@uurtech/jdf-cli 0.1.26 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -1,13 +1,20 @@
1
1
  #!/usr/bin/env node
2
- import fs from 'fs';
3
- import path2 from 'path';
2
+ import fs7 from 'fs';
3
+ import path7 from 'path';
4
4
  import { fileURLToPath } from 'url';
5
5
  import Ajv from 'ajv';
6
6
  import addFormats from 'ajv-formats';
7
7
  import JSZip from 'jszip';
8
8
  import crypto, { createHash } from 'crypto';
9
9
  import { readFile } from 'fs/promises';
10
- import { execFileSync } from 'child_process';
10
+ import { execFileSync, spawnSync } from 'child_process';
11
+
12
+ var __require = /* @__PURE__ */ ((x) => typeof require !== "undefined" ? require : typeof Proxy !== "undefined" ? new Proxy(x, {
13
+ get: (a, b) => (typeof require !== "undefined" ? require : a)[b]
14
+ }) : x)(function(x) {
15
+ if (typeof require !== "undefined") return require.apply(this, arguments);
16
+ throw Error('Dynamic require of "' + x + '" is not supported');
17
+ });
11
18
 
12
19
  // ../../packages/jdf-core/src/manifest.ts
13
20
  var JDFX_MANIFEST_VERSION = "1.0.0";
@@ -16,17 +23,17 @@ var JDFX_MANIFEST_PATH = "manifest.json";
16
23
  var JDFX_ASSET_DIR = "assets";
17
24
 
18
25
  // src/commands/validate.ts
19
- var __dirname$1 = path2.dirname(fileURLToPath(import.meta.url));
26
+ var __dirname$1 = path7.dirname(fileURLToPath(import.meta.url));
20
27
  function resolveSchemaPath() {
21
- const bundled = path2.resolve(__dirname$1, "jdf-schema.json");
22
- if (fs.existsSync(bundled)) return bundled;
23
- const dev = path2.resolve(__dirname$1, "../../../../spec/jdf-schema.json");
28
+ const bundled = path7.resolve(__dirname$1, "jdf-schema.json");
29
+ if (fs7.existsSync(bundled)) return bundled;
30
+ const dev = path7.resolve(__dirname$1, "../../../../spec/jdf-schema.json");
24
31
  return dev;
25
32
  }
26
33
  var SCHEMA_PATH = resolveSchemaPath();
27
34
  async function loadDocument(filePath) {
28
35
  if (filePath.toLowerCase().endsWith(".jdfx")) {
29
- const zip = await JSZip.loadAsync(fs.readFileSync(filePath));
36
+ const zip = await JSZip.loadAsync(fs7.readFileSync(filePath));
30
37
  const docFile = zip.file(JDFX_DOCUMENT_PATH);
31
38
  if (!docFile) {
32
39
  console.error(`\u2717 Bundle missing ${JDFX_DOCUMENT_PATH}`);
@@ -51,11 +58,11 @@ async function loadDocument(filePath) {
51
58
  }
52
59
  return { doc, bundle: { manifest, assetCount } };
53
60
  }
54
- return { doc: JSON.parse(fs.readFileSync(filePath, "utf-8")) };
61
+ return { doc: JSON.parse(fs7.readFileSync(filePath, "utf-8")) };
55
62
  }
56
63
  async function validate(file) {
57
- const filePath = path2.resolve(file);
58
- if (!fs.existsSync(filePath)) {
64
+ const filePath = path7.resolve(file);
65
+ if (!fs7.existsSync(filePath)) {
59
66
  console.error(`File not found: ${filePath}`);
60
67
  return false;
61
68
  }
@@ -68,11 +75,11 @@ async function validate(file) {
68
75
  }
69
76
  if (!loaded) return false;
70
77
  const { doc, bundle } = loaded;
71
- if (!fs.existsSync(SCHEMA_PATH)) {
78
+ if (!fs7.existsSync(SCHEMA_PATH)) {
72
79
  console.error(`Schema not found at ${SCHEMA_PATH}`);
73
80
  return false;
74
81
  }
75
- const schema = JSON.parse(fs.readFileSync(SCHEMA_PATH, "utf-8"));
82
+ const schema = JSON.parse(fs7.readFileSync(SCHEMA_PATH, "utf-8"));
76
83
  const ajv = new Ajv({ allErrors: true, strict: false });
77
84
  addFormats(ajv);
78
85
  const validateFn = ajv.compile(schema);
@@ -81,7 +88,7 @@ async function validate(file) {
81
88
  const d = doc;
82
89
  const pageCount = Array.isArray(d.pages) ? d.pages.length : 0;
83
90
  const elCount = Array.isArray(d.pages) ? d.pages.reduce((acc, p) => acc + (Array.isArray(p?.elements) ? p.elements.length : 0), 0) : 0;
84
- console.log(`\u2713 Valid: ${path2.basename(filePath)}`);
91
+ console.log(`\u2713 Valid: ${path7.basename(filePath)}`);
85
92
  console.log(` Format: ${d.$jdf}${bundle ? " (jdfx bundle)" : ""}`);
86
93
  console.log(` Title: ${d.meta?.title}`);
87
94
  console.log(` Pages: ${pageCount}`);
@@ -94,7 +101,7 @@ async function validate(file) {
94
101
  }
95
102
  return true;
96
103
  }
97
- console.error(`\u2717 Invalid: ${path2.basename(filePath)}`);
104
+ console.error(`\u2717 Invalid: ${path7.basename(filePath)}`);
98
105
  for (const err of validateFn.errors || []) {
99
106
  const loc = err.instancePath || "(root)";
100
107
  console.error(` ${loc} \u2014 ${err.message}`);
@@ -109,15 +116,15 @@ function hashBytes(bytes) {
109
116
  return createHash("sha1").update(bytes).digest("hex").slice(0, 16);
110
117
  }
111
118
  function rewriteResourceRefs(doc, oldKey, newKey) {
112
- function walk(els) {
119
+ function walk2(els) {
113
120
  if (!els) return;
114
121
  for (const el of els) {
115
122
  if (el?.resource === oldKey) el.resource = newKey;
116
- if (el?.elements) walk(el.elements);
117
- if (el?.children) walk(el.children);
123
+ if (el?.elements) walk2(el.elements);
124
+ if (el?.children) walk2(el.children);
118
125
  }
119
126
  }
120
- for (const page of doc.pages || []) walk(page.elements);
127
+ for (const page of doc.pages || []) walk2(page.elements);
121
128
  }
122
129
  function extractAssets(doc) {
123
130
  const assets = [];
@@ -149,10 +156,10 @@ function extractAssets(doc) {
149
156
  assets.push({ id, bytes, mimeType, ext });
150
157
  return id;
151
158
  }
152
- function walk(els) {
159
+ function walk2(els) {
153
160
  if (!els) return;
154
161
  for (const el of els) {
155
- if (el?.type === "image" && typeof el.src === "string" && el.src.startsWith("data:")) {
162
+ if ((el?.type === "image" || el?.type === "video") && typeof el.src === "string" && el.src.startsWith("data:")) {
156
163
  const m = el.src.match(/^data:([^;]+);base64,(.*)$/);
157
164
  if (m) {
158
165
  const mimeType = m[1];
@@ -162,19 +169,21 @@ function extractAssets(doc) {
162
169
  el.resource = id;
163
170
  }
164
171
  }
165
- if (el?.elements) walk(el.elements);
166
- if (el?.children) walk(el.children);
172
+ if (el?.elements) walk2(el.elements);
173
+ if (el?.children) walk2(el.children);
167
174
  }
168
175
  }
169
- for (const page of cloned.pages || []) walk(page.elements);
170
- if (cloned.resources?.images) {
171
- for (const [key, res] of Object.entries(cloned.resources.images)) {
176
+ for (const page of cloned.pages || []) walk2(page.elements);
177
+ for (const bucket of ["images", "videos"]) {
178
+ const store = cloned.resources?.[bucket];
179
+ if (!store) continue;
180
+ for (const [key, res] of Object.entries(store)) {
172
181
  if (!res || typeof res !== "object" || !("data" in res) || !res.data) continue;
173
182
  const data = String(res.data);
174
183
  const m = data.match(/^data:([^;]+);base64,(.*)$/);
175
184
  const b64 = m ? m[2] : data;
176
- const mimeType = m ? m[1] : res.mimeType || "image/png";
177
- const ext = mimeType.split("/")[1]?.replace("jpeg", "jpg") || "bin";
185
+ const mimeType = m ? m[1] : res.mimeType || (bucket === "videos" ? "video/mp4" : "image/png");
186
+ const ext = mimeType.split("/")[1]?.replace("jpeg", "jpg").replace("quicktime", "mov") || "bin";
178
187
  const bytes = decodeBase64(b64);
179
188
  const h = hashBytes(bytes);
180
189
  let canonicalId = hashToId.get(h);
@@ -188,10 +197,10 @@ function extractAssets(doc) {
188
197
  delete updated.data;
189
198
  updated.src = "embedded";
190
199
  if (canonicalId !== key) {
191
- delete cloned.resources.images[key];
200
+ delete store[key];
192
201
  rewriteResourceRefs(cloned, key, canonicalId);
193
202
  } else {
194
- cloned.resources.images[key] = updated;
203
+ store[key] = updated;
195
204
  }
196
205
  }
197
206
  }
@@ -221,21 +230,22 @@ async function packJdfx(doc) {
221
230
  return { bytes, manifest };
222
231
  }
223
232
  function shouldUseJdfx(doc) {
224
- const images = doc.resources?.images ?? {};
225
- for (const v of Object.values(images)) {
226
- if (v && typeof v === "object" && "data" in v && v.data) return true;
233
+ for (const store of [doc.resources?.images ?? {}, doc.resources?.videos ?? {}]) {
234
+ for (const v of Object.values(store)) {
235
+ if (v && typeof v === "object" && "data" in v && v.data) return true;
236
+ }
227
237
  }
228
- function walk(els) {
238
+ function walk2(els) {
229
239
  if (!els) return false;
230
240
  for (const el of els) {
231
- if (el?.type === "image" && typeof el.src === "string" && el.src.startsWith("data:")) return true;
232
- if (el?.elements && walk(el.elements)) return true;
233
- if (el?.children && walk(el.children)) return true;
241
+ if ((el?.type === "image" || el?.type === "video") && typeof el.src === "string" && el.src.startsWith("data:")) return true;
242
+ if (el?.elements && walk2(el.elements)) return true;
243
+ if (el?.children && walk2(el.children)) return true;
234
244
  }
235
245
  return false;
236
246
  }
237
247
  for (const page of doc.pages || []) {
238
- if (walk(page.elements)) return true;
248
+ if (walk2(page.elements)) return true;
239
249
  }
240
250
  return false;
241
251
  }
@@ -294,13 +304,13 @@ function stripInline(text) {
294
304
  return parseInline(text).map((r) => r.text).join("");
295
305
  }
296
306
  async function importMarkdown(inputPath, outputPath) {
297
- const input = path2.resolve(inputPath);
307
+ const input = path7.resolve(inputPath);
298
308
  console.log(`Importing: ${input}`);
299
- const content = fs.readFileSync(input, "utf-8");
300
- const doc = convertMarkdownToJdf(content, path2.basename(input, path2.extname(input)), path2.dirname(input));
309
+ const content = fs7.readFileSync(input, "utf-8");
310
+ const doc = convertMarkdownToJdf(content, path7.basename(input, path7.extname(input)), path7.dirname(input));
301
311
  let output;
302
312
  if (outputPath) {
303
- output = path2.resolve(outputPath);
313
+ output = path7.resolve(outputPath);
304
314
  } else {
305
315
  const stem = input.replace(/\.(md|markdown)$/i, "");
306
316
  output = stem + (shouldUseJdfx(doc) ? ".jdfx" : ".jdf");
@@ -308,11 +318,11 @@ async function importMarkdown(inputPath, outputPath) {
308
318
  console.log(`Output: ${output}`);
309
319
  if (output.toLowerCase().endsWith(".jdfx")) {
310
320
  const { bytes, manifest } = await packJdfx(doc);
311
- fs.writeFileSync(output, bytes);
321
+ fs7.writeFileSync(output, bytes);
312
322
  console.log(`
313
323
  Done! Created ${doc.pages.length} page(s), ${manifest.assets.length} asset(s) bundled`);
314
324
  } else {
315
- fs.writeFileSync(output, JSON.stringify(doc, null, 2));
325
+ fs7.writeFileSync(output, JSON.stringify(doc, null, 2));
316
326
  console.log(`
317
327
  Done! Created ${doc.pages.length} page(s)`);
318
328
  }
@@ -329,10 +339,10 @@ var MIME_BY_EXT2 = {
329
339
  };
330
340
  function resolveImageSrc(src, baseDir) {
331
341
  if (/^(https?:|data:|file:)/i.test(src)) return src;
332
- const abs = path2.isAbsolute(src) ? src : path2.resolve(baseDir, src);
342
+ const abs = path7.isAbsolute(src) ? src : path7.resolve(baseDir, src);
333
343
  try {
334
- const bytes = fs.readFileSync(abs);
335
- const ext = path2.extname(abs).slice(1).toLowerCase();
344
+ const bytes = fs7.readFileSync(abs);
345
+ const ext = path7.extname(abs).slice(1).toLowerCase();
336
346
  const mime = MIME_BY_EXT2[ext] || "application/octet-stream";
337
347
  return `data:${mime};base64,${bytes.toString("base64")}`;
338
348
  } catch {
@@ -613,8 +623,232 @@ function convertMarkdownToJdf(md, title, baseDir = process.cwd()) {
613
623
  };
614
624
  }
615
625
 
616
- // ../../packages/jdf-pdf-import/src/core.ts
626
+ // ../../packages/jdf-pdf-import/src/tables.ts
617
627
  var PT_TO_MM = 0.352778;
628
+ function calibrateGlyphWidth(runs) {
629
+ const ks = [];
630
+ for (const r of runs) {
631
+ const t = r.text;
632
+ if (/\s$/.test(t) || t.trim().length < 3 || r.width <= 0) continue;
633
+ ks.push(r.width / (t.length * r.fontSize * PT_TO_MM));
634
+ }
635
+ if (ks.length < 3) return 0.55;
636
+ ks.sort((a, b) => a - b);
637
+ return Math.min(0.7, Math.max(0.4, ks[Math.floor(ks.length / 2)]));
638
+ }
639
+ function hasStretchedSpaces(runs, k) {
640
+ let n = 0, wide = 0;
641
+ for (const r of runs) {
642
+ if (!/\s$/.test(r.text) || r.text.trim().length === 0) continue;
643
+ const em = r.fontSize * PT_TO_MM;
644
+ const est = r.text.trim().length * em * k + em * 0.25;
645
+ n++;
646
+ if (r.width > est * 1.4) wide++;
647
+ }
648
+ return n >= 4 && wide / n >= 0.3;
649
+ }
650
+ function textExtent(r, k, stretched) {
651
+ const em = r.fontSize * PT_TO_MM;
652
+ if (!/\s$/.test(r.text)) return Math.max(em * 0.5, r.width);
653
+ const chars = Math.max(1, r.text.trim().length);
654
+ const est = chars * em * k + em * 0.25;
655
+ if (stretched) return Math.max(em * 0.5, Math.min(r.width, est));
656
+ return Math.max(em * 0.5, r.width > est * 1.4 ? est : r.width);
657
+ }
658
+ function groupRows(runs, skip) {
659
+ const k = calibrateGlyphWidth(runs);
660
+ const stretched = hasStretchedSpaces(runs, k);
661
+ const idx = runs.map((_, i) => i).filter((i) => !skip(runs[i]) && runs[i].text.trim().length > 0);
662
+ idx.sort((a, b) => runs[a].y - runs[b].y || runs[a].x - runs[b].x);
663
+ const rows = [];
664
+ for (const i of idx) {
665
+ const r = runs[i];
666
+ const tol = Math.max(0.8, r.fontSize * PT_TO_MM * 0.35);
667
+ const last = rows[rows.length - 1];
668
+ const cell = { run: r, idx: i, x0: r.x, x1: r.x + textExtent(r, k, stretched) };
669
+ if (last && Math.abs(last.y - r.y) <= tol) {
670
+ last.cells.push(cell);
671
+ last.h = Math.max(last.h, r.height);
672
+ } else {
673
+ rows.push({ y: r.y, h: r.height, cells: [cell] });
674
+ }
675
+ }
676
+ for (const row of rows) {
677
+ row.cells.sort((a, b) => a.x0 - b.x0);
678
+ const merged = [];
679
+ for (const c of row.cells) {
680
+ const last = merged[merged.length - 1];
681
+ const em = c.run.fontSize * PT_TO_MM;
682
+ if (last && c.x0 - last.x1 <= em * 1) {
683
+ last.x1 = Math.max(last.x1, c.x1);
684
+ last.run = { ...last.run, text: `${last.run.text.replace(/\s+$/, "")} ${c.run.text.replace(/^\s+/, "")}`, width: last.x1 - last.x0 };
685
+ last.extra = [...last.extra ?? [], c.idx];
686
+ } else merged.push({ ...c });
687
+ }
688
+ row.cells = merged;
689
+ }
690
+ return rows;
691
+ }
692
+ function columnBands(rows) {
693
+ const cells = rows.flatMap((r) => r.cells);
694
+ const sorted = cells.slice().sort((a, b) => a.x0 - b.x0);
695
+ const bands = [];
696
+ for (const c of sorted) {
697
+ const last = bands[bands.length - 1];
698
+ if (last && c.x0 <= last.x1 - 0.2) {
699
+ last.x1 = Math.max(last.x1, c.x1);
700
+ last.members.push(c);
701
+ } else bands.push({ x0: c.x0, x1: c.x1, members: [c] });
702
+ }
703
+ for (const b of bands) {
704
+ const rowsSeen = /* @__PURE__ */ new Set();
705
+ for (const m of b.members) {
706
+ const row = rows.find((r) => r.cells.includes(m));
707
+ if (rowsSeen.has(row)) return null;
708
+ rowsSeen.add(row);
709
+ }
710
+ }
711
+ return bands.map(({ x0, x1 }) => ({ x0, x1 }));
712
+ }
713
+ var numeric = (s) => /^[\s$€£¥+\-−–]*[\d.,]+\s*(%|ms|s|k|m|b|M|K|B|x|×)?\s*(\/\w+)?$/i.test(s.trim()) || /^[+\-−]?\d/.test(s.trim()) && /\d$/.test(s.trim().replace(/[%)]$/, ""));
714
+ function detectTables(runs, shapes, pageWidthMm) {
715
+ const out = [];
716
+ const used = /* @__PURE__ */ new Set();
717
+ const rows = groupRows(runs, () => false);
718
+ const pageW = pageWidthMm;
719
+ let i = 0;
720
+ while (i < rows.length) {
721
+ if (rows[i].cells.length < 2) {
722
+ i++;
723
+ continue;
724
+ }
725
+ let j = i;
726
+ let cur = columnBands([rows[i]]);
727
+ let best = null;
728
+ while (j + 1 < rows.length && cur) {
729
+ const next = rows[j + 1];
730
+ const gap = next.y - (rows[j].y + rows[j].h);
731
+ const rowH = Math.max(rows[j].h, next.h);
732
+ if (gap > rowH * 2.2) break;
733
+ if (next.cells.length === 1) {
734
+ const c = next.cells[0];
735
+ const inBand = cur.findIndex((b) => c.x0 < b.x1 - 0.2 && c.x1 > b.x0 + 0.2);
736
+ const spansSeveral = cur.filter((b) => c.x0 < b.x1 - 0.2 && c.x1 > b.x0 + 0.2).length > 1;
737
+ if (inBand <= 0 || spansSeveral) break;
738
+ j++;
739
+ continue;
740
+ }
741
+ const nb = columnBands(rows.slice(i, j + 2));
742
+ if (!nb || nb.length < 2) break;
743
+ if (nb.length > cur.length && j - i >= 2) break;
744
+ cur = nb;
745
+ j++;
746
+ if (cur.length >= 2) best = { j, bands: cur };
747
+ }
748
+ const multiRows = best ? rows.slice(i, best.j + 1).filter((r) => r.cells.length >= 2).length : 0;
749
+ const blockRows = best ? rows.slice(i, best.j + 1) : [];
750
+ const bbox = blockRows.length ? {
751
+ x0: Math.min(...best.bands.map((b) => b.x0)) - 5,
752
+ x1: Math.max(...best.bands.map((b) => b.x1)) + 5,
753
+ // Cell padding puts backgrounds/borders well above the first baseline and below the last.
754
+ y0: blockRows[0].y - blockRows[0].h * 2.5,
755
+ y1: blockRows[blockRows.length - 1].y + blockRows[blockRows.length - 1].h * 3
756
+ } : null;
757
+ const gridShapes = bbox ? shapes.map((s, k) => ({ s, k })).filter(({ s }) => s.x >= bbox.x0 - 1 && s.x + s.width <= bbox.x1 + 1 && s.y >= bbox.y0 - 1 && s.y + s.height <= bbox.y1 + 1 && (s.kind === "line" || s.kind === "rect")) : [];
758
+ const hasLattice = gridShapes.length >= 3;
759
+ if (!best || multiRows < (hasLattice ? 2 : 3) || best.bands.length < 2) {
760
+ i++;
761
+ continue;
762
+ }
763
+ const bands = best.bands;
764
+ const cellText = (row, b) => row.cells.filter((c) => c.x0 < bands[b].x1 - 0.2 && c.x1 > bands[b].x0 + 0.2).map((c) => c.run.text.trim()).join(" ").trim();
765
+ const grid = [];
766
+ const lineIdx = [];
767
+ for (const row of blockRows) {
768
+ if (row.cells.length === 1 && grid.length) {
769
+ const c = row.cells[0];
770
+ const b = bands.findIndex((bb) => c.x0 < bb.x1 - 0.2 && c.x1 > bb.x0 + 0.2);
771
+ if (b > 0) {
772
+ grid[grid.length - 1][b] = (grid[grid.length - 1][b] + " " + c.run.text.trim()).trim();
773
+ lineIdx.push(c.idx, ...c.extra ?? []);
774
+ continue;
775
+ }
776
+ }
777
+ grid.push(bands.map((_, b) => cellText(row, b)));
778
+ for (const c of row.cells) {
779
+ lineIdx.push(c.idx);
780
+ for (const k of c.extra ?? []) lineIdx.push(k);
781
+ }
782
+ }
783
+ if (lineIdx.some((k) => used.has(k))) {
784
+ i = best.j + 1;
785
+ continue;
786
+ }
787
+ const first = blockRows[0];
788
+ const tableW = bbox.x1 - bbox.x0;
789
+ const rowFill = (row) => {
790
+ const cy = row.y + row.h * 0.5;
791
+ const rects = gridShapes.filter(({ s }) => s.kind === "rect" && s.fill && s.fill.toLowerCase() !== "#ffffff" && s.height >= row.h * 0.6 && s.height < row.h * 4.5 && s.y <= cy && s.y + s.height >= cy);
792
+ const covered = rects.reduce((a, { s }) => a + s.width, 0);
793
+ if (!rects.length || covered < tableW * 0.5) return null;
794
+ const counts = /* @__PURE__ */ new Map();
795
+ for (const { s } of rects) counts.set(s.fill, (counts.get(s.fill) ?? 0) + s.width);
796
+ const fill = [...counts.entries()].sort((a, b) => b[1] - a[1])[0][0];
797
+ return { fill, rects };
798
+ };
799
+ const headerBg = rowFill(first);
800
+ const firstBold = first.cells.every((c) => c.run.bold);
801
+ const bodyFills = blockRows.slice(1).map(rowFill);
802
+ const headerDistinct = !!headerBg && !bodyFills.every((f) => f?.fill === headerBg.fill);
803
+ const isHeader = headerDistinct || firstBold && !blockRows.slice(1).every((r) => r.cells.every((c) => c.run.bold));
804
+ const columns = bands.map((b, k) => {
805
+ const vals = grid.slice(isHeader ? 1 : 0).map((r) => r[k]).filter(Boolean);
806
+ const numericShare = vals.length ? vals.filter(numeric).length / vals.length : 0;
807
+ const col = { width: Math.round((b.x1 - b.x0) * 10) / 10 };
808
+ if (numericShare >= 0.7) col.align = "right";
809
+ return col;
810
+ });
811
+ const x0 = Math.max(0, bands[0].x0 - 2.5);
812
+ const x1 = Math.min(pageW, bands[bands.length - 1].x1 + 2.5);
813
+ for (let k = 0; k < bands.length; k++) {
814
+ const left = k === 0 ? x0 : (bands[k - 1].x1 + bands[k].x0) / 2;
815
+ const right = k === bands.length - 1 ? x1 : (bands[k].x1 + bands[k + 1].x0) / 2;
816
+ columns[k].width = Math.round((right - left) * 10) / 10;
817
+ }
818
+ const rowFills = (isHeader ? bodyFills : [headerBg, ...bodyFills]).map((f) => f?.fill ?? null);
819
+ const odd = rowFills.filter((_, k) => k % 2 === 1), even = rowFills.filter((_, k) => k % 2 === 0);
820
+ const altColor = odd.length && odd[0] && odd.every((f) => f === odd[0]) && even.every((f) => f !== odd[0]) ? odd[0] : void 0;
821
+ const borderShape = gridShapes.find(({ s }) => s.kind === "line" || s.kind === "rect" && (s.height < 0.6 || s.width < 0.6) && (s.fill || s.stroke));
822
+ const borderColor = borderShape ? borderShape.s.stroke || borderShape.s.fill : void 0;
823
+ const fontSize = Math.round(first.cells[0].run.fontSize * 10) / 10;
824
+ const y0 = headerBg ? Math.min(...headerBg.rects.map(({ s }) => s.y)) : first.y - first.h * 0.5;
825
+ const element = {
826
+ type: "table",
827
+ position: { x: Math.round(x0 * 100) / 100, y: Math.round(Math.max(0, y0) * 100) / 100 },
828
+ width: Math.round((x1 - x0) * 100) / 100,
829
+ columns,
830
+ rows: isHeader ? grid.slice(1) : grid,
831
+ style: { fontSize }
832
+ };
833
+ if (isHeader) {
834
+ element.headers = grid[0];
835
+ const hs = { fontWeight: "bold" };
836
+ if (headerBg) hs.backgroundColor = headerBg.fill;
837
+ const hc = first.cells[0].run.color;
838
+ if (hc && hc !== "#000000") hs.color = hc;
839
+ element.headerStyle = hs;
840
+ }
841
+ if (altColor) element.alternatingRowColor = altColor;
842
+ element.borders = borderColor ? { outer: true, inner: true, color: borderColor, width: 1 } : false;
843
+ for (const k of lineIdx) used.add(k);
844
+ out.push({ element, lineIdx, shapeIdx: gridShapes.map(({ k }) => k) });
845
+ i = best.j + 1;
846
+ }
847
+ return out;
848
+ }
849
+
850
+ // ../../packages/jdf-pdf-import/src/core.ts
851
+ var PT_TO_MM2 = 0.352778;
618
852
  function classifyFont(name) {
619
853
  const n = (name || "").toLowerCase();
620
854
  const bold = /bold|black|heavy|semibold|demibold|extrabold/.test(n);
@@ -724,19 +958,36 @@ async function walkOps(page, OPS, viewport) {
724
958
  const minY = Math.min(...ys), maxY = Math.max(...ys);
725
959
  imagePositions.push({
726
960
  name,
727
- x: minX * PT_TO_MM,
728
- y: minY * PT_TO_MM,
729
- w: (maxX - minX) * PT_TO_MM,
730
- h: (maxY - minY) * PT_TO_MM,
961
+ x: minX * PT_TO_MM2,
962
+ y: minY * PT_TO_MM2,
963
+ w: (maxX - minX) * PT_TO_MM2,
964
+ h: (maxY - minY) * PT_TO_MM2,
731
965
  inline,
732
966
  maskFill
733
967
  });
734
968
  }
735
969
  let pathSegments = [];
736
970
  let pathRect = null;
971
+ let pathRects = [];
737
972
  let pathStart = null;
738
973
  let pathLast = null;
739
974
  function flushPath(isFill, isStroke) {
975
+ for (const r of pathRects) {
976
+ const tl = toViewport(r.x, r.y + r.h);
977
+ const br = toViewport(r.x + r.w, r.y);
978
+ shapes.push({
979
+ kind: "rect",
980
+ x: Math.min(tl.x, br.x) * PT_TO_MM2,
981
+ y: Math.min(tl.y, br.y) * PT_TO_MM2,
982
+ width: Math.abs(br.x - tl.x) * PT_TO_MM2,
983
+ height: Math.abs(br.y - tl.y) * PT_TO_MM2,
984
+ fill: isFill ? gs.fill : void 0,
985
+ stroke: isStroke ? gs.stroke : void 0,
986
+ strokeWidth: isStroke ? gs.lineWidth * PT_TO_MM2 : void 0,
987
+ opacity: isFill ? gs.fillAlpha : gs.strokeAlpha
988
+ });
989
+ }
990
+ pathRects = [];
740
991
  if (pathRect) {
741
992
  const tl = toViewport(pathRect.x, pathRect.y + pathRect.h);
742
993
  const br = toViewport(pathRect.x + pathRect.w, pathRect.y);
@@ -746,13 +997,13 @@ async function walkOps(page, OPS, viewport) {
746
997
  const h = Math.abs(br.y - tl.y);
747
998
  shapes.push({
748
999
  kind: "rect",
749
- x: x * PT_TO_MM,
750
- y: y * PT_TO_MM,
751
- width: w * PT_TO_MM,
752
- height: h * PT_TO_MM,
1000
+ x: x * PT_TO_MM2,
1001
+ y: y * PT_TO_MM2,
1002
+ width: w * PT_TO_MM2,
1003
+ height: h * PT_TO_MM2,
753
1004
  fill: isFill ? gs.fill : void 0,
754
1005
  stroke: isStroke ? gs.stroke : void 0,
755
- strokeWidth: isStroke ? gs.lineWidth * PT_TO_MM : void 0,
1006
+ strokeWidth: isStroke ? gs.lineWidth * PT_TO_MM2 : void 0,
756
1007
  opacity: isFill ? gs.fillAlpha : gs.strokeAlpha
757
1008
  });
758
1009
  } else if (pathSegments.length === 2 && pathSegments[0].type === "M" && pathSegments[1].type === "L") {
@@ -764,35 +1015,35 @@ async function walkOps(page, OPS, viewport) {
764
1015
  const minY = Math.min(va.y, vb.y);
765
1016
  const maxX = Math.max(va.x, vb.x);
766
1017
  const maxY = Math.max(va.y, vb.y);
767
- const x1Local = (va.x - minX) * PT_TO_MM;
768
- const y1Local = (va.y - minY) * PT_TO_MM;
769
- const x2Local = (vb.x - minX) * PT_TO_MM;
770
- const y2Local = (vb.y - minY) * PT_TO_MM;
771
- const wLocal = Math.max(0.05, (maxX - minX) * PT_TO_MM);
772
- const hLocal = Math.max(0.05, (maxY - minY) * PT_TO_MM);
1018
+ const x1Local = (va.x - minX) * PT_TO_MM2;
1019
+ const y1Local = (va.y - minY) * PT_TO_MM2;
1020
+ const x2Local = (vb.x - minX) * PT_TO_MM2;
1021
+ const y2Local = (vb.y - minY) * PT_TO_MM2;
1022
+ const wLocal = Math.max(0.05, (maxX - minX) * PT_TO_MM2);
1023
+ const hLocal = Math.max(0.05, (maxY - minY) * PT_TO_MM2);
773
1024
  const dx = Math.abs(va.x - vb.x);
774
1025
  const dy = Math.abs(va.y - vb.y);
775
1026
  const axisAligned = dx < 0.5 || dy < 0.5;
776
1027
  if (axisAligned) {
777
1028
  shapes.push({
778
1029
  kind: "line",
779
- x: minX * PT_TO_MM,
780
- y: minY * PT_TO_MM,
1030
+ x: minX * PT_TO_MM2,
1031
+ y: minY * PT_TO_MM2,
781
1032
  width: wLocal,
782
1033
  height: hLocal,
783
1034
  stroke: isStroke ? gs.stroke : void 0,
784
- strokeWidth: isStroke ? gs.lineWidth * PT_TO_MM : void 0,
1035
+ strokeWidth: isStroke ? gs.lineWidth * PT_TO_MM2 : void 0,
785
1036
  opacity: gs.strokeAlpha
786
1037
  });
787
1038
  } else {
788
1039
  shapes.push({
789
1040
  kind: "path",
790
- x: minX * PT_TO_MM,
791
- y: minY * PT_TO_MM,
1041
+ x: minX * PT_TO_MM2,
1042
+ y: minY * PT_TO_MM2,
792
1043
  width: wLocal,
793
1044
  height: hLocal,
794
1045
  stroke: isStroke ? gs.stroke : void 0,
795
- strokeWidth: isStroke ? gs.lineWidth * PT_TO_MM : void 0,
1046
+ strokeWidth: isStroke ? gs.lineWidth * PT_TO_MM2 : void 0,
796
1047
  opacity: gs.strokeAlpha,
797
1048
  path: `M ${x1Local.toFixed(2)} ${y1Local.toFixed(2)} L ${x2Local.toFixed(2)} ${y2Local.toFixed(2)}`
798
1049
  });
@@ -825,20 +1076,20 @@ async function walkOps(page, OPS, viewport) {
825
1076
  if (seg.type === "Z") return "Z";
826
1077
  const p = [];
827
1078
  for (let i = 0; i < seg.pts.length; i += 2) {
828
- p.push(((seg.pts[i] - minX) * PT_TO_MM).toFixed(2));
829
- p.push(((seg.pts[i + 1] - minY) * PT_TO_MM).toFixed(2));
1079
+ p.push(((seg.pts[i] - minX) * PT_TO_MM2).toFixed(2));
1080
+ p.push(((seg.pts[i + 1] - minY) * PT_TO_MM2).toFixed(2));
830
1081
  }
831
1082
  return `${seg.type} ${p.join(" ")}`;
832
1083
  }).join(" ");
833
1084
  shapes.push({
834
1085
  kind: "path",
835
- x: minX * PT_TO_MM,
836
- y: minY * PT_TO_MM,
837
- width: bw * PT_TO_MM,
838
- height: bh * PT_TO_MM,
1086
+ x: minX * PT_TO_MM2,
1087
+ y: minY * PT_TO_MM2,
1088
+ width: bw * PT_TO_MM2,
1089
+ height: bh * PT_TO_MM2,
839
1090
  fill: isFill ? gs.fill : void 0,
840
1091
  stroke: isStroke ? gs.stroke : void 0,
841
- strokeWidth: isStroke ? gs.lineWidth * PT_TO_MM : void 0,
1092
+ strokeWidth: isStroke ? gs.lineWidth * PT_TO_MM2 : void 0,
842
1093
  opacity: isFill ? gs.fillAlpha : gs.strokeAlpha,
843
1094
  path: d
844
1095
  });
@@ -1000,6 +1251,12 @@ async function walkOps(page, OPS, viewport) {
1000
1251
  } else if (op === OPS.closePath) {
1001
1252
  pathSegments.push({ type: "Z", pts: [] });
1002
1253
  if (pathStart) pathLast = { ...pathStart };
1254
+ } else if (op === OPS.rectangle) {
1255
+ const x = pathArgs[ai], y = pathArgs[ai + 1], w = pathArgs[ai + 2], h = pathArgs[ai + 3];
1256
+ ai += 4;
1257
+ const p1 = tx(gs.ctm, x, y);
1258
+ const p3 = tx(gs.ctm, x + w, y + h);
1259
+ pathRects.push({ x: Math.min(p1.x, p3.x), y: Math.min(p1.y, p3.y), w: Math.abs(p3.x - p1.x), h: Math.abs(p3.y - p1.y) });
1003
1260
  }
1004
1261
  }
1005
1262
  } else if (fn === OPS.fill || fn === OPS.stroke || fn === OPS.fillStroke || fn === OPS.eoFill || fn === OPS.eoFillStroke || fn === OPS.closeFillStroke || fn === OPS.closeStroke || fn === OPS.closeEOFillStroke) {
@@ -1008,6 +1265,7 @@ async function walkOps(page, OPS, viewport) {
1008
1265
  flushPath(isFill, isStroke);
1009
1266
  } else if (fn === OPS.endPath || fn === OPS.clip || fn === OPS.eoClip) {
1010
1267
  pathSegments = [];
1268
+ pathRects = [];
1011
1269
  pathRect = null;
1012
1270
  pathStart = null;
1013
1271
  pathLast = null;
@@ -1172,10 +1430,10 @@ async function extractLinks(doc, page, viewport) {
1172
1430
  const xMax = Math.max(c1.x, c2.x);
1173
1431
  const yMax = Math.max(c1.y, c2.y);
1174
1432
  const rectMm = {
1175
- x: xMin * PT_TO_MM,
1176
- y: yMin * PT_TO_MM,
1177
- w: (xMax - xMin) * PT_TO_MM,
1178
- h: (yMax - yMin) * PT_TO_MM
1433
+ x: xMin * PT_TO_MM2,
1434
+ y: yMin * PT_TO_MM2,
1435
+ w: (xMax - xMin) * PT_TO_MM2,
1436
+ h: (yMax - yMin) * PT_TO_MM2
1179
1437
  };
1180
1438
  const url = a.url || a.unsafeUrl;
1181
1439
  const destPage = url ? void 0 : await resolveDestPage(doc, a.dest);
@@ -1213,10 +1471,10 @@ async function extractFormWidgets(page, viewport) {
1213
1471
  })).filter((o) => o.value !== "") : [];
1214
1472
  out.push({
1215
1473
  rectMm: {
1216
- x: xMin * PT_TO_MM,
1217
- y: yMin * PT_TO_MM,
1218
- w: (xMax - xMin) * PT_TO_MM,
1219
- h: (yMax - yMin) * PT_TO_MM
1474
+ x: xMin * PT_TO_MM2,
1475
+ y: yMin * PT_TO_MM2,
1476
+ w: (xMax - xMin) * PT_TO_MM2,
1477
+ h: (yMax - yMin) * PT_TO_MM2
1220
1478
  },
1221
1479
  fieldType: a.fieldType || "",
1222
1480
  fieldName: a.fieldName || `field-${out.length + 1}`,
@@ -1236,16 +1494,16 @@ async function extractFormWidgets(page, viewport) {
1236
1494
  async function flattenOutline(doc, outline) {
1237
1495
  if (!outline) return [];
1238
1496
  const out = [];
1239
- async function walk(items, depth) {
1497
+ async function walk2(items, depth) {
1240
1498
  for (const item of items) {
1241
1499
  const idx = await resolveDestPage(doc, item.dest);
1242
1500
  if (idx != null && typeof item.title === "string" && item.title.trim()) {
1243
1501
  out.push({ title: item.title.trim(), pageIndex: idx, depth });
1244
1502
  }
1245
- if (item.items?.length) await walk(item.items, depth + 1);
1503
+ if (item.items?.length) await walk2(item.items, depth + 1);
1246
1504
  }
1247
1505
  }
1248
- await walk(outline, 1);
1506
+ await walk2(outline, 1);
1249
1507
  return out;
1250
1508
  }
1251
1509
  var normTitle = (s) => s.toLowerCase().replace(/[\s\u00a0]+/g, " ").replace(/[^\p{L}\p{N} ]/gu, "").trim();
@@ -1498,12 +1756,12 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
1498
1756
  const w = safeNum(it.width, 0);
1499
1757
  runs.push({
1500
1758
  text: it.str,
1501
- x: safeNum(vx * PT_TO_MM, 0),
1502
- y: safeNum(yTop * PT_TO_MM, 0),
1759
+ x: safeNum(vx * PT_TO_MM2, 0),
1760
+ y: safeNum(yTop * PT_TO_MM2, 0),
1503
1761
  fontSize: safeNum(fontSize, 10),
1504
1762
  fontName: it.fontName,
1505
- width: safeNum(w * PT_TO_MM, 0),
1506
- height: safeNum((it.height || fontSize) * PT_TO_MM, fontSize * PT_TO_MM),
1763
+ width: safeNum(w * PT_TO_MM2, 0),
1764
+ height: safeNum((it.height || fontSize) * PT_TO_MM2, fontSize * PT_TO_MM2),
1507
1765
  color: op?.fill || "#000000",
1508
1766
  opacity: invisible ? 0 : safeNum(op?.alpha, 1)
1509
1767
  });
@@ -1511,6 +1769,12 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
1511
1769
  runs.sort((a, b) => a.y - b.y || a.x - b.x);
1512
1770
  const lines = [];
1513
1771
  const Y_TOL = 0.6;
1772
+ const kGlyph = calibrateGlyphWidth(runs);
1773
+ const stretchedSpaces = hasStretchedSpaces(runs, kGlyph);
1774
+ const fontKey = (name) => {
1775
+ const c = fontMap.get(name) || classifyFont(name || "");
1776
+ return `${c.family}|${c.weight || ""}|${c.style || ""}`;
1777
+ };
1514
1778
  for (const r of runs) {
1515
1779
  if (!r.text.length) continue;
1516
1780
  const last = lines[lines.length - 1];
@@ -1519,24 +1783,48 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
1519
1783
  continue;
1520
1784
  }
1521
1785
  const sameLine = Math.abs(last.y - r.y) <= Y_TOL;
1522
- const sameStyle = Math.abs(last.fontSize - r.fontSize) < 0.4 && last.fontName === r.fontName && last.color === r.color && Math.abs(last.opacity - r.opacity) < 0.05;
1523
- const gapMm = r.x - (last.x + last.width);
1524
- const emMm = r.fontSize * PT_TO_MM;
1525
- const mergeOk = sameLine && sameStyle && gapMm >= -0.2 && gapMm <= emMm * 0.45;
1786
+ const sameStyle = Math.abs(last.fontSize - r.fontSize) < 0.4 && (last.fontName === r.fontName || fontKey(last.fontName) === fontKey(r.fontName)) && last.color === r.color && Math.abs(last.opacity - r.opacity) < 0.05;
1787
+ const emMm = r.fontSize * PT_TO_MM2;
1788
+ const extent = (t) => {
1789
+ if (!/\s$/.test(t.text)) return t.width;
1790
+ const em = t.fontSize * PT_TO_MM2;
1791
+ const est = Math.max(1, t.text.trim().length) * em * kGlyph + em * 0.25;
1792
+ if (stretchedSpaces) return Math.min(t.width, est);
1793
+ return t.width > est * 1.4 ? est : t.width;
1794
+ };
1795
+ const gapMm = r.x - (last.x + extent(last));
1796
+ const mergeOk = sameLine && sameStyle && gapMm >= -emMm * 0.5 && gapMm <= emMm * 0.45;
1526
1797
  if (mergeOk) {
1527
1798
  const lastEndsSpace = /\s$/.test(last.text);
1528
1799
  const currStartsSpace = /^\s/.test(r.text);
1529
1800
  const sep = gapMm > emMm * 0.08 && !lastEndsSpace && !currStartsSpace ? " " : "";
1530
1801
  last.text = last.text + sep + r.text;
1531
1802
  const newExtent = r.x - last.x + r.width;
1532
- last.width = Math.max(last.width, newExtent);
1803
+ last.width = Math.max(extent(last), newExtent);
1533
1804
  } else {
1534
1805
  lines.push({ ...r });
1535
1806
  }
1536
1807
  }
1537
1808
  const elements = [];
1538
- for (const sh of ops.shapes) {
1539
- if (sh.width < 0.3 && sh.height < 0.3) continue;
1809
+ const tRuns = lines.map((l) => {
1810
+ const cls = fontMap.get(l.fontName) || classifyFont(l.fontName || "");
1811
+ return { text: l.text, x: l.x, y: l.y, width: l.width, height: l.height, fontSize: l.fontSize, fontName: l.fontName, color: l.color, bold: cls.weight === "bold" };
1812
+ });
1813
+ const detected = options.detectTables === false ? [] : detectTables(tRuns, ops.shapes, pageW * PT_TO_MM2);
1814
+ const consumedLines = /* @__PURE__ */ new Set();
1815
+ const consumedShapes = /* @__PURE__ */ new Set();
1816
+ const tableAtLine = /* @__PURE__ */ new Map();
1817
+ for (const t of detected) {
1818
+ for (const k of t.lineIdx) consumedLines.add(k);
1819
+ for (const k of t.shapeIdx) consumedShapes.add(k);
1820
+ tableAtLine.set(Math.min(...t.lineIdx), t.element);
1821
+ }
1822
+ const pageWmm = pageW * PT_TO_MM2, pageHmm = pageH * PT_TO_MM2;
1823
+ ops.shapes.forEach((sh, shapeIdx) => {
1824
+ if (consumedShapes.has(shapeIdx)) return;
1825
+ if (sh.width < 0.3 && sh.height < 0.3) return;
1826
+ if (sh.x + sh.width <= 0 || sh.y + sh.height <= 0 || sh.x >= pageWmm || sh.y >= pageHmm) return;
1827
+ if (sh.kind === "rect" && sh.fill && !sh.stroke && sh.width * sh.height >= pageWmm * pageHmm * 0.9) return;
1540
1828
  const shapeType = sh.kind;
1541
1829
  const shape = {
1542
1830
  type: "shape",
@@ -1552,7 +1840,7 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
1552
1840
  shape.style = { opacity: Math.round(sh.opacity * 100) / 100 };
1553
1841
  }
1554
1842
  elements.push(shape);
1555
- }
1843
+ });
1556
1844
  const imgs = await extractImages(page, ops.imagePositions, runtime, dataUrlCache);
1557
1845
  for (const { pos, dataUrl } of imgs) {
1558
1846
  let resourceKey = resourceKeyByName.get(pos.name);
@@ -1575,7 +1863,16 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
1575
1863
  fit: "fill"
1576
1864
  });
1577
1865
  }
1866
+ const sizeChars = /* @__PURE__ */ new Map();
1578
1867
  for (const l of lines) {
1868
+ const k = Math.round(l.fontSize * 2) / 2;
1869
+ sizeChars.set(k, (sizeChars.get(k) ?? 0) + l.text.length);
1870
+ }
1871
+ const bodyFontSize = [...sizeChars.entries()].sort((a, b) => b[1] - a[1])[0]?.[0] ?? 0;
1872
+ lines.forEach((l, lineIdx) => {
1873
+ const tableEl = tableAtLine.get(lineIdx);
1874
+ if (tableEl) elements.push(tableEl);
1875
+ if (consumedLines.has(lineIdx)) return;
1579
1876
  const cls = fontMap.get(l.fontName) || classifyFont(l.fontName || "");
1580
1877
  const style = {
1581
1878
  fontSize: Math.round(l.fontSize * 10) / 10,
@@ -1586,8 +1883,7 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
1586
1883
  if (l.color !== "#000000") style.color = l.color;
1587
1884
  if (l.opacity < 0.999) style.opacity = Math.round(l.opacity * 100) / 100;
1588
1885
  const link = findLinkForRun2(l);
1589
- const pageWmm = pageW * PT_TO_MM;
1590
- const measured = Math.max(l.width + l.fontSize * PT_TO_MM * 0.4, l.fontSize * PT_TO_MM);
1886
+ const measured = Math.max(l.width + l.fontSize * PT_TO_MM2 * 0.4, l.fontSize * PT_TO_MM2);
1591
1887
  const remaining = Math.max(measured, pageWmm - l.x);
1592
1888
  const elWidth = Math.min(measured, remaining);
1593
1889
  const text = {
@@ -1597,18 +1893,26 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
1597
1893
  width: Math.max(2, Math.round(elWidth * 100) / 100),
1598
1894
  style
1599
1895
  };
1600
- if (cls.weight === "bold") {
1601
- if (l.fontSize >= 22) text.heading = 1;
1602
- else if (l.fontSize >= 17) text.heading = 2;
1603
- else if (l.fontSize >= 16) text.heading = 3;
1896
+ if (cls.weight === "bold" && l.text.trim().length <= 120 && !consumedLines.has(lineIdx)) {
1897
+ const ratio = bodyFontSize > 0 ? l.fontSize / bodyFontSize : 1;
1898
+ if (l.fontSize >= 22 || ratio >= 1.8) text.heading = 1;
1899
+ else if (l.fontSize >= 17 || ratio >= 1.35) text.heading = 2;
1900
+ else if (l.fontSize >= 16 || ratio >= 1.2) text.heading = 3;
1604
1901
  }
1605
1902
  if (text.heading) text.tocEntry = text.content;
1606
1903
  if (link) {
1607
1904
  if (link.url) text.link = link.url;
1608
1905
  else if (link.destPage != null) text.link = { type: "internal", target: `#page-${link.destPage + 1}` };
1609
1906
  }
1907
+ const prev = elements[elements.length - 1];
1908
+ if (text.heading && prev && prev.type === "text" && prev.heading === text.heading && !link && !prev.link && Math.abs(prev.style?.fontSize - style.fontSize) < 0.5 && Math.abs(prev.position.x - text.position.x) < 1 && text.position.y - prev.position.y < l.fontSize * PT_TO_MM2 * 2.2 && text.position.y > prev.position.y) {
1909
+ prev.content = `${prev.content} ${text.content}`.replace(/\s+/g, " ");
1910
+ prev.tocEntry = prev.content;
1911
+ prev.width = Math.max(prev.width ?? 0, text.width ?? 0);
1912
+ return;
1913
+ }
1610
1914
  elements.push(text);
1611
- }
1915
+ });
1612
1916
  for (const w of formWidgets) {
1613
1917
  if (w.pushButton) continue;
1614
1918
  const baseEl = {
@@ -1643,7 +1947,7 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
1643
1947
  }
1644
1948
  pages.push({
1645
1949
  id: `page-${pi}`,
1646
- pageSize: { width: Math.round(pageW * PT_TO_MM * 100) / 100, height: Math.round(pageH * PT_TO_MM * 100) / 100 },
1950
+ pageSize: { width: Math.round(pageW * PT_TO_MM2 * 100) / 100, height: Math.round(pageH * PT_TO_MM2 * 100) / 100 },
1647
1951
  margins: { top: 0, right: 0, bottom: 0, left: 0 },
1648
1952
  elements
1649
1953
  });
@@ -1796,13 +2100,13 @@ async function importPdfToJdf2(source, title, options = {}) {
1796
2100
 
1797
2101
  // src/commands/import-pdf.ts
1798
2102
  async function importPdf(inputPath, outputPath, options = {}) {
1799
- const input = path2.resolve(inputPath);
1800
- if (!fs.existsSync(input)) {
2103
+ const input = path7.resolve(inputPath);
2104
+ if (!fs7.existsSync(input)) {
1801
2105
  console.error(`File not found: ${input}`);
1802
2106
  process.exit(1);
1803
2107
  }
1804
2108
  console.log(`Importing: ${input}`);
1805
- const title = path2.basename(input, path2.extname(input));
2109
+ const title = path7.basename(input, path7.extname(input));
1806
2110
  const t0 = Date.now();
1807
2111
  const doc = await importPdfToJdf2(input, title, {
1808
2112
  password: options.password,
@@ -1811,7 +2115,7 @@ async function importPdf(inputPath, outputPath, options = {}) {
1811
2115
  console.log(`Parsed in ${((Date.now() - t0) / 1e3).toFixed(1)}s \u2014 ${doc.pages.length} page(s)`);
1812
2116
  let output;
1813
2117
  if (outputPath) {
1814
- output = path2.resolve(outputPath);
2118
+ output = path7.resolve(outputPath);
1815
2119
  } else {
1816
2120
  const stem = input.replace(/\.pdf$/i, "");
1817
2121
  const wantJdfx = !options.forceJson && shouldUseJdfx(doc);
@@ -1820,11 +2124,11 @@ async function importPdf(inputPath, outputPath, options = {}) {
1820
2124
  console.log(`Output: ${output}`);
1821
2125
  if (output.toLowerCase().endsWith(".jdfx")) {
1822
2126
  const { bytes, manifest } = await packJdfx(doc);
1823
- fs.writeFileSync(output, bytes);
2127
+ fs7.writeFileSync(output, bytes);
1824
2128
  console.log(`
1825
2129
  Done! Created ${doc.pages.length} page(s), ${manifest.assets.length} asset(s) bundled`);
1826
2130
  } else {
1827
- fs.writeFileSync(output, JSON.stringify(doc, null, 2));
2131
+ fs7.writeFileSync(output, JSON.stringify(doc, null, 2));
1828
2132
  console.log(`
1829
2133
  Done! Created ${doc.pages.length} page(s)`);
1830
2134
  }
@@ -1837,23 +2141,23 @@ var ImportJsonError = class extends Error {
1837
2141
  }
1838
2142
  };
1839
2143
  async function importJson(inputPath, outputPath, options = {}) {
1840
- const input = path2.resolve(inputPath);
1841
- if (!fs.existsSync(input)) {
2144
+ const input = path7.resolve(inputPath);
2145
+ if (!fs7.existsSync(input)) {
1842
2146
  throw new ImportJsonError(`File not found: ${input}`);
1843
2147
  }
1844
2148
  console.log(`Importing: ${input}`);
1845
- const raw = fs.readFileSync(input, "utf-8");
2149
+ const raw = fs7.readFileSync(input, "utf-8");
1846
2150
  let parsed;
1847
2151
  try {
1848
2152
  parsed = JSON.parse(raw);
1849
2153
  } catch (e) {
1850
2154
  throw new ImportJsonError(`Not valid JSON: ${e.message}`);
1851
2155
  }
1852
- const title = path2.basename(input, path2.extname(input));
2156
+ const title = path7.basename(input, path7.extname(input));
1853
2157
  const doc = normaliseToJdf(parsed, title);
1854
2158
  let output;
1855
2159
  if (outputPath) {
1856
- output = path2.resolve(outputPath);
2160
+ output = path7.resolve(outputPath);
1857
2161
  } else {
1858
2162
  const stem = input.replace(/\.json$/i, "");
1859
2163
  const wantJdfx = !options.forceJson && shouldUseJdfx(doc);
@@ -1862,11 +2166,11 @@ async function importJson(inputPath, outputPath, options = {}) {
1862
2166
  console.log(`Output: ${output}`);
1863
2167
  if (output.toLowerCase().endsWith(".jdfx")) {
1864
2168
  const { bytes, manifest } = await packJdfx(doc);
1865
- fs.writeFileSync(output, bytes);
2169
+ fs7.writeFileSync(output, bytes);
1866
2170
  console.log(`
1867
2171
  Done! Created ${doc.pages.length} page(s), ${manifest.assets.length} asset(s) bundled`);
1868
2172
  } else {
1869
- fs.writeFileSync(output, JSON.stringify(doc, null, 2));
2173
+ fs7.writeFileSync(output, JSON.stringify(doc, null, 2));
1870
2174
  console.log(`
1871
2175
  Done! Created ${doc.pages.length} page(s)`);
1872
2176
  }
@@ -1942,6 +2246,47 @@ function wrapElements(elements, title, meta) {
1942
2246
  ]
1943
2247
  };
1944
2248
  }
2249
+ var DEFAULT_TRANSCRIPT_WINDOW = 45;
2250
+ var fmtTime = (sec) => {
2251
+ const s = Math.max(0, Math.round(sec));
2252
+ const h = Math.floor(s / 3600), m = Math.floor(s % 3600 / 60), r = s % 60;
2253
+ return h ? `${h}:${String(m).padStart(2, "0")}:${String(r).padStart(2, "0")}` : `${String(m).padStart(2, "0")}:${String(r).padStart(2, "0")}`;
2254
+ };
2255
+ function transcriptChunks(el, elementId2, page, crumb, windowSec, maxTokens) {
2256
+ const segs = Array.isArray(el?.transcript?.segments) ? el.transcript.segments : [];
2257
+ if (!segs.length) return [];
2258
+ const chapters = Array.isArray(el?.chapters) ? [...el.chapters].sort((a, b) => a.t - b.t) : [];
2259
+ const chapterAt = (t) => {
2260
+ let cur = null;
2261
+ for (const c of chapters) {
2262
+ if (c.t <= t + 1e-6) cur = c;
2263
+ else break;
2264
+ }
2265
+ return cur;
2266
+ };
2267
+ const out = [];
2268
+ let win = [];
2269
+ const flush = () => {
2270
+ if (!win.length) return;
2271
+ const t0 = win[0].t0, t1 = win[win.length - 1].t1;
2272
+ const body = win.map((sg) => sg.speaker ? `${sg.speaker}: ${sg.text}` : sg.text).join(" ").replace(/\s+/g, " ").trim();
2273
+ const text = `[${fmtTime(t0)}\u2013${fmtTime(t1)}] ${body}`;
2274
+ const chapter = chapterAt(t0);
2275
+ const path9 = [...crumb, ...el.title ? [String(el.title)] : [], ...chapter ? [String(chapter.title)] : []];
2276
+ out.push({ id: `${elementId2}@${Math.round(t0)}`, text, path: path9, page, types: ["video"], tokens: estimateTokens(text), hash: hashText(text), media: { element: elementId2, t0, t1 } });
2277
+ win = [];
2278
+ };
2279
+ for (const sg of segs) {
2280
+ if (typeof sg?.text !== "string" || !sg.text.trim()) continue;
2281
+ const startsNewChapter = win.length && chapterAt(sg.t0) !== chapterAt(win[0].t0);
2282
+ const spansWindow = win.length && sg.t1 - win[0].t0 > windowSec;
2283
+ const overBudget = win.length && estimateTokens(win.map((w) => w.text).join(" ") + sg.text) > maxTokens;
2284
+ if (startsNewChapter || spansWindow || overBudget) flush();
2285
+ win.push(sg);
2286
+ }
2287
+ flush();
2288
+ return out;
2289
+ }
1945
2290
  var DEFAULT_MAX_TOKENS = 512;
1946
2291
  function estimateTokens(text) {
1947
2292
  return Math.ceil(text.length / 4);
@@ -1964,11 +2309,11 @@ function serializeElement(el) {
1964
2309
  case "richtext":
1965
2310
  return (e.runs || []).map((r) => r.text ?? "").join("").trim();
1966
2311
  case "list": {
1967
- const walk = (items, depth = 0) => (items || []).flatMap((it) => {
2312
+ const walk2 = (items, depth = 0) => (items || []).flatMap((it) => {
1968
2313
  const line = " ".repeat(depth) + "- " + String(it.content ?? "").trim();
1969
- return it.children?.length ? [line, ...walk(it.children, depth + 1)] : [line];
2314
+ return it.children?.length ? [line, ...walk2(it.children, depth + 1)] : [line];
1970
2315
  });
1971
- return walk(e.items).join("\n");
2316
+ return walk2(e.items).join("\n");
1972
2317
  }
1973
2318
  case "table": {
1974
2319
  const headers = e.headers ?? e.columns?.map((c) => c.header || "").filter((h) => h) ?? [];
@@ -1999,6 +2344,8 @@ function serializeElement(el) {
1999
2344
  return `${e.checked ? "[x]" : "[ ]"} ${e.label ?? ""}`.trim();
2000
2345
  case "image":
2001
2346
  return e.alt ? `[image: ${e.alt}]` : "";
2347
+ case "video":
2348
+ return e.title ? `[video: ${e.title}]` : "";
2002
2349
  case "toc":
2003
2350
  case "shape":
2004
2351
  case "signature":
@@ -2038,8 +2385,15 @@ function makeChunk(group, breadcrumb) {
2038
2385
  function chunkDocument(doc, options = {}) {
2039
2386
  const strategy = options.strategy ?? "section";
2040
2387
  const maxTokens = options.maxTokens ?? DEFAULT_MAX_TOKENS;
2388
+ const windowSec = options.transcriptWindowSec ?? DEFAULT_TRANSCRIPT_WINDOW;
2041
2389
  const flat = flatten(doc);
2042
2390
  const chunks = [];
2391
+ const withTranscripts = (group, crumb2, c) => {
2392
+ if (c) chunks.push(c);
2393
+ for (const f of group) {
2394
+ if (f.el.type === "video") chunks.push(...transcriptChunks(f.el, f.id, f.page, crumb2, windowSec, maxTokens));
2395
+ }
2396
+ };
2043
2397
  if (strategy === "element") {
2044
2398
  const crumb2 = [];
2045
2399
  for (const f of flat) {
@@ -2048,8 +2402,7 @@ function chunkDocument(doc, options = {}) {
2048
2402
  crumb2.length = Math.max(0, lvl - 1);
2049
2403
  crumb2[lvl - 1] = serializeElement(f.el);
2050
2404
  }
2051
- const c = makeChunk([f], crumb2);
2052
- if (c) chunks.push(c);
2405
+ withTranscripts([f], crumb2, makeChunk([f], crumb2));
2053
2406
  }
2054
2407
  return chunks;
2055
2408
  }
@@ -2058,8 +2411,7 @@ function chunkDocument(doc, options = {}) {
2058
2411
  let buf2 = [];
2059
2412
  let bufTokens = 0;
2060
2413
  const flush = () => {
2061
- const c = makeChunk(buf2, crumb2);
2062
- if (c) chunks.push(c);
2414
+ withTranscripts(buf2, crumb2, makeChunk(buf2, crumb2));
2063
2415
  buf2 = [];
2064
2416
  bufTokens = 0;
2065
2417
  };
@@ -2086,22 +2438,21 @@ function chunkDocument(doc, options = {}) {
2086
2438
  for (const f of buf) {
2087
2439
  const t = estimateTokens(serializeElement(f.el));
2088
2440
  if (subTokens + t > maxTokens && sub.length > 0) {
2089
- const c2 = makeChunk(sub, crumb);
2090
- if (c2) chunks.push(c2);
2441
+ withTranscripts(sub, crumb, makeChunk(sub, crumb));
2091
2442
  sub = [];
2092
2443
  subTokens = 0;
2093
2444
  }
2094
2445
  sub.push(f);
2095
2446
  subTokens += t;
2096
2447
  }
2097
- const c = makeChunk(sub, crumb);
2098
- if (c) chunks.push(c);
2448
+ withTranscripts(sub, crumb, makeChunk(sub, crumb));
2099
2449
  buf = [];
2100
2450
  };
2101
2451
  for (const f of flat) {
2102
2452
  const lvl = headingLevel(f.el);
2103
2453
  if (lvl != null) {
2104
- flushSection();
2454
+ const onlyHeadings = buf.length > 0 && buf.every((b) => headingLevel(b.el) != null);
2455
+ if (!onlyHeadings) flushSection();
2105
2456
  crumb.length = Math.max(0, lvl - 1);
2106
2457
  crumb[lvl - 1] = serializeElement(f.el);
2107
2458
  }
@@ -2112,34 +2463,34 @@ function chunkDocument(doc, options = {}) {
2112
2463
  }
2113
2464
  async function loadJdf(filePath) {
2114
2465
  if (filePath.toLowerCase().endsWith(".jdfx")) {
2115
- const zip = await JSZip.loadAsync(fs.readFileSync(filePath));
2466
+ const zip = await JSZip.loadAsync(fs7.readFileSync(filePath));
2116
2467
  const docFile = zip.file(JDFX_DOCUMENT_PATH);
2117
2468
  if (!docFile) throw new Error(`Bundle missing ${JDFX_DOCUMENT_PATH}`);
2118
2469
  return JSON.parse(await docFile.async("string"));
2119
2470
  }
2120
- return JSON.parse(fs.readFileSync(filePath, "utf-8"));
2471
+ return JSON.parse(fs7.readFileSync(filePath, "utf-8"));
2121
2472
  }
2122
2473
  async function chunkFile(inputPath, opts = {}) {
2123
- const input = path2.resolve(inputPath);
2124
- if (!fs.existsSync(input)) throw new Error(`File not found: ${input}`);
2474
+ const input = path7.resolve(inputPath);
2475
+ if (!fs7.existsSync(input)) throw new Error(`File not found: ${input}`);
2125
2476
  const doc = await loadJdf(input);
2126
2477
  const strategy = opts.strategy ?? "section";
2127
- const chunks = chunkDocument(doc, { strategy, maxTokens: opts.maxTokens });
2478
+ const chunks = chunkDocument(doc, { strategy, maxTokens: opts.maxTokens, transcriptWindowSec: opts.transcriptWindowSec });
2128
2479
  const format = opts.format ?? "jsonl";
2129
2480
  console.log(`Chunking: ${input}`);
2130
2481
  console.log(`Strategy: ${strategy}${opts.maxTokens ? ` (max ${opts.maxTokens} tokens)` : ""}`);
2131
2482
  if (format === "inline") {
2132
- const out = opts.output ? path2.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".jdf");
2483
+ const out = opts.output ? path7.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".jdf");
2133
2484
  const withIndex = { ...doc, index: { chunker: `jdf-${strategy}-v1`, chunks } };
2134
- fs.writeFileSync(out, JSON.stringify(withIndex, null, 2));
2485
+ fs7.writeFileSync(out, JSON.stringify(withIndex, null, 2));
2135
2486
  console.log(`Output: ${out} (${chunks.length} chunks in "index" block)`);
2136
2487
  } else if (format === "json") {
2137
- const out = opts.output ? path2.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".chunks.json");
2138
- fs.writeFileSync(out, JSON.stringify(chunks, null, 2));
2488
+ const out = opts.output ? path7.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".chunks.json");
2489
+ fs7.writeFileSync(out, JSON.stringify(chunks, null, 2));
2139
2490
  console.log(`Output: ${out} (${chunks.length} chunks)`);
2140
2491
  } else {
2141
- const out = opts.output ? path2.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".chunks.jsonl");
2142
- fs.writeFileSync(out, chunks.map((c) => JSON.stringify(c)).join("\n") + "\n");
2492
+ const out = opts.output ? path7.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".chunks.jsonl");
2493
+ fs7.writeFileSync(out, chunks.map((c) => JSON.stringify(c)).join("\n") + "\n");
2143
2494
  console.log(`Output: ${out} (${chunks.length} chunks)`);
2144
2495
  }
2145
2496
  const totalTokens = chunks.reduce((a, c) => a + c.tokens, 0);
@@ -2311,17 +2662,17 @@ async function embedOpenAI(model, inputs) {
2311
2662
  }
2312
2663
  async function loadJdf2(filePath) {
2313
2664
  if (filePath.toLowerCase().endsWith(".jdfx")) {
2314
- const zip = await JSZip.loadAsync(fs.readFileSync(filePath));
2665
+ const zip = await JSZip.loadAsync(fs7.readFileSync(filePath));
2315
2666
  const docFile = zip.file(JDFX_DOCUMENT_PATH);
2316
2667
  if (!docFile) throw new Error(`Bundle missing ${JDFX_DOCUMENT_PATH}`);
2317
2668
  return JSON.parse(await docFile.async("string"));
2318
2669
  }
2319
- return JSON.parse(fs.readFileSync(filePath, "utf-8"));
2670
+ return JSON.parse(fs7.readFileSync(filePath, "utf-8"));
2320
2671
  }
2321
2672
  function loadCache(cachePath) {
2322
2673
  try {
2323
- if (!fs.existsSync(cachePath)) return null;
2324
- return JSON.parse(fs.readFileSync(cachePath, "utf-8"));
2674
+ if (!fs7.existsSync(cachePath)) return null;
2675
+ return JSON.parse(fs7.readFileSync(cachePath, "utf-8"));
2325
2676
  } catch {
2326
2677
  return null;
2327
2678
  }
@@ -2332,15 +2683,15 @@ function batched(items, size) {
2332
2683
  return out;
2333
2684
  }
2334
2685
  async function embedFile(inputPath, opts = {}) {
2335
- const input = path2.resolve(inputPath);
2336
- if (!fs.existsSync(input)) throw new Error(`File not found: ${input}`);
2686
+ const input = path7.resolve(inputPath);
2687
+ if (!fs7.existsSync(input)) throw new Error(`File not found: ${input}`);
2337
2688
  const provider = opts.provider ?? "ollama";
2338
2689
  const model = opts.model ?? DEFAULT_MODEL[provider];
2339
2690
  const strategy = opts.strategy ?? "section";
2340
2691
  const doc = await loadJdf2(input);
2341
- const chunks = chunkDocument(doc, { strategy, maxTokens: opts.maxTokens });
2342
- const output = opts.output ? path2.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".embeddings.json");
2343
- const cachePath = opts.cache ? path2.resolve(opts.cache) : output;
2692
+ const chunks = chunkDocument(doc, { strategy, maxTokens: opts.maxTokens, transcriptWindowSec: opts.transcriptWindowSec });
2693
+ const output = opts.output ? path7.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".embeddings.json");
2694
+ const cachePath = opts.cache ? path7.resolve(opts.cache) : output;
2344
2695
  console.log(`Embedding: ${input}`);
2345
2696
  console.log(`Provider: ${provider} / ${model}${provider === "ollama" ? " (local \u2014 no data leaves this machine)" : " (remote API)"}`);
2346
2697
  console.log(`Strategy: ${strategy} \u2192 ${chunks.length} chunks`);
@@ -2379,11 +2730,295 @@ async function embedFile(inputPath, opts = {}) {
2379
2730
  chunker: `jdf-${strategy}-v1`,
2380
2731
  vectors
2381
2732
  };
2382
- fs.writeFileSync(output, JSON.stringify(sidecar));
2733
+ fs7.writeFileSync(output, JSON.stringify(sidecar));
2383
2734
  console.log(`
2384
2735
  Done! ${Object.keys(vectors).length} vectors (${dims}-dim) \u2192 ${output}`);
2385
2736
  return sidecar;
2386
2737
  }
2738
+ var toSec = (ts) => {
2739
+ const m = ts.trim().replace(",", ".").match(/^(?:(\d+):)?(\d{1,2}):(\d{2}(?:\.\d+)?)$/);
2740
+ if (!m) throw new Error(`bad timestamp "${ts}"`);
2741
+ return (m[1] ? Number(m[1]) * 3600 : 0) + Number(m[2]) * 60 + Number(m[3]);
2742
+ };
2743
+ function parseSubtitles(text, filename = "") {
2744
+ const trimmed = text.replace(/^/, "").trim();
2745
+ if (trimmed.startsWith("{") || trimmed.startsWith("[")) {
2746
+ const j = JSON.parse(trimmed);
2747
+ const arr = Array.isArray(j) ? j : Array.isArray(j.segments) ? j.segments : [];
2748
+ return arr.map((sg) => ({
2749
+ t0: Number(sg.t0 ?? sg.start ?? sg.from ?? 0),
2750
+ t1: Number(sg.t1 ?? sg.end ?? sg.to ?? 0),
2751
+ text: String(sg.text ?? "").trim(),
2752
+ ...sg.speaker ? { speaker: String(sg.speaker) } : {}
2753
+ })).filter((sg) => sg.text);
2754
+ }
2755
+ const segs = [];
2756
+ for (const block of trimmed.split(/\r?\n\r?\n+/)) {
2757
+ const lines = block.split(/\r?\n/).filter((l) => l.trim() !== "" && l.trim() !== "WEBVTT");
2758
+ const ti = lines.findIndex((l) => l.includes("-->"));
2759
+ if (ti < 0) continue;
2760
+ const [a, b] = lines[ti].split("-->").map((x) => x.trim().split(/\s+/)[0]);
2761
+ const body = lines.slice(ti + 1).join(" ").replace(/<[^>]+>/g, "").replace(/\s+/g, " ").trim();
2762
+ if (!body) continue;
2763
+ segs.push({ t0: toSec(a), t1: toSec(b), text: body });
2764
+ }
2765
+ if (!segs.length) throw new Error(`no cues found in ${filename || "subtitle input"} (expected SRT, WebVTT or JSON segments)`);
2766
+ return segs;
2767
+ }
2768
+ function parseChapters(text) {
2769
+ const t = text.trim();
2770
+ if (t.startsWith("[")) return JSON.parse(t).map((c) => ({ t: Number(c.t ?? c.start ?? 0), title: String(c.title ?? "") }));
2771
+ return t.split(/\r?\n/).map((l) => l.trim()).filter(Boolean).map((l) => {
2772
+ const m = l.match(/^(\S+)\s+(.+)$/);
2773
+ if (!m) throw new Error(`bad chapter line "${l}" (expected "mm:ss Title")`);
2774
+ return { t: toSec(m[1]), title: m[2].trim() };
2775
+ });
2776
+ }
2777
+ async function loadDoc(file) {
2778
+ if (file.toLowerCase().endsWith(".jdfx")) {
2779
+ const zip = await JSZip.loadAsync(fs7.readFileSync(file));
2780
+ const f = zip.file(JDFX_DOCUMENT_PATH);
2781
+ if (!f) throw new Error(`Bundle missing ${JDFX_DOCUMENT_PATH}`);
2782
+ const doc = JSON.parse(await f.async("string"));
2783
+ const manifest = zip.file("manifest.json") ? JSON.parse(await zip.file("manifest.json").async("string")) : { assets: [] };
2784
+ for (const a of manifest.assets ?? []) {
2785
+ const af = zip.file(a.path);
2786
+ if (!af) continue;
2787
+ const data = (await af.async("nodebuffer")).toString("base64");
2788
+ const res = { src: "embedded", mimeType: a.mimeType, data };
2789
+ doc.resources ??= {};
2790
+ if (/^video\//i.test(a.mimeType || "")) (doc.resources.videos ??= {})[a.id] = res;
2791
+ else (doc.resources.images ??= {})[a.id] = res;
2792
+ }
2793
+ return { doc, bundle: true, zip };
2794
+ }
2795
+ return { doc: JSON.parse(fs7.readFileSync(file, "utf-8")), bundle: false };
2796
+ }
2797
+ function findVideos(doc) {
2798
+ const out = [];
2799
+ const walk2 = (els, page) => {
2800
+ for (const el of els ?? []) {
2801
+ if (el?.type === "video") out.push({ el, page, index: out.length });
2802
+ if (el?.elements) walk2(el.elements, page);
2803
+ }
2804
+ };
2805
+ doc.pages.forEach((p, i) => walk2(p.elements, i + 1));
2806
+ return out;
2807
+ }
2808
+ async function clipToTempFile(doc, el, docDir) {
2809
+ const tmp = path7.join(fs7.mkdtempSync(path7.join(__require("os").tmpdir(), "jdf-transcribe-")), "clip.mp4");
2810
+ const res = el.resource ? doc.resources?.videos?.[el.resource] ?? doc.resources?.images?.[el.resource] : void 0;
2811
+ if (res?.data) {
2812
+ fs7.writeFileSync(tmp, Buffer.from(res.data.replace(/^data:[^,]*,/, ""), "base64"));
2813
+ return tmp;
2814
+ }
2815
+ if (res?.path) return path7.resolve(docDir, res.path);
2816
+ const src = el.src;
2817
+ if (!src) return null;
2818
+ if (src.startsWith("data:")) {
2819
+ fs7.writeFileSync(tmp, Buffer.from(src.replace(/^data:[^,]*,/, ""), "base64"));
2820
+ return tmp;
2821
+ }
2822
+ if (/^https?:\/\//i.test(src)) {
2823
+ const r = await fetch(src);
2824
+ if (!r.ok) throw new Error(`download failed ${r.status}: ${src}`);
2825
+ fs7.writeFileSync(tmp, Buffer.from(await r.arrayBuffer()));
2826
+ return tmp;
2827
+ }
2828
+ const local = path7.resolve(docDir, src);
2829
+ return fs7.existsSync(local) ? local : null;
2830
+ }
2831
+ function whisperCli(clip, model, language, prompt2) {
2832
+ const ffmpeg = spawnSync("ffmpeg", ["-version"]);
2833
+ if (ffmpeg.error) throw new Error("ffmpeg not found \u2014 needed to extract audio for whisper-cli (brew install ffmpeg)");
2834
+ const wav = clip.replace(/\.[^.]+$/, "") + ".16k.wav";
2835
+ const ex = spawnSync("ffmpeg", ["-y", "-i", clip, "-vn", "-ac", "1", "-ar", "16000", "-f", "wav", wav], { encoding: "utf-8" });
2836
+ if (ex.status !== 0) throw new Error(`ffmpeg failed: ${ex.stderr.slice(-400)}`);
2837
+ const args = ["-f", wav, "-oj", "-of", wav.replace(/\.wav$/, "")];
2838
+ if (model) args.push("-m", model);
2839
+ if (language) args.push("-l", language);
2840
+ if (prompt2) args.push("--prompt", prompt2);
2841
+ const run = spawnSync("whisper-cli", args, { encoding: "utf-8" });
2842
+ if (run.error) throw new Error("whisper-cli not found \u2014 install whisper.cpp (brew install whisper-cpp) or use --from / --provider openai");
2843
+ if (run.status !== 0) throw new Error(`whisper-cli failed: ${run.stderr.slice(-400)}`);
2844
+ const j = JSON.parse(fs7.readFileSync(wav.replace(/\.wav$/, "") + ".json", "utf-8"));
2845
+ const segs = j.transcription ?? j.segments ?? [];
2846
+ const ms = (x) => typeof x === "number" ? x / 1e3 : toSec(String(x).replace(",", "."));
2847
+ return segs.map((sg) => ({ t0: ms(sg.offsets?.from ?? sg.start), t1: ms(sg.offsets?.to ?? sg.end), text: String(sg.text ?? "").trim() })).filter((sg) => sg.text);
2848
+ }
2849
+ async function openaiTranscribe(clip, model, language, prompt2) {
2850
+ const key = process.env.OPENAI_API_KEY;
2851
+ if (!key) throw new Error("OPENAI_API_KEY is not set");
2852
+ const base = process.env.OPENAI_BASE_URL || "https://api.openai.com/v1";
2853
+ const form = new FormData();
2854
+ form.append("file", new Blob([fs7.readFileSync(clip)]), path7.basename(clip));
2855
+ form.append("model", model || "whisper-1");
2856
+ form.append("response_format", "verbose_json");
2857
+ form.append("timestamp_granularities[]", "segment");
2858
+ if (language) form.append("language", language);
2859
+ if (prompt2) form.append("prompt", prompt2);
2860
+ const r = await fetch(`${base}/audio/transcriptions`, { method: "POST", headers: { Authorization: `Bearer ${key}` }, body: form });
2861
+ if (!r.ok) throw new Error(`OpenAI transcription failed ${r.status}: ${(await r.text()).slice(0, 300)}`);
2862
+ const j = await r.json();
2863
+ return (j.segments ?? []).map((sg) => ({ t0: Number(sg.start), t1: Number(sg.end), text: String(sg.text).trim() })).filter((sg) => sg.text);
2864
+ }
2865
+ async function transcribeFile(inputPath, opts = {}) {
2866
+ const input = path7.resolve(inputPath);
2867
+ if (!fs7.existsSync(input)) throw new Error(`File not found: ${input}`);
2868
+ const { doc, bundle } = await loadDoc(input);
2869
+ const videos = findVideos(doc);
2870
+ if (!videos.length) throw new Error("document has no video element");
2871
+ let target = videos[0];
2872
+ if (opts.element != null) {
2873
+ const byId = videos.find((v) => v.el.id === opts.element);
2874
+ const byIdx = /^\d+$/.test(opts.element) ? videos[Number(opts.element)] : void 0;
2875
+ target = byId ?? byIdx ?? (() => {
2876
+ throw new Error(`no video element "${opts.element}" (have: ${videos.map((v) => v.el.id ?? `#${v.index}`).join(", ")})`);
2877
+ })();
2878
+ } else if (videos.length > 1) {
2879
+ throw new Error(`document has ${videos.length} videos \u2014 pick one with --element <id|index>`);
2880
+ }
2881
+ let segments;
2882
+ let source;
2883
+ if (opts.from) {
2884
+ segments = parseSubtitles(fs7.readFileSync(path7.resolve(opts.from), "utf-8"), opts.from);
2885
+ source = `${path7.extname(opts.from).slice(1).toLowerCase() || "file"}-import`;
2886
+ } else {
2887
+ const provider = opts.provider ?? "whisper-cli";
2888
+ const clip = await clipToTempFile(doc, target.el, path7.dirname(input));
2889
+ if (!clip) throw new Error("could not locate the clip bytes (no bundled asset, data URL, local path or http URL) \u2014 use --from to import subtitles instead");
2890
+ segments = provider === "openai" ? await openaiTranscribe(clip, opts.model, opts.language, opts.prompt) : whisperCli(clip, opts.model, opts.language, opts.prompt);
2891
+ source = provider === "openai" ? `openai:${opts.model || "whisper-1"}` : `whisper-cli${opts.model ? ":" + path7.basename(opts.model) : ""}`;
2892
+ }
2893
+ segments.sort((a, b) => a.t0 - b.t0);
2894
+ const transcript = { ...opts.language ? { language: opts.language } : {}, source, created: (/* @__PURE__ */ new Date()).toISOString(), segments };
2895
+ target.el.transcript = transcript;
2896
+ if (opts.chapters) target.el.chapters = parseChapters(fs7.readFileSync(path7.resolve(opts.chapters), "utf-8"));
2897
+ if (!target.el.id) target.el.id = `video-${target.index + 1}`;
2898
+ const output = opts.output ? path7.resolve(opts.output) : input;
2899
+ if (output.toLowerCase().endsWith(".jdfx") || bundle && !opts.output) {
2900
+ const { bytes } = await packJdfx(doc);
2901
+ fs7.writeFileSync(output, bytes);
2902
+ } else {
2903
+ fs7.writeFileSync(output, JSON.stringify(doc, null, 2));
2904
+ }
2905
+ const dur = segments.length ? segments[segments.length - 1].t1 : 0;
2906
+ console.log(`Transcribed: ${path7.basename(input)} \u2192 element "${target.el.id}" (${segments.length} segments, ${Math.round(dur)} s, source ${source})`);
2907
+ if (target.el.chapters) console.log(`Chapters: ${target.el.chapters.length}`);
2908
+ console.log(`Output: ${output}
2909
+ Next: jdf chunk ${path7.basename(output)} # transcript \u2192 time-windowed chunks with media.t0/t1`);
2910
+ return transcript;
2911
+ }
2912
+ var CONFIG_NAME = "jdf.rag.json";
2913
+ var OUT_DIR = ".jdf-rag";
2914
+ function walk(dir, acc = []) {
2915
+ for (const ent of fs7.readdirSync(dir, { withFileTypes: true })) {
2916
+ if (ent.name === "node_modules" || ent.name === OUT_DIR || ent.name.startsWith(".")) continue;
2917
+ const p = path7.join(dir, ent.name);
2918
+ if (ent.isDirectory()) walk(p, acc);
2919
+ else if (/\.(jdf|jdfx)$/i.test(ent.name)) acc.push(p);
2920
+ }
2921
+ return acc.sort();
2922
+ }
2923
+ async function readDoc(file) {
2924
+ if (file.toLowerCase().endsWith(".jdfx")) {
2925
+ const zip = await JSZip.loadAsync(fs7.readFileSync(file));
2926
+ return JSON.parse(await zip.file(JDFX_DOCUMENT_PATH).async("string"));
2927
+ }
2928
+ return JSON.parse(fs7.readFileSync(file, "utf-8"));
2929
+ }
2930
+ function videosIn(doc) {
2931
+ const out = [];
2932
+ const w = (els) => {
2933
+ for (const el of els ?? []) {
2934
+ if (el?.type === "video") out.push({ id: el.id, hasTranscript: !!el.transcript?.segments?.length });
2935
+ if (el?.elements) w(el.elements);
2936
+ }
2937
+ };
2938
+ for (const p of doc.pages ?? []) w(p.elements);
2939
+ return out;
2940
+ }
2941
+ async function ragFolder(dirPath, cli = {}) {
2942
+ const dir = path7.resolve(dirPath);
2943
+ if (!fs7.existsSync(dir) || !fs7.statSync(dir).isDirectory()) throw new Error(`Not a directory: ${dir}`);
2944
+ const cfgPath = path7.join(dir, CONFIG_NAME);
2945
+ const cfg = fs7.existsSync(cfgPath) ? JSON.parse(fs7.readFileSync(cfgPath, "utf-8")) : {};
2946
+ const opts = { ...cfg, ...Object.fromEntries(Object.entries(cli).filter(([, v]) => v !== void 0)) };
2947
+ const provider = opts.provider ?? "ollama";
2948
+ const transcribe = opts.transcribe ?? "none";
2949
+ const outDir = path7.resolve(opts.out ?? path7.join(dir, OUT_DIR));
2950
+ const files = walk(dir);
2951
+ console.log(`jdf rag: ${dir}
2952
+ files: ${files.length} (.jdf/.jdfx)${fs7.existsSync(cfgPath) ? `
2953
+ config: ${CONFIG_NAME}` : ""}
2954
+ embeddings: ${opts.noEmbed ? "skipped (--no-embed)" : `${provider}${opts.model ? " / " + opts.model : ""}`}
2955
+ transcribe: ${transcribe}${opts.dryRun ? "\n DRY RUN \u2014 nothing written" : ""}
2956
+ `);
2957
+ if (!files.length) {
2958
+ console.log("Nothing to do.");
2959
+ return;
2960
+ }
2961
+ const manifest = { dir, created: (/* @__PURE__ */ new Date()).toISOString(), provider: opts.noEmbed ? null : provider, model: opts.model ?? null, strategy: opts.strategy ?? "section", transcribe, files: [], totals: { files: files.length, chunks: 0, videoChunks: 0, videos: 0, transcribed: 0, untranscribed: 0 } };
2962
+ const indexLines = [];
2963
+ for (const file of files) {
2964
+ const rel = path7.relative(dir, file);
2965
+ const doc = await readDoc(file);
2966
+ const vids = videosIn(doc);
2967
+ let transcribedHere = 0;
2968
+ for (const v of vids) {
2969
+ if (v.hasTranscript) continue;
2970
+ if (transcribe === "none") {
2971
+ manifest.totals.untranscribed++;
2972
+ continue;
2973
+ }
2974
+ if (opts.dryRun) {
2975
+ transcribedHere++;
2976
+ continue;
2977
+ }
2978
+ try {
2979
+ await transcribeFile(file, { provider: transcribe, element: v.id, model: opts.transcribeModel, language: opts.language, prompt: opts.prompt });
2980
+ transcribedHere++;
2981
+ } catch (e) {
2982
+ console.warn(` ! ${rel}: transcription failed for video ${v.id ?? "#?"}: ${e.message}`);
2983
+ manifest.totals.untranscribed++;
2984
+ }
2985
+ }
2986
+ manifest.totals.videos += vids.length;
2987
+ manifest.totals.transcribed += transcribedHere;
2988
+ if (opts.dryRun) {
2989
+ manifest.files.push({ file: rel, videos: vids.length, wouldTranscribe: transcribedHere });
2990
+ continue;
2991
+ }
2992
+ const chunkOpts = { strategy: opts.strategy, maxTokens: opts.maxTokens, transcriptWindowSec: opts.transcriptWindowSec };
2993
+ const chunkOut = path7.join(outDir, "chunks", rel.replace(/\.(jdf|jdfx)$/i, ".chunks.jsonl"));
2994
+ fs7.mkdirSync(path7.dirname(chunkOut), { recursive: true });
2995
+ let chunks;
2996
+ if (opts.noEmbed) {
2997
+ chunks = await chunkFile(file, { ...chunkOpts, format: "jsonl", output: chunkOut });
2998
+ } else {
2999
+ const side = await embedFile(file, { ...chunkOpts, provider, model: opts.model, incremental: true });
3000
+ chunks = await chunkFile(file, { ...chunkOpts, format: "jsonl", output: chunkOut });
3001
+ manifest.files.push({ file: rel, chunks: chunks.length, vectors: Object.keys(side.vectors).length, sidecar: path7.relative(dir, file.replace(/\.(jdf|jdfx)$/i, ".embeddings.json")), videos: vids.length, transcribed: transcribedHere });
3002
+ }
3003
+ if (opts.noEmbed) manifest.files.push({ file: rel, chunks: chunks.length, videos: vids.length, transcribed: transcribedHere });
3004
+ for (const c of chunks) {
3005
+ indexLines.push(JSON.stringify({ file: rel, ...c }));
3006
+ manifest.totals.chunks++;
3007
+ if (c.media) manifest.totals.videoChunks++;
3008
+ }
3009
+ }
3010
+ if (!opts.dryRun) {
3011
+ fs7.mkdirSync(outDir, { recursive: true });
3012
+ fs7.writeFileSync(path7.join(outDir, "index.jsonl"), indexLines.join("\n") + (indexLines.length ? "\n" : ""));
3013
+ fs7.writeFileSync(path7.join(outDir, "manifest.json"), JSON.stringify(manifest, null, 2) + "\n");
3014
+ }
3015
+ const t = manifest.totals;
3016
+ console.log(`
3017
+ Done. ${t.files} files \u2192 ${t.chunks} chunks (${t.videoChunks} from video transcripts); videos ${t.videos}, transcribed now ${t.transcribed}, still without transcript ${t.untranscribed}${t.untranscribed && transcribe === "none" ? " (pass --transcribe whisper-cli|openai, or jdf transcribe --from subs.srt)" : ""}.`);
3018
+ if (!opts.dryRun) console.log(`Index: ${path7.join(outDir, "index.jsonl")}
3019
+ Report: ${path7.join(outDir, "manifest.json")}${opts.noEmbed ? "" : `
3020
+ Vectors: one <file>.embeddings.json next to each document (incremental \u2014 re-run any time)`}`);
3021
+ }
2387
3022
 
2388
3023
  // src/index.ts
2389
3024
  var HELP = `jdf \u2014 JSON Document Format CLI
@@ -2395,12 +3030,18 @@ The CLI exists for these workflows:
2395
3030
  into a validated .jdf (or .jdfx) you can ship.
2396
3031
  \u2022 JDF \u2192 chunks turn a document into retrieval-ready chunks (RAG).
2397
3032
  \u2022 JDF \u2192 vectors embed those chunks, incrementally, for a vector store.
3033
+ \u2022 video \u2192 text attach a time-stamped transcript to a video element so
3034
+ RAG retrieves "video at 02:13", not just "a video".
3035
+ \u2022 folder \u2192 index one command over a directory of .jdf/.jdfx: transcribe,
3036
+ chunk, embed incrementally, write .jdf-rag/index.jsonl.
2398
3037
 
2399
3038
  Usage:
2400
3039
  jdf validate <file.jdf>
2401
3040
  jdf convert <file.{pdf,json,md}> [-o output.{jdf,jdfx}] [--json] [--password PW] [--drop-invisible-text]
2402
3041
  jdf chunk <file.{jdf,jdfx}> [--strategy section|element|fixed] [--format jsonl|json|inline] [--max-tokens N] [-o out]
2403
3042
  jdf embed <file.{jdf,jdfx}> [--provider ollama|openai] [--model NAME] [--strategy \u2026] [--incremental] [-o out]
3043
+ jdf transcribe <file.{jdf,jdfx}> [--from subs.srt|.vtt|.json] [--provider whisper-cli|openai] [--element ID] [--chapters FILE] [-o out]
3044
+ jdf rag <dir> [--provider ollama|openai] [--model NAME] [--transcribe none|whisper-cli|openai] [--no-embed] [--dry-run] [--out DIR]
2404
3045
  jdf --help
2405
3046
 
2406
3047
  Commands:
@@ -2408,6 +3049,8 @@ Commands:
2408
3049
  convert Convert a PDF, JSON, or Markdown file into JDF (alias: import)
2409
3050
  chunk Split a JDF document into retrieval-ready chunks (offline, deterministic)
2410
3051
  embed Compute embeddings for the chunks (local via Ollama by default)
3052
+ transcribe Store time-stamped text on a video element (import SRT/VTT/JSON, or run Whisper)
3053
+ rag Make a whole folder retrieval-ready (finds .jdf/.jdfx, transcribes, chunks, embeds, indexes)
2411
3054
 
2412
3055
  Flags:
2413
3056
  -o, --output <path> Explicit output path
@@ -2424,7 +3067,20 @@ Flags:
2424
3067
  --provider <p> embed: ollama (default, local) | openai (remote API)
2425
3068
  --model <name> embed: model id (default: nomic-embed-text / text-embedding-3-small)
2426
3069
  --incremental embed: skip chunks whose content hash is unchanged
3070
+ --cache <path> embed: sidecar to reuse vectors from (default: the
3071
+ output path itself)
2427
3072
  --no-auto-start embed(ollama): don't auto-launch Ollama via Docker
3073
+ --from <file> transcribe: import subtitles (.srt / .vtt / JSON segments) \u2014 offline, no model
3074
+ --element <id|n> transcribe: which video element (id, or 0-based index); default the only one
3075
+ --chapters <file> transcribe: JSON [{t,title}] or "mm:ss Title" lines \u2192 chapter breadcrumbs
3076
+ --language <tag> transcribe: BCP-47 language hint for Whisper
3077
+ --prompt <text> transcribe: Whisper vocabulary hint (names, acronyms) \u2014 not an instruction
3078
+ --window <sec> chunk/embed/rag: transcript window per video chunk (default 45)
3079
+ --transcribe <p> rag: none (default) | whisper-cli | openai \u2014 for videos that have no transcript yet
3080
+ --no-embed rag: chunk + index only
3081
+ --dry-run rag: list what would happen, write nothing
3082
+ --out <dir> rag: index folder (default <dir>/.jdf-rag)
3083
+ rag reads defaults from <dir>/jdf.rag.json (same keys as the flags; flags win)
2428
3084
 
2429
3085
  Environment (embed):
2430
3086
  ollama: OLLAMA_HOST (default http://localhost:11434)
@@ -2438,8 +3094,11 @@ Examples:
2438
3094
  jdf chunk report.jdf --format inline # embed the chunk index into the .jdf
2439
3095
  jdf embed report.jdf # local embeddings via Ollama (auto-setup)
2440
3096
  jdf embed report.jdf --provider openai --incremental
3097
+ jdf transcribe talk.jdfx --from talk.srt --chapters chapters.txt # then: jdf chunk talk.jdfx
3098
+ jdf transcribe talk.jdfx --provider openai --language en --prompt "JDF, jdfx, Ollama"
3099
+ jdf rag ./knowledge-base --transcribe openai # whole folder \u2192 .jdf-rag/index.jsonl
2441
3100
  `;
2442
- var BOOLEAN_FLAGS = /* @__PURE__ */ new Set(["help", "h", "json", "verbose", "skip-validate", "incremental", "no-auto-start", "drop-invisible-text"]);
3101
+ var BOOLEAN_FLAGS = /* @__PURE__ */ new Set(["help", "h", "json", "verbose", "skip-validate", "incremental", "no-auto-start", "drop-invisible-text", "no-embed", "dry-run"]);
2443
3102
  function parseArgs(argv) {
2444
3103
  const positional = [];
2445
3104
  const flags = {};
@@ -2543,14 +3202,55 @@ async function main() {
2543
3202
  strategy: typeof flags.strategy === "string" ? flags.strategy : void 0,
2544
3203
  format: typeof flags.format === "string" ? flags.format : void 0,
2545
3204
  maxTokens: typeof flags["max-tokens"] === "string" ? parseInt(flags["max-tokens"], 10) : void 0,
3205
+ transcriptWindowSec: typeof flags.window === "string" ? parseInt(flags.window, 10) : void 0,
3206
+ output: typeof flags.output === "string" ? flags.output : void 0
3207
+ });
3208
+ process.exit(0);
3209
+ }
3210
+ case "transcribe": {
3211
+ const input = positional[0];
3212
+ if (!input) {
3213
+ console.error("Usage: jdf transcribe <file.{jdf,jdfx}> [--from subs.srt|.vtt|.json] [--provider whisper-cli|openai] [--model M] [--language tag] [--element id|n] [--chapters file] [-o out]");
3214
+ process.exit(1);
3215
+ }
3216
+ await transcribeFile(input, {
3217
+ from: typeof flags.from === "string" ? flags.from : void 0,
3218
+ provider: typeof flags.provider === "string" ? flags.provider : void 0,
3219
+ model: typeof flags.model === "string" ? flags.model : void 0,
3220
+ language: typeof flags.language === "string" ? flags.language : void 0,
3221
+ prompt: typeof flags.prompt === "string" ? flags.prompt : void 0,
3222
+ element: typeof flags.element === "string" ? flags.element : void 0,
3223
+ chapters: typeof flags.chapters === "string" ? flags.chapters : void 0,
2546
3224
  output: typeof flags.output === "string" ? flags.output : void 0
2547
3225
  });
2548
3226
  process.exit(0);
2549
3227
  }
3228
+ case "rag": {
3229
+ const input = positional[0];
3230
+ if (!input) {
3231
+ console.error("Usage: jdf rag <dir> [--provider ollama|openai] [--model NAME] [--strategy \u2026] [--window sec] [--transcribe none|whisper-cli|openai] [--language tag] [--prompt text] [--no-embed] [--dry-run] [--out DIR]");
3232
+ process.exit(1);
3233
+ }
3234
+ await ragFolder(input, {
3235
+ provider: typeof flags.provider === "string" ? flags.provider : void 0,
3236
+ model: typeof flags.model === "string" ? flags.model : void 0,
3237
+ strategy: typeof flags.strategy === "string" ? flags.strategy : void 0,
3238
+ maxTokens: typeof flags["max-tokens"] === "string" ? parseInt(flags["max-tokens"], 10) : void 0,
3239
+ transcriptWindowSec: typeof flags.window === "string" ? parseInt(flags.window, 10) : void 0,
3240
+ transcribe: typeof flags.transcribe === "string" ? flags.transcribe : void 0,
3241
+ transcribeModel: typeof flags["transcribe-model"] === "string" ? flags["transcribe-model"] : void 0,
3242
+ language: typeof flags.language === "string" ? flags.language : void 0,
3243
+ prompt: typeof flags.prompt === "string" ? flags.prompt : void 0,
3244
+ noEmbed: flags["no-embed"] === true,
3245
+ dryRun: flags["dry-run"] === true,
3246
+ out: typeof flags.out === "string" ? flags.out : void 0
3247
+ });
3248
+ process.exit(0);
3249
+ }
2550
3250
  case "embed": {
2551
3251
  const input = positional[0];
2552
3252
  if (!input) {
2553
- console.error("Usage: jdf embed <file.{jdf,jdfx}> [--provider ollama|openai] [--model NAME] [--strategy \u2026] [--incremental] [--no-auto-start] [-o out]");
3253
+ console.error("Usage: jdf embed <file.{jdf,jdfx}> [--provider ollama|openai] [--model NAME] [--strategy \u2026] [--incremental] [--cache prev.embeddings.json] [--no-auto-start] [-o out]");
2554
3254
  process.exit(1);
2555
3255
  }
2556
3256
  await embedFile(input, {
@@ -2559,8 +3259,10 @@ async function main() {
2559
3259
  strategy: typeof flags.strategy === "string" ? flags.strategy : void 0,
2560
3260
  maxTokens: typeof flags["max-tokens"] === "string" ? parseInt(flags["max-tokens"], 10) : void 0,
2561
3261
  incremental: flags.incremental === true,
3262
+ transcriptWindowSec: typeof flags.window === "string" ? parseInt(flags.window, 10) : void 0,
2562
3263
  autoStart: flags["no-auto-start"] !== true,
2563
- output: typeof flags.output === "string" ? flags.output : void 0
3264
+ output: typeof flags.output === "string" ? flags.output : void 0,
3265
+ cache: typeof flags.cache === "string" ? flags.cache : void 0
2564
3266
  });
2565
3267
  process.exit(0);
2566
3268
  }