@uurtech/jdf-cli 0.1.26 → 0.2.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -1,13 +1,20 @@
1
1
  #!/usr/bin/env node
2
- import fs from 'fs';
3
- import path2 from 'path';
2
+ import fs7 from 'fs';
3
+ import path7 from 'path';
4
4
  import { fileURLToPath } from 'url';
5
5
  import Ajv from 'ajv';
6
6
  import addFormats from 'ajv-formats';
7
7
  import JSZip from 'jszip';
8
8
  import crypto, { createHash } from 'crypto';
9
9
  import { readFile } from 'fs/promises';
10
- import { execFileSync } from 'child_process';
10
+ import { execFileSync, spawnSync } from 'child_process';
11
+
12
+ var __require = /* @__PURE__ */ ((x) => typeof require !== "undefined" ? require : typeof Proxy !== "undefined" ? new Proxy(x, {
13
+ get: (a, b) => (typeof require !== "undefined" ? require : a)[b]
14
+ }) : x)(function(x) {
15
+ if (typeof require !== "undefined") return require.apply(this, arguments);
16
+ throw Error('Dynamic require of "' + x + '" is not supported');
17
+ });
11
18
 
12
19
  // ../../packages/jdf-core/src/manifest.ts
13
20
  var JDFX_MANIFEST_VERSION = "1.0.0";
@@ -16,17 +23,17 @@ var JDFX_MANIFEST_PATH = "manifest.json";
16
23
  var JDFX_ASSET_DIR = "assets";
17
24
 
18
25
  // src/commands/validate.ts
19
- var __dirname$1 = path2.dirname(fileURLToPath(import.meta.url));
26
+ var __dirname$1 = path7.dirname(fileURLToPath(import.meta.url));
20
27
  function resolveSchemaPath() {
21
- const bundled = path2.resolve(__dirname$1, "jdf-schema.json");
22
- if (fs.existsSync(bundled)) return bundled;
23
- const dev = path2.resolve(__dirname$1, "../../../../spec/jdf-schema.json");
28
+ const bundled = path7.resolve(__dirname$1, "jdf-schema.json");
29
+ if (fs7.existsSync(bundled)) return bundled;
30
+ const dev = path7.resolve(__dirname$1, "../../../../spec/jdf-schema.json");
24
31
  return dev;
25
32
  }
26
33
  var SCHEMA_PATH = resolveSchemaPath();
27
34
  async function loadDocument(filePath) {
28
35
  if (filePath.toLowerCase().endsWith(".jdfx")) {
29
- const zip = await JSZip.loadAsync(fs.readFileSync(filePath));
36
+ const zip = await JSZip.loadAsync(fs7.readFileSync(filePath));
30
37
  const docFile = zip.file(JDFX_DOCUMENT_PATH);
31
38
  if (!docFile) {
32
39
  console.error(`\u2717 Bundle missing ${JDFX_DOCUMENT_PATH}`);
@@ -51,11 +58,11 @@ async function loadDocument(filePath) {
51
58
  }
52
59
  return { doc, bundle: { manifest, assetCount } };
53
60
  }
54
- return { doc: JSON.parse(fs.readFileSync(filePath, "utf-8")) };
61
+ return { doc: JSON.parse(fs7.readFileSync(filePath, "utf-8")) };
55
62
  }
56
63
  async function validate(file) {
57
- const filePath = path2.resolve(file);
58
- if (!fs.existsSync(filePath)) {
64
+ const filePath = path7.resolve(file);
65
+ if (!fs7.existsSync(filePath)) {
59
66
  console.error(`File not found: ${filePath}`);
60
67
  return false;
61
68
  }
@@ -68,11 +75,11 @@ async function validate(file) {
68
75
  }
69
76
  if (!loaded) return false;
70
77
  const { doc, bundle } = loaded;
71
- if (!fs.existsSync(SCHEMA_PATH)) {
78
+ if (!fs7.existsSync(SCHEMA_PATH)) {
72
79
  console.error(`Schema not found at ${SCHEMA_PATH}`);
73
80
  return false;
74
81
  }
75
- const schema = JSON.parse(fs.readFileSync(SCHEMA_PATH, "utf-8"));
82
+ const schema = JSON.parse(fs7.readFileSync(SCHEMA_PATH, "utf-8"));
76
83
  const ajv = new Ajv({ allErrors: true, strict: false });
77
84
  addFormats(ajv);
78
85
  const validateFn = ajv.compile(schema);
@@ -81,7 +88,7 @@ async function validate(file) {
81
88
  const d = doc;
82
89
  const pageCount = Array.isArray(d.pages) ? d.pages.length : 0;
83
90
  const elCount = Array.isArray(d.pages) ? d.pages.reduce((acc, p) => acc + (Array.isArray(p?.elements) ? p.elements.length : 0), 0) : 0;
84
- console.log(`\u2713 Valid: ${path2.basename(filePath)}`);
91
+ console.log(`\u2713 Valid: ${path7.basename(filePath)}`);
85
92
  console.log(` Format: ${d.$jdf}${bundle ? " (jdfx bundle)" : ""}`);
86
93
  console.log(` Title: ${d.meta?.title}`);
87
94
  console.log(` Pages: ${pageCount}`);
@@ -94,7 +101,7 @@ async function validate(file) {
94
101
  }
95
102
  return true;
96
103
  }
97
- console.error(`\u2717 Invalid: ${path2.basename(filePath)}`);
104
+ console.error(`\u2717 Invalid: ${path7.basename(filePath)}`);
98
105
  for (const err of validateFn.errors || []) {
99
106
  const loc = err.instancePath || "(root)";
100
107
  console.error(` ${loc} \u2014 ${err.message}`);
@@ -109,15 +116,15 @@ function hashBytes(bytes) {
109
116
  return createHash("sha1").update(bytes).digest("hex").slice(0, 16);
110
117
  }
111
118
  function rewriteResourceRefs(doc, oldKey, newKey) {
112
- function walk(els) {
119
+ function walk2(els) {
113
120
  if (!els) return;
114
121
  for (const el of els) {
115
122
  if (el?.resource === oldKey) el.resource = newKey;
116
- if (el?.elements) walk(el.elements);
117
- if (el?.children) walk(el.children);
123
+ if (el?.elements) walk2(el.elements);
124
+ if (el?.children) walk2(el.children);
118
125
  }
119
126
  }
120
- for (const page of doc.pages || []) walk(page.elements);
127
+ for (const page of doc.pages || []) walk2(page.elements);
121
128
  }
122
129
  function extractAssets(doc) {
123
130
  const assets = [];
@@ -149,10 +156,10 @@ function extractAssets(doc) {
149
156
  assets.push({ id, bytes, mimeType, ext });
150
157
  return id;
151
158
  }
152
- function walk(els) {
159
+ function walk2(els) {
153
160
  if (!els) return;
154
161
  for (const el of els) {
155
- if (el?.type === "image" && typeof el.src === "string" && el.src.startsWith("data:")) {
162
+ if ((el?.type === "image" || el?.type === "video") && typeof el.src === "string" && el.src.startsWith("data:")) {
156
163
  const m = el.src.match(/^data:([^;]+);base64,(.*)$/);
157
164
  if (m) {
158
165
  const mimeType = m[1];
@@ -162,19 +169,21 @@ function extractAssets(doc) {
162
169
  el.resource = id;
163
170
  }
164
171
  }
165
- if (el?.elements) walk(el.elements);
166
- if (el?.children) walk(el.children);
172
+ if (el?.elements) walk2(el.elements);
173
+ if (el?.children) walk2(el.children);
167
174
  }
168
175
  }
169
- for (const page of cloned.pages || []) walk(page.elements);
170
- if (cloned.resources?.images) {
171
- for (const [key, res] of Object.entries(cloned.resources.images)) {
176
+ for (const page of cloned.pages || []) walk2(page.elements);
177
+ for (const bucket of ["images", "videos"]) {
178
+ const store = cloned.resources?.[bucket];
179
+ if (!store) continue;
180
+ for (const [key, res] of Object.entries(store)) {
172
181
  if (!res || typeof res !== "object" || !("data" in res) || !res.data) continue;
173
182
  const data = String(res.data);
174
183
  const m = data.match(/^data:([^;]+);base64,(.*)$/);
175
184
  const b64 = m ? m[2] : data;
176
- const mimeType = m ? m[1] : res.mimeType || "image/png";
177
- const ext = mimeType.split("/")[1]?.replace("jpeg", "jpg") || "bin";
185
+ const mimeType = m ? m[1] : res.mimeType || (bucket === "videos" ? "video/mp4" : "image/png");
186
+ const ext = mimeType.split("/")[1]?.replace("jpeg", "jpg").replace("quicktime", "mov") || "bin";
178
187
  const bytes = decodeBase64(b64);
179
188
  const h = hashBytes(bytes);
180
189
  let canonicalId = hashToId.get(h);
@@ -188,10 +197,10 @@ function extractAssets(doc) {
188
197
  delete updated.data;
189
198
  updated.src = "embedded";
190
199
  if (canonicalId !== key) {
191
- delete cloned.resources.images[key];
200
+ delete store[key];
192
201
  rewriteResourceRefs(cloned, key, canonicalId);
193
202
  } else {
194
- cloned.resources.images[key] = updated;
203
+ store[key] = updated;
195
204
  }
196
205
  }
197
206
  }
@@ -221,21 +230,22 @@ async function packJdfx(doc) {
221
230
  return { bytes, manifest };
222
231
  }
223
232
  function shouldUseJdfx(doc) {
224
- const images = doc.resources?.images ?? {};
225
- for (const v of Object.values(images)) {
226
- if (v && typeof v === "object" && "data" in v && v.data) return true;
233
+ for (const store of [doc.resources?.images ?? {}, doc.resources?.videos ?? {}]) {
234
+ for (const v of Object.values(store)) {
235
+ if (v && typeof v === "object" && "data" in v && v.data) return true;
236
+ }
227
237
  }
228
- function walk(els) {
238
+ function walk2(els) {
229
239
  if (!els) return false;
230
240
  for (const el of els) {
231
- if (el?.type === "image" && typeof el.src === "string" && el.src.startsWith("data:")) return true;
232
- if (el?.elements && walk(el.elements)) return true;
233
- if (el?.children && walk(el.children)) return true;
241
+ if ((el?.type === "image" || el?.type === "video") && typeof el.src === "string" && el.src.startsWith("data:")) return true;
242
+ if (el?.elements && walk2(el.elements)) return true;
243
+ if (el?.children && walk2(el.children)) return true;
234
244
  }
235
245
  return false;
236
246
  }
237
247
  for (const page of doc.pages || []) {
238
- if (walk(page.elements)) return true;
248
+ if (walk2(page.elements)) return true;
239
249
  }
240
250
  return false;
241
251
  }
@@ -294,13 +304,13 @@ function stripInline(text) {
294
304
  return parseInline(text).map((r) => r.text).join("");
295
305
  }
296
306
  async function importMarkdown(inputPath, outputPath) {
297
- const input = path2.resolve(inputPath);
307
+ const input = path7.resolve(inputPath);
298
308
  console.log(`Importing: ${input}`);
299
- const content = fs.readFileSync(input, "utf-8");
300
- const doc = convertMarkdownToJdf(content, path2.basename(input, path2.extname(input)), path2.dirname(input));
309
+ const content = fs7.readFileSync(input, "utf-8");
310
+ const doc = convertMarkdownToJdf(content, path7.basename(input, path7.extname(input)), path7.dirname(input));
301
311
  let output;
302
312
  if (outputPath) {
303
- output = path2.resolve(outputPath);
313
+ output = path7.resolve(outputPath);
304
314
  } else {
305
315
  const stem = input.replace(/\.(md|markdown)$/i, "");
306
316
  output = stem + (shouldUseJdfx(doc) ? ".jdfx" : ".jdf");
@@ -308,11 +318,11 @@ async function importMarkdown(inputPath, outputPath) {
308
318
  console.log(`Output: ${output}`);
309
319
  if (output.toLowerCase().endsWith(".jdfx")) {
310
320
  const { bytes, manifest } = await packJdfx(doc);
311
- fs.writeFileSync(output, bytes);
321
+ fs7.writeFileSync(output, bytes);
312
322
  console.log(`
313
323
  Done! Created ${doc.pages.length} page(s), ${manifest.assets.length} asset(s) bundled`);
314
324
  } else {
315
- fs.writeFileSync(output, JSON.stringify(doc, null, 2));
325
+ fs7.writeFileSync(output, JSON.stringify(doc, null, 2));
316
326
  console.log(`
317
327
  Done! Created ${doc.pages.length} page(s)`);
318
328
  }
@@ -329,10 +339,10 @@ var MIME_BY_EXT2 = {
329
339
  };
330
340
  function resolveImageSrc(src, baseDir) {
331
341
  if (/^(https?:|data:|file:)/i.test(src)) return src;
332
- const abs = path2.isAbsolute(src) ? src : path2.resolve(baseDir, src);
342
+ const abs = path7.isAbsolute(src) ? src : path7.resolve(baseDir, src);
333
343
  try {
334
- const bytes = fs.readFileSync(abs);
335
- const ext = path2.extname(abs).slice(1).toLowerCase();
344
+ const bytes = fs7.readFileSync(abs);
345
+ const ext = path7.extname(abs).slice(1).toLowerCase();
336
346
  const mime = MIME_BY_EXT2[ext] || "application/octet-stream";
337
347
  return `data:${mime};base64,${bytes.toString("base64")}`;
338
348
  } catch {
@@ -613,8 +623,232 @@ function convertMarkdownToJdf(md, title, baseDir = process.cwd()) {
613
623
  };
614
624
  }
615
625
 
616
- // ../../packages/jdf-pdf-import/src/core.ts
626
+ // ../../packages/jdf-pdf-import/src/tables.ts
617
627
  var PT_TO_MM = 0.352778;
628
+ function calibrateGlyphWidth(runs) {
629
+ const ks = [];
630
+ for (const r of runs) {
631
+ const t = r.text;
632
+ if (/\s$/.test(t) || t.trim().length < 3 || r.width <= 0) continue;
633
+ ks.push(r.width / (t.length * r.fontSize * PT_TO_MM));
634
+ }
635
+ if (ks.length < 3) return 0.55;
636
+ ks.sort((a, b) => a - b);
637
+ return Math.min(0.7, Math.max(0.4, ks[Math.floor(ks.length / 2)]));
638
+ }
639
+ function hasStretchedSpaces(runs, k) {
640
+ let n = 0, wide = 0;
641
+ for (const r of runs) {
642
+ if (!/\s$/.test(r.text) || r.text.trim().length === 0) continue;
643
+ const em = r.fontSize * PT_TO_MM;
644
+ const est = r.text.trim().length * em * k + em * 0.25;
645
+ n++;
646
+ if (r.width > est * 1.4) wide++;
647
+ }
648
+ return n >= 4 && wide / n >= 0.3;
649
+ }
650
+ function textExtent(r, k, stretched) {
651
+ const em = r.fontSize * PT_TO_MM;
652
+ if (!/\s$/.test(r.text)) return Math.max(em * 0.5, r.width);
653
+ const chars = Math.max(1, r.text.trim().length);
654
+ const est = chars * em * k + em * 0.25;
655
+ if (stretched) return Math.max(em * 0.5, Math.min(r.width, est));
656
+ return Math.max(em * 0.5, r.width > est * 1.4 ? est : r.width);
657
+ }
658
+ function groupRows(runs, skip) {
659
+ const k = calibrateGlyphWidth(runs);
660
+ const stretched = hasStretchedSpaces(runs, k);
661
+ const idx = runs.map((_, i) => i).filter((i) => !skip(runs[i]) && runs[i].text.trim().length > 0);
662
+ idx.sort((a, b) => runs[a].y - runs[b].y || runs[a].x - runs[b].x);
663
+ const rows = [];
664
+ for (const i of idx) {
665
+ const r = runs[i];
666
+ const tol = Math.max(0.8, r.fontSize * PT_TO_MM * 0.35);
667
+ const last = rows[rows.length - 1];
668
+ const cell = { run: r, idx: i, x0: r.x, x1: r.x + textExtent(r, k, stretched) };
669
+ if (last && Math.abs(last.y - r.y) <= tol) {
670
+ last.cells.push(cell);
671
+ last.h = Math.max(last.h, r.height);
672
+ } else {
673
+ rows.push({ y: r.y, h: r.height, cells: [cell] });
674
+ }
675
+ }
676
+ for (const row of rows) {
677
+ row.cells.sort((a, b) => a.x0 - b.x0);
678
+ const merged = [];
679
+ for (const c of row.cells) {
680
+ const last = merged[merged.length - 1];
681
+ const em = c.run.fontSize * PT_TO_MM;
682
+ if (last && c.x0 - last.x1 <= em * 1) {
683
+ last.x1 = Math.max(last.x1, c.x1);
684
+ last.run = { ...last.run, text: `${last.run.text.replace(/\s+$/, "")} ${c.run.text.replace(/^\s+/, "")}`, width: last.x1 - last.x0 };
685
+ last.extra = [...last.extra ?? [], c.idx];
686
+ } else merged.push({ ...c });
687
+ }
688
+ row.cells = merged;
689
+ }
690
+ return rows;
691
+ }
692
+ function columnBands(rows) {
693
+ const cells = rows.flatMap((r) => r.cells);
694
+ const sorted = cells.slice().sort((a, b) => a.x0 - b.x0);
695
+ const bands = [];
696
+ for (const c of sorted) {
697
+ const last = bands[bands.length - 1];
698
+ if (last && c.x0 <= last.x1 - 0.2) {
699
+ last.x1 = Math.max(last.x1, c.x1);
700
+ last.members.push(c);
701
+ } else bands.push({ x0: c.x0, x1: c.x1, members: [c] });
702
+ }
703
+ for (const b of bands) {
704
+ const rowsSeen = /* @__PURE__ */ new Set();
705
+ for (const m of b.members) {
706
+ const row = rows.find((r) => r.cells.includes(m));
707
+ if (rowsSeen.has(row)) return null;
708
+ rowsSeen.add(row);
709
+ }
710
+ }
711
+ return bands.map(({ x0, x1 }) => ({ x0, x1 }));
712
+ }
713
+ var numeric = (s) => /^[\s$€£¥+\-−–]*[\d.,]+\s*(%|ms|s|k|m|b|M|K|B|x|×)?\s*(\/\w+)?$/i.test(s.trim()) || /^[+\-−]?\d/.test(s.trim()) && /\d$/.test(s.trim().replace(/[%)]$/, ""));
714
+ function detectTables(runs, shapes, pageWidthMm) {
715
+ const out = [];
716
+ const used = /* @__PURE__ */ new Set();
717
+ const rows = groupRows(runs, () => false);
718
+ const pageW = pageWidthMm;
719
+ let i = 0;
720
+ while (i < rows.length) {
721
+ if (rows[i].cells.length < 2) {
722
+ i++;
723
+ continue;
724
+ }
725
+ let j = i;
726
+ let cur = columnBands([rows[i]]);
727
+ let best = null;
728
+ while (j + 1 < rows.length && cur) {
729
+ const next = rows[j + 1];
730
+ const gap = next.y - (rows[j].y + rows[j].h);
731
+ const rowH = Math.max(rows[j].h, next.h);
732
+ if (gap > rowH * 2.2) break;
733
+ if (next.cells.length === 1) {
734
+ const c = next.cells[0];
735
+ const inBand = cur.findIndex((b) => c.x0 < b.x1 - 0.2 && c.x1 > b.x0 + 0.2);
736
+ const spansSeveral = cur.filter((b) => c.x0 < b.x1 - 0.2 && c.x1 > b.x0 + 0.2).length > 1;
737
+ if (inBand <= 0 || spansSeveral) break;
738
+ j++;
739
+ continue;
740
+ }
741
+ const nb = columnBands(rows.slice(i, j + 2));
742
+ if (!nb || nb.length < 2) break;
743
+ if (nb.length > cur.length && j - i >= 2) break;
744
+ cur = nb;
745
+ j++;
746
+ if (cur.length >= 2) best = { j, bands: cur };
747
+ }
748
+ const multiRows = best ? rows.slice(i, best.j + 1).filter((r) => r.cells.length >= 2).length : 0;
749
+ const blockRows = best ? rows.slice(i, best.j + 1) : [];
750
+ const bbox = blockRows.length ? {
751
+ x0: Math.min(...best.bands.map((b) => b.x0)) - 5,
752
+ x1: Math.max(...best.bands.map((b) => b.x1)) + 5,
753
+ // Cell padding puts backgrounds/borders well above the first baseline and below the last.
754
+ y0: blockRows[0].y - blockRows[0].h * 2.5,
755
+ y1: blockRows[blockRows.length - 1].y + blockRows[blockRows.length - 1].h * 3
756
+ } : null;
757
+ const gridShapes = bbox ? shapes.map((s, k) => ({ s, k })).filter(({ s }) => s.x >= bbox.x0 - 1 && s.x + s.width <= bbox.x1 + 1 && s.y >= bbox.y0 - 1 && s.y + s.height <= bbox.y1 + 1 && (s.kind === "line" || s.kind === "rect")) : [];
758
+ const hasLattice = gridShapes.length >= 3;
759
+ if (!best || multiRows < (hasLattice ? 2 : 3) || best.bands.length < 2) {
760
+ i++;
761
+ continue;
762
+ }
763
+ const bands = best.bands;
764
+ const cellText = (row, b) => row.cells.filter((c) => c.x0 < bands[b].x1 - 0.2 && c.x1 > bands[b].x0 + 0.2).map((c) => c.run.text.trim()).join(" ").trim();
765
+ const grid = [];
766
+ const lineIdx = [];
767
+ for (const row of blockRows) {
768
+ if (row.cells.length === 1 && grid.length) {
769
+ const c = row.cells[0];
770
+ const b = bands.findIndex((bb) => c.x0 < bb.x1 - 0.2 && c.x1 > bb.x0 + 0.2);
771
+ if (b > 0) {
772
+ grid[grid.length - 1][b] = (grid[grid.length - 1][b] + " " + c.run.text.trim()).trim();
773
+ lineIdx.push(c.idx, ...c.extra ?? []);
774
+ continue;
775
+ }
776
+ }
777
+ grid.push(bands.map((_, b) => cellText(row, b)));
778
+ for (const c of row.cells) {
779
+ lineIdx.push(c.idx);
780
+ for (const k of c.extra ?? []) lineIdx.push(k);
781
+ }
782
+ }
783
+ if (lineIdx.some((k) => used.has(k))) {
784
+ i = best.j + 1;
785
+ continue;
786
+ }
787
+ const first = blockRows[0];
788
+ const tableW = bbox.x1 - bbox.x0;
789
+ const rowFill = (row) => {
790
+ const cy = row.y + row.h * 0.5;
791
+ const rects = gridShapes.filter(({ s }) => s.kind === "rect" && s.fill && s.fill.toLowerCase() !== "#ffffff" && s.height >= row.h * 0.6 && s.height < row.h * 4.5 && s.y <= cy && s.y + s.height >= cy);
792
+ const covered = rects.reduce((a, { s }) => a + s.width, 0);
793
+ if (!rects.length || covered < tableW * 0.5) return null;
794
+ const counts = /* @__PURE__ */ new Map();
795
+ for (const { s } of rects) counts.set(s.fill, (counts.get(s.fill) ?? 0) + s.width);
796
+ const fill = [...counts.entries()].sort((a, b) => b[1] - a[1])[0][0];
797
+ return { fill, rects };
798
+ };
799
+ const headerBg = rowFill(first);
800
+ const firstBold = first.cells.every((c) => c.run.bold);
801
+ const bodyFills = blockRows.slice(1).map(rowFill);
802
+ const headerDistinct = !!headerBg && !bodyFills.every((f) => f?.fill === headerBg.fill);
803
+ const isHeader = headerDistinct || firstBold && !blockRows.slice(1).every((r) => r.cells.every((c) => c.run.bold));
804
+ const columns = bands.map((b, k) => {
805
+ const vals = grid.slice(isHeader ? 1 : 0).map((r) => r[k]).filter(Boolean);
806
+ const numericShare = vals.length ? vals.filter(numeric).length / vals.length : 0;
807
+ const col = { width: Math.round((b.x1 - b.x0) * 10) / 10 };
808
+ if (numericShare >= 0.7) col.align = "right";
809
+ return col;
810
+ });
811
+ const x0 = Math.max(0, bands[0].x0 - 2.5);
812
+ const x1 = Math.min(pageW, bands[bands.length - 1].x1 + 2.5);
813
+ for (let k = 0; k < bands.length; k++) {
814
+ const left = k === 0 ? x0 : (bands[k - 1].x1 + bands[k].x0) / 2;
815
+ const right = k === bands.length - 1 ? x1 : (bands[k].x1 + bands[k + 1].x0) / 2;
816
+ columns[k].width = Math.round((right - left) * 10) / 10;
817
+ }
818
+ const rowFills = (isHeader ? bodyFills : [headerBg, ...bodyFills]).map((f) => f?.fill ?? null);
819
+ const odd = rowFills.filter((_, k) => k % 2 === 1), even = rowFills.filter((_, k) => k % 2 === 0);
820
+ const altColor = odd.length && odd[0] && odd.every((f) => f === odd[0]) && even.every((f) => f !== odd[0]) ? odd[0] : void 0;
821
+ const borderShape = gridShapes.find(({ s }) => s.kind === "line" || s.kind === "rect" && (s.height < 0.6 || s.width < 0.6) && (s.fill || s.stroke));
822
+ const borderColor = borderShape ? borderShape.s.stroke || borderShape.s.fill : void 0;
823
+ const fontSize = Math.round(first.cells[0].run.fontSize * 10) / 10;
824
+ const y0 = headerBg ? Math.min(...headerBg.rects.map(({ s }) => s.y)) : first.y - first.h * 0.5;
825
+ const element = {
826
+ type: "table",
827
+ position: { x: Math.round(x0 * 100) / 100, y: Math.round(Math.max(0, y0) * 100) / 100 },
828
+ width: Math.round((x1 - x0) * 100) / 100,
829
+ columns,
830
+ rows: isHeader ? grid.slice(1) : grid,
831
+ style: { fontSize }
832
+ };
833
+ if (isHeader) {
834
+ element.headers = grid[0];
835
+ const hs = { fontWeight: "bold" };
836
+ if (headerBg) hs.backgroundColor = headerBg.fill;
837
+ const hc = first.cells[0].run.color;
838
+ if (hc && hc !== "#000000") hs.color = hc;
839
+ element.headerStyle = hs;
840
+ }
841
+ if (altColor) element.alternatingRowColor = altColor;
842
+ element.borders = borderColor ? { outer: true, inner: true, color: borderColor, width: 1 } : false;
843
+ for (const k of lineIdx) used.add(k);
844
+ out.push({ element, lineIdx, shapeIdx: gridShapes.map(({ k }) => k) });
845
+ i = best.j + 1;
846
+ }
847
+ return out;
848
+ }
849
+
850
+ // ../../packages/jdf-pdf-import/src/core.ts
851
+ var PT_TO_MM2 = 0.352778;
618
852
  function classifyFont(name) {
619
853
  const n = (name || "").toLowerCase();
620
854
  const bold = /bold|black|heavy|semibold|demibold|extrabold/.test(n);
@@ -724,19 +958,36 @@ async function walkOps(page, OPS, viewport) {
724
958
  const minY = Math.min(...ys), maxY = Math.max(...ys);
725
959
  imagePositions.push({
726
960
  name,
727
- x: minX * PT_TO_MM,
728
- y: minY * PT_TO_MM,
729
- w: (maxX - minX) * PT_TO_MM,
730
- h: (maxY - minY) * PT_TO_MM,
961
+ x: minX * PT_TO_MM2,
962
+ y: minY * PT_TO_MM2,
963
+ w: (maxX - minX) * PT_TO_MM2,
964
+ h: (maxY - minY) * PT_TO_MM2,
731
965
  inline,
732
966
  maskFill
733
967
  });
734
968
  }
735
969
  let pathSegments = [];
736
970
  let pathRect = null;
971
+ let pathRects = [];
737
972
  let pathStart = null;
738
973
  let pathLast = null;
739
974
  function flushPath(isFill, isStroke) {
975
+ for (const r of pathRects) {
976
+ const tl = toViewport(r.x, r.y + r.h);
977
+ const br = toViewport(r.x + r.w, r.y);
978
+ shapes.push({
979
+ kind: "rect",
980
+ x: Math.min(tl.x, br.x) * PT_TO_MM2,
981
+ y: Math.min(tl.y, br.y) * PT_TO_MM2,
982
+ width: Math.abs(br.x - tl.x) * PT_TO_MM2,
983
+ height: Math.abs(br.y - tl.y) * PT_TO_MM2,
984
+ fill: isFill ? gs.fill : void 0,
985
+ stroke: isStroke ? gs.stroke : void 0,
986
+ strokeWidth: isStroke ? gs.lineWidth * PT_TO_MM2 : void 0,
987
+ opacity: isFill ? gs.fillAlpha : gs.strokeAlpha
988
+ });
989
+ }
990
+ pathRects = [];
740
991
  if (pathRect) {
741
992
  const tl = toViewport(pathRect.x, pathRect.y + pathRect.h);
742
993
  const br = toViewport(pathRect.x + pathRect.w, pathRect.y);
@@ -746,13 +997,13 @@ async function walkOps(page, OPS, viewport) {
746
997
  const h = Math.abs(br.y - tl.y);
747
998
  shapes.push({
748
999
  kind: "rect",
749
- x: x * PT_TO_MM,
750
- y: y * PT_TO_MM,
751
- width: w * PT_TO_MM,
752
- height: h * PT_TO_MM,
1000
+ x: x * PT_TO_MM2,
1001
+ y: y * PT_TO_MM2,
1002
+ width: w * PT_TO_MM2,
1003
+ height: h * PT_TO_MM2,
753
1004
  fill: isFill ? gs.fill : void 0,
754
1005
  stroke: isStroke ? gs.stroke : void 0,
755
- strokeWidth: isStroke ? gs.lineWidth * PT_TO_MM : void 0,
1006
+ strokeWidth: isStroke ? gs.lineWidth * PT_TO_MM2 : void 0,
756
1007
  opacity: isFill ? gs.fillAlpha : gs.strokeAlpha
757
1008
  });
758
1009
  } else if (pathSegments.length === 2 && pathSegments[0].type === "M" && pathSegments[1].type === "L") {
@@ -764,35 +1015,35 @@ async function walkOps(page, OPS, viewport) {
764
1015
  const minY = Math.min(va.y, vb.y);
765
1016
  const maxX = Math.max(va.x, vb.x);
766
1017
  const maxY = Math.max(va.y, vb.y);
767
- const x1Local = (va.x - minX) * PT_TO_MM;
768
- const y1Local = (va.y - minY) * PT_TO_MM;
769
- const x2Local = (vb.x - minX) * PT_TO_MM;
770
- const y2Local = (vb.y - minY) * PT_TO_MM;
771
- const wLocal = Math.max(0.05, (maxX - minX) * PT_TO_MM);
772
- const hLocal = Math.max(0.05, (maxY - minY) * PT_TO_MM);
1018
+ const x1Local = (va.x - minX) * PT_TO_MM2;
1019
+ const y1Local = (va.y - minY) * PT_TO_MM2;
1020
+ const x2Local = (vb.x - minX) * PT_TO_MM2;
1021
+ const y2Local = (vb.y - minY) * PT_TO_MM2;
1022
+ const wLocal = Math.max(0.05, (maxX - minX) * PT_TO_MM2);
1023
+ const hLocal = Math.max(0.05, (maxY - minY) * PT_TO_MM2);
773
1024
  const dx = Math.abs(va.x - vb.x);
774
1025
  const dy = Math.abs(va.y - vb.y);
775
1026
  const axisAligned = dx < 0.5 || dy < 0.5;
776
1027
  if (axisAligned) {
777
1028
  shapes.push({
778
1029
  kind: "line",
779
- x: minX * PT_TO_MM,
780
- y: minY * PT_TO_MM,
1030
+ x: minX * PT_TO_MM2,
1031
+ y: minY * PT_TO_MM2,
781
1032
  width: wLocal,
782
1033
  height: hLocal,
783
1034
  stroke: isStroke ? gs.stroke : void 0,
784
- strokeWidth: isStroke ? gs.lineWidth * PT_TO_MM : void 0,
1035
+ strokeWidth: isStroke ? gs.lineWidth * PT_TO_MM2 : void 0,
785
1036
  opacity: gs.strokeAlpha
786
1037
  });
787
1038
  } else {
788
1039
  shapes.push({
789
1040
  kind: "path",
790
- x: minX * PT_TO_MM,
791
- y: minY * PT_TO_MM,
1041
+ x: minX * PT_TO_MM2,
1042
+ y: minY * PT_TO_MM2,
792
1043
  width: wLocal,
793
1044
  height: hLocal,
794
1045
  stroke: isStroke ? gs.stroke : void 0,
795
- strokeWidth: isStroke ? gs.lineWidth * PT_TO_MM : void 0,
1046
+ strokeWidth: isStroke ? gs.lineWidth * PT_TO_MM2 : void 0,
796
1047
  opacity: gs.strokeAlpha,
797
1048
  path: `M ${x1Local.toFixed(2)} ${y1Local.toFixed(2)} L ${x2Local.toFixed(2)} ${y2Local.toFixed(2)}`
798
1049
  });
@@ -825,20 +1076,20 @@ async function walkOps(page, OPS, viewport) {
825
1076
  if (seg.type === "Z") return "Z";
826
1077
  const p = [];
827
1078
  for (let i = 0; i < seg.pts.length; i += 2) {
828
- p.push(((seg.pts[i] - minX) * PT_TO_MM).toFixed(2));
829
- p.push(((seg.pts[i + 1] - minY) * PT_TO_MM).toFixed(2));
1079
+ p.push(((seg.pts[i] - minX) * PT_TO_MM2).toFixed(2));
1080
+ p.push(((seg.pts[i + 1] - minY) * PT_TO_MM2).toFixed(2));
830
1081
  }
831
1082
  return `${seg.type} ${p.join(" ")}`;
832
1083
  }).join(" ");
833
1084
  shapes.push({
834
1085
  kind: "path",
835
- x: minX * PT_TO_MM,
836
- y: minY * PT_TO_MM,
837
- width: bw * PT_TO_MM,
838
- height: bh * PT_TO_MM,
1086
+ x: minX * PT_TO_MM2,
1087
+ y: minY * PT_TO_MM2,
1088
+ width: bw * PT_TO_MM2,
1089
+ height: bh * PT_TO_MM2,
839
1090
  fill: isFill ? gs.fill : void 0,
840
1091
  stroke: isStroke ? gs.stroke : void 0,
841
- strokeWidth: isStroke ? gs.lineWidth * PT_TO_MM : void 0,
1092
+ strokeWidth: isStroke ? gs.lineWidth * PT_TO_MM2 : void 0,
842
1093
  opacity: isFill ? gs.fillAlpha : gs.strokeAlpha,
843
1094
  path: d
844
1095
  });
@@ -1000,6 +1251,12 @@ async function walkOps(page, OPS, viewport) {
1000
1251
  } else if (op === OPS.closePath) {
1001
1252
  pathSegments.push({ type: "Z", pts: [] });
1002
1253
  if (pathStart) pathLast = { ...pathStart };
1254
+ } else if (op === OPS.rectangle) {
1255
+ const x = pathArgs[ai], y = pathArgs[ai + 1], w = pathArgs[ai + 2], h = pathArgs[ai + 3];
1256
+ ai += 4;
1257
+ const p1 = tx(gs.ctm, x, y);
1258
+ const p3 = tx(gs.ctm, x + w, y + h);
1259
+ pathRects.push({ x: Math.min(p1.x, p3.x), y: Math.min(p1.y, p3.y), w: Math.abs(p3.x - p1.x), h: Math.abs(p3.y - p1.y) });
1003
1260
  }
1004
1261
  }
1005
1262
  } else if (fn === OPS.fill || fn === OPS.stroke || fn === OPS.fillStroke || fn === OPS.eoFill || fn === OPS.eoFillStroke || fn === OPS.closeFillStroke || fn === OPS.closeStroke || fn === OPS.closeEOFillStroke) {
@@ -1008,6 +1265,7 @@ async function walkOps(page, OPS, viewport) {
1008
1265
  flushPath(isFill, isStroke);
1009
1266
  } else if (fn === OPS.endPath || fn === OPS.clip || fn === OPS.eoClip) {
1010
1267
  pathSegments = [];
1268
+ pathRects = [];
1011
1269
  pathRect = null;
1012
1270
  pathStart = null;
1013
1271
  pathLast = null;
@@ -1172,10 +1430,10 @@ async function extractLinks(doc, page, viewport) {
1172
1430
  const xMax = Math.max(c1.x, c2.x);
1173
1431
  const yMax = Math.max(c1.y, c2.y);
1174
1432
  const rectMm = {
1175
- x: xMin * PT_TO_MM,
1176
- y: yMin * PT_TO_MM,
1177
- w: (xMax - xMin) * PT_TO_MM,
1178
- h: (yMax - yMin) * PT_TO_MM
1433
+ x: xMin * PT_TO_MM2,
1434
+ y: yMin * PT_TO_MM2,
1435
+ w: (xMax - xMin) * PT_TO_MM2,
1436
+ h: (yMax - yMin) * PT_TO_MM2
1179
1437
  };
1180
1438
  const url = a.url || a.unsafeUrl;
1181
1439
  const destPage = url ? void 0 : await resolveDestPage(doc, a.dest);
@@ -1213,10 +1471,10 @@ async function extractFormWidgets(page, viewport) {
1213
1471
  })).filter((o) => o.value !== "") : [];
1214
1472
  out.push({
1215
1473
  rectMm: {
1216
- x: xMin * PT_TO_MM,
1217
- y: yMin * PT_TO_MM,
1218
- w: (xMax - xMin) * PT_TO_MM,
1219
- h: (yMax - yMin) * PT_TO_MM
1474
+ x: xMin * PT_TO_MM2,
1475
+ y: yMin * PT_TO_MM2,
1476
+ w: (xMax - xMin) * PT_TO_MM2,
1477
+ h: (yMax - yMin) * PT_TO_MM2
1220
1478
  },
1221
1479
  fieldType: a.fieldType || "",
1222
1480
  fieldName: a.fieldName || `field-${out.length + 1}`,
@@ -1236,16 +1494,16 @@ async function extractFormWidgets(page, viewport) {
1236
1494
  async function flattenOutline(doc, outline) {
1237
1495
  if (!outline) return [];
1238
1496
  const out = [];
1239
- async function walk(items, depth) {
1497
+ async function walk2(items, depth) {
1240
1498
  for (const item of items) {
1241
1499
  const idx = await resolveDestPage(doc, item.dest);
1242
1500
  if (idx != null && typeof item.title === "string" && item.title.trim()) {
1243
1501
  out.push({ title: item.title.trim(), pageIndex: idx, depth });
1244
1502
  }
1245
- if (item.items?.length) await walk(item.items, depth + 1);
1503
+ if (item.items?.length) await walk2(item.items, depth + 1);
1246
1504
  }
1247
1505
  }
1248
- await walk(outline, 1);
1506
+ await walk2(outline, 1);
1249
1507
  return out;
1250
1508
  }
1251
1509
  var normTitle = (s) => s.toLowerCase().replace(/[\s\u00a0]+/g, " ").replace(/[^\p{L}\p{N} ]/gu, "").trim();
@@ -1441,6 +1699,10 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
1441
1699
  if (arr) arr.push(op);
1442
1700
  else opBins.set(key, [op]);
1443
1701
  }
1702
+ const sizePenalty = (op, fontSize) => {
1703
+ if (!op.fontSize || !fontSize) return 0;
1704
+ return Math.abs(Math.log(op.fontSize / fontSize)) * 6;
1705
+ };
1444
1706
  const findOp = (x, y, fontSize) => {
1445
1707
  let best = null;
1446
1708
  let bestD = Infinity;
@@ -1450,7 +1712,7 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
1450
1712
  const arr = opBins.get(`${bx + dx},${by + dy}`);
1451
1713
  if (!arr) continue;
1452
1714
  for (const op of arr) {
1453
- const d = Math.hypot(op.x - x, op.y - y);
1715
+ const d = Math.hypot(op.x - x, op.y - y) + sizePenalty(op, fontSize);
1454
1716
  if (d < bestD) {
1455
1717
  bestD = d;
1456
1718
  best = op;
@@ -1460,8 +1722,9 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
1460
1722
  }
1461
1723
  if (best) return best;
1462
1724
  const tol = Math.max(2, fontSize * 0.6);
1725
+ const sizeOk = (op) => !op.fontSize || !fontSize || op.fontSize / fontSize > 0.6 && op.fontSize / fontSize < 1.7;
1463
1726
  for (const op of ops.textOps) {
1464
- if (Math.abs(op.y - y) > tol) continue;
1727
+ if (!sizeOk(op) || Math.abs(op.y - y) > tol) continue;
1465
1728
  const d = Math.abs(op.x - x) + Math.abs(op.y - y) * 4;
1466
1729
  if (d < bestD) {
1467
1730
  bestD = d;
@@ -1470,6 +1733,7 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
1470
1733
  }
1471
1734
  if (best) return best;
1472
1735
  for (const op of ops.textOps) {
1736
+ if (!sizeOk(op)) continue;
1473
1737
  const d = Math.hypot(op.x - x, op.y - y);
1474
1738
  if (d < bestD) {
1475
1739
  bestD = d;
@@ -1488,6 +1752,7 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
1488
1752
  const conv = viewport.convertToViewportPoint(baseX, baseY);
1489
1753
  const vx = safeNum(conv?.[0], 0);
1490
1754
  const vy = safeNum(conv?.[1], 0);
1755
+ if (fontSize < 1.5) return;
1491
1756
  const op = findOp(vx, vy, fontSize);
1492
1757
  const mode = op?.mode ?? 0;
1493
1758
  if (mode === 7) return;
@@ -1498,12 +1763,12 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
1498
1763
  const w = safeNum(it.width, 0);
1499
1764
  runs.push({
1500
1765
  text: it.str,
1501
- x: safeNum(vx * PT_TO_MM, 0),
1502
- y: safeNum(yTop * PT_TO_MM, 0),
1766
+ x: safeNum(vx * PT_TO_MM2, 0),
1767
+ y: safeNum(yTop * PT_TO_MM2, 0),
1503
1768
  fontSize: safeNum(fontSize, 10),
1504
1769
  fontName: it.fontName,
1505
- width: safeNum(w * PT_TO_MM, 0),
1506
- height: safeNum((it.height || fontSize) * PT_TO_MM, fontSize * PT_TO_MM),
1770
+ width: safeNum(w * PT_TO_MM2, 0),
1771
+ height: safeNum((it.height || fontSize) * PT_TO_MM2, fontSize * PT_TO_MM2),
1507
1772
  color: op?.fill || "#000000",
1508
1773
  opacity: invisible ? 0 : safeNum(op?.alpha, 1)
1509
1774
  });
@@ -1511,6 +1776,12 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
1511
1776
  runs.sort((a, b) => a.y - b.y || a.x - b.x);
1512
1777
  const lines = [];
1513
1778
  const Y_TOL = 0.6;
1779
+ const kGlyph = calibrateGlyphWidth(runs);
1780
+ const stretchedSpaces = hasStretchedSpaces(runs, kGlyph);
1781
+ const fontKey = (name) => {
1782
+ const c = fontMap.get(name) || classifyFont(name || "");
1783
+ return `${c.family}|${c.weight || ""}|${c.style || ""}`;
1784
+ };
1514
1785
  for (const r of runs) {
1515
1786
  if (!r.text.length) continue;
1516
1787
  const last = lines[lines.length - 1];
@@ -1519,24 +1790,48 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
1519
1790
  continue;
1520
1791
  }
1521
1792
  const sameLine = Math.abs(last.y - r.y) <= Y_TOL;
1522
- const sameStyle = Math.abs(last.fontSize - r.fontSize) < 0.4 && last.fontName === r.fontName && last.color === r.color && Math.abs(last.opacity - r.opacity) < 0.05;
1523
- const gapMm = r.x - (last.x + last.width);
1524
- const emMm = r.fontSize * PT_TO_MM;
1525
- const mergeOk = sameLine && sameStyle && gapMm >= -0.2 && gapMm <= emMm * 0.45;
1793
+ const sameStyle = Math.abs(last.fontSize - r.fontSize) < 0.4 && (last.fontName === r.fontName || fontKey(last.fontName) === fontKey(r.fontName)) && last.color === r.color && Math.abs(last.opacity - r.opacity) < 0.05;
1794
+ const emMm = r.fontSize * PT_TO_MM2;
1795
+ const extent = (t) => {
1796
+ if (!/\s$/.test(t.text)) return t.width;
1797
+ const em = t.fontSize * PT_TO_MM2;
1798
+ const est = Math.max(1, t.text.trim().length) * em * kGlyph + em * 0.25;
1799
+ if (stretchedSpaces) return Math.min(t.width, est);
1800
+ return t.width > est * 1.4 ? est : t.width;
1801
+ };
1802
+ const gapMm = r.x - (last.x + extent(last));
1803
+ const mergeOk = sameLine && sameStyle && gapMm >= -emMm * 0.5 && gapMm <= emMm * 0.45;
1526
1804
  if (mergeOk) {
1527
1805
  const lastEndsSpace = /\s$/.test(last.text);
1528
1806
  const currStartsSpace = /^\s/.test(r.text);
1529
1807
  const sep = gapMm > emMm * 0.08 && !lastEndsSpace && !currStartsSpace ? " " : "";
1530
1808
  last.text = last.text + sep + r.text;
1531
1809
  const newExtent = r.x - last.x + r.width;
1532
- last.width = Math.max(last.width, newExtent);
1810
+ last.width = Math.max(extent(last), newExtent);
1533
1811
  } else {
1534
1812
  lines.push({ ...r });
1535
1813
  }
1536
1814
  }
1537
1815
  const elements = [];
1538
- for (const sh of ops.shapes) {
1539
- if (sh.width < 0.3 && sh.height < 0.3) continue;
1816
+ const tRuns = lines.map((l) => {
1817
+ const cls = fontMap.get(l.fontName) || classifyFont(l.fontName || "");
1818
+ return { text: l.text, x: l.x, y: l.y, width: l.width, height: l.height, fontSize: l.fontSize, fontName: l.fontName, color: l.color, bold: cls.weight === "bold" };
1819
+ });
1820
+ const detected = options.detectTables === false ? [] : detectTables(tRuns, ops.shapes, pageW * PT_TO_MM2);
1821
+ const consumedLines = /* @__PURE__ */ new Set();
1822
+ const consumedShapes = /* @__PURE__ */ new Set();
1823
+ const tableAtLine = /* @__PURE__ */ new Map();
1824
+ for (const t of detected) {
1825
+ for (const k of t.lineIdx) consumedLines.add(k);
1826
+ for (const k of t.shapeIdx) consumedShapes.add(k);
1827
+ tableAtLine.set(Math.min(...t.lineIdx), t.element);
1828
+ }
1829
+ const pageWmm = pageW * PT_TO_MM2, pageHmm = pageH * PT_TO_MM2;
1830
+ ops.shapes.forEach((sh, shapeIdx) => {
1831
+ if (consumedShapes.has(shapeIdx)) return;
1832
+ if (sh.width < 0.3 && sh.height < 0.3) return;
1833
+ if (sh.x + sh.width <= 0 || sh.y + sh.height <= 0 || sh.x >= pageWmm || sh.y >= pageHmm) return;
1834
+ if (sh.kind === "rect" && sh.fill && !sh.stroke && sh.width * sh.height >= pageWmm * pageHmm * 0.9) return;
1540
1835
  const shapeType = sh.kind;
1541
1836
  const shape = {
1542
1837
  type: "shape",
@@ -1552,7 +1847,7 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
1552
1847
  shape.style = { opacity: Math.round(sh.opacity * 100) / 100 };
1553
1848
  }
1554
1849
  elements.push(shape);
1555
- }
1850
+ });
1556
1851
  const imgs = await extractImages(page, ops.imagePositions, runtime, dataUrlCache);
1557
1852
  for (const { pos, dataUrl } of imgs) {
1558
1853
  let resourceKey = resourceKeyByName.get(pos.name);
@@ -1575,7 +1870,98 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
1575
1870
  fit: "fill"
1576
1871
  });
1577
1872
  }
1873
+ const sizeChars = /* @__PURE__ */ new Map();
1578
1874
  for (const l of lines) {
1875
+ const k = Math.round(l.fontSize * 2) / 2;
1876
+ sizeChars.set(k, (sizeChars.get(k) ?? 0) + l.text.length);
1877
+ }
1878
+ const bodyFontSize = [...sizeChars.entries()].sort((a, b) => b[1] - a[1])[0]?.[0] ?? 0;
1879
+ const rowOf = /* @__PURE__ */ new Map();
1880
+ const rowStartOf = /* @__PURE__ */ new Map();
1881
+ const nextOnRow = /* @__PURE__ */ new Map();
1882
+ {
1883
+ const order = lines.map((_, i) => i).filter((i) => !consumedLines.has(i));
1884
+ for (let a = 0; a < order.length; a++) {
1885
+ const i = order[a], li = lines[i];
1886
+ const tolY = Math.max(0.6, li.fontSize * PT_TO_MM2 * 0.35);
1887
+ let bestNext = -1, bestX = Infinity;
1888
+ for (let b = 0; b < order.length; b++) {
1889
+ const j = order[b], lj = lines[j];
1890
+ if (j === i || Math.abs(lj.y - li.y) > tolY || lj.x <= li.x) continue;
1891
+ if (lj.x < bestX) {
1892
+ bestX = lj.x;
1893
+ bestNext = j;
1894
+ }
1895
+ }
1896
+ if (bestNext >= 0) nextOnRow.set(i, bestNext);
1897
+ }
1898
+ const seen = /* @__PURE__ */ new Set();
1899
+ for (const i of order) {
1900
+ if (seen.has(i)) continue;
1901
+ const row = [i];
1902
+ seen.add(i);
1903
+ let cur = i;
1904
+ while (nextOnRow.has(cur)) {
1905
+ const j = nextOnRow.get(cur), lc = lines[cur], lj = lines[j];
1906
+ const em = Math.min(lc.fontSize, lj.fontSize) * PT_TO_MM2;
1907
+ const gap = lj.x - (lc.x + lc.width);
1908
+ if (gap < -em * 0.3 || gap > em * 0.6) break;
1909
+ row.push(j);
1910
+ seen.add(j);
1911
+ cur = j;
1912
+ }
1913
+ rowOf.set(i, row);
1914
+ for (const j of row) rowStartOf.set(j, i);
1915
+ }
1916
+ }
1917
+ const runStyle = (l) => {
1918
+ const cls = fontMap.get(l.fontName) || classifyFont(l.fontName || "");
1919
+ return { cls, bold: cls.weight === "bold", italic: cls.style === "italic" };
1920
+ };
1921
+ lines.forEach((l, lineIdx) => {
1922
+ const tableEl = tableAtLine.get(lineIdx);
1923
+ if (tableEl) elements.push(tableEl);
1924
+ if (consumedLines.has(lineIdx)) return;
1925
+ const row = rowOf.get(lineIdx);
1926
+ if (!row) return;
1927
+ if (row.length > 1) {
1928
+ const first = lines[row[0]], last = lines[row[row.length - 1]];
1929
+ const base = runStyle(first);
1930
+ const rowEnd = last.x + last.width;
1931
+ const measuredW = Math.max((rowEnd - first.x) * 1.2 + first.fontSize * PT_TO_MM2 * 0.4, first.fontSize * PT_TO_MM2);
1932
+ const nextIdx2 = nextOnRow.get(row[row.length - 1]);
1933
+ const cap2 = nextIdx2 != null ? lines[nextIdx2].x - first.x - first.fontSize * PT_TO_MM2 * 0.3 : pageWmm - first.x;
1934
+ const runs2 = [];
1935
+ row.forEach((idx, k) => {
1936
+ const r = lines[idx];
1937
+ const st = runStyle(r);
1938
+ let text2 = r.text;
1939
+ if (k > 0) {
1940
+ const prev2 = lines[row[k - 1]];
1941
+ const gap = r.x - (prev2.x + prev2.width);
1942
+ if (gap > r.fontSize * PT_TO_MM2 * 0.08 && !/\s$/.test(prev2.text) && !/^\s/.test(text2)) text2 = " " + text2;
1943
+ }
1944
+ const run = { text: text2 };
1945
+ if (st.bold) run.bold = true;
1946
+ if (st.italic) run.italic = true;
1947
+ if (r.color !== "#000000") run.color = r.color;
1948
+ if (Math.abs(r.fontSize - first.fontSize) >= 0.5) run.fontSize = Math.round(r.fontSize * 10) / 10;
1949
+ if (st.cls.family !== base.cls.family) run.fontFamily = st.cls.family;
1950
+ const lk = findLinkForRun2(r);
1951
+ if (lk) run.link = lk.url ? lk.url : lk.destPage != null ? { type: "internal", target: `#page-${lk.destPage + 1}` } : void 0;
1952
+ runs2.push(run);
1953
+ });
1954
+ const style2 = { fontSize: Math.round(first.fontSize * 10) / 10, fontFamily: base.cls.family };
1955
+ if (first.opacity < 0.999) style2.opacity = Math.round(first.opacity * 100) / 100;
1956
+ elements.push({
1957
+ type: "richtext",
1958
+ runs: runs2,
1959
+ position: { x: Math.max(0, Math.round(first.x * 100) / 100), y: Math.max(0, Math.round(Math.min(...row.map((i) => lines[i].y)) * 100) / 100) },
1960
+ width: Math.max(2, Math.round(Math.max(first.fontSize * PT_TO_MM2, Math.min(measuredW, cap2)) * 100) / 100),
1961
+ style: style2
1962
+ });
1963
+ return;
1964
+ }
1579
1965
  const cls = fontMap.get(l.fontName) || classifyFont(l.fontName || "");
1580
1966
  const style = {
1581
1967
  fontSize: Math.round(l.fontSize * 10) / 10,
@@ -1586,10 +1972,11 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
1586
1972
  if (l.color !== "#000000") style.color = l.color;
1587
1973
  if (l.opacity < 0.999) style.opacity = Math.round(l.opacity * 100) / 100;
1588
1974
  const link = findLinkForRun2(l);
1589
- const pageWmm = pageW * PT_TO_MM;
1590
- const measured = Math.max(l.width + l.fontSize * PT_TO_MM * 0.4, l.fontSize * PT_TO_MM);
1975
+ const measured = Math.max(l.width * 1.2 + l.fontSize * PT_TO_MM2 * 0.4, l.fontSize * PT_TO_MM2);
1591
1976
  const remaining = Math.max(measured, pageWmm - l.x);
1592
- const elWidth = Math.min(measured, remaining);
1977
+ const nextIdx = nextOnRow.get(lineIdx);
1978
+ const cap = nextIdx != null ? Math.max(l.fontSize * PT_TO_MM2, lines[nextIdx].x - l.x - l.fontSize * PT_TO_MM2 * 0.3) : Infinity;
1979
+ const elWidth = Math.min(measured, remaining, cap);
1593
1980
  const text = {
1594
1981
  type: "text",
1595
1982
  content: l.text,
@@ -1597,18 +1984,26 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
1597
1984
  width: Math.max(2, Math.round(elWidth * 100) / 100),
1598
1985
  style
1599
1986
  };
1600
- if (cls.weight === "bold") {
1601
- if (l.fontSize >= 22) text.heading = 1;
1602
- else if (l.fontSize >= 17) text.heading = 2;
1603
- else if (l.fontSize >= 16) text.heading = 3;
1987
+ if (cls.weight === "bold" && l.text.trim().length <= 120 && !consumedLines.has(lineIdx)) {
1988
+ const ratio = bodyFontSize > 0 ? l.fontSize / bodyFontSize : 1;
1989
+ if (l.fontSize >= 22 || ratio >= 1.8) text.heading = 1;
1990
+ else if (l.fontSize >= 17 || ratio >= 1.35) text.heading = 2;
1991
+ else if (l.fontSize >= 16 || ratio >= 1.2) text.heading = 3;
1604
1992
  }
1605
1993
  if (text.heading) text.tocEntry = text.content;
1606
1994
  if (link) {
1607
1995
  if (link.url) text.link = link.url;
1608
1996
  else if (link.destPage != null) text.link = { type: "internal", target: `#page-${link.destPage + 1}` };
1609
1997
  }
1998
+ const prev = elements[elements.length - 1];
1999
+ if (text.heading && prev && prev.type === "text" && prev.heading === text.heading && !link && !prev.link && Math.abs(prev.style?.fontSize - style.fontSize) < 0.5 && Math.abs(prev.position.x - text.position.x) < 1 && text.position.y - prev.position.y < l.fontSize * PT_TO_MM2 * 2.2 && text.position.y > prev.position.y) {
2000
+ prev.content = `${prev.content} ${text.content}`.replace(/\s+/g, " ");
2001
+ prev.tocEntry = prev.content;
2002
+ prev.width = Math.max(prev.width ?? 0, text.width ?? 0);
2003
+ return;
2004
+ }
1610
2005
  elements.push(text);
1611
- }
2006
+ });
1612
2007
  for (const w of formWidgets) {
1613
2008
  if (w.pushButton) continue;
1614
2009
  const baseEl = {
@@ -1643,7 +2038,7 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
1643
2038
  }
1644
2039
  pages.push({
1645
2040
  id: `page-${pi}`,
1646
- pageSize: { width: Math.round(pageW * PT_TO_MM * 100) / 100, height: Math.round(pageH * PT_TO_MM * 100) / 100 },
2041
+ pageSize: { width: Math.round(pageW * PT_TO_MM2 * 100) / 100, height: Math.round(pageH * PT_TO_MM2 * 100) / 100 },
1647
2042
  margins: { top: 0, right: 0, bottom: 0, left: 0 },
1648
2043
  elements
1649
2044
  });
@@ -1796,13 +2191,13 @@ async function importPdfToJdf2(source, title, options = {}) {
1796
2191
 
1797
2192
  // src/commands/import-pdf.ts
1798
2193
  async function importPdf(inputPath, outputPath, options = {}) {
1799
- const input = path2.resolve(inputPath);
1800
- if (!fs.existsSync(input)) {
2194
+ const input = path7.resolve(inputPath);
2195
+ if (!fs7.existsSync(input)) {
1801
2196
  console.error(`File not found: ${input}`);
1802
2197
  process.exit(1);
1803
2198
  }
1804
2199
  console.log(`Importing: ${input}`);
1805
- const title = path2.basename(input, path2.extname(input));
2200
+ const title = path7.basename(input, path7.extname(input));
1806
2201
  const t0 = Date.now();
1807
2202
  const doc = await importPdfToJdf2(input, title, {
1808
2203
  password: options.password,
@@ -1811,7 +2206,7 @@ async function importPdf(inputPath, outputPath, options = {}) {
1811
2206
  console.log(`Parsed in ${((Date.now() - t0) / 1e3).toFixed(1)}s \u2014 ${doc.pages.length} page(s)`);
1812
2207
  let output;
1813
2208
  if (outputPath) {
1814
- output = path2.resolve(outputPath);
2209
+ output = path7.resolve(outputPath);
1815
2210
  } else {
1816
2211
  const stem = input.replace(/\.pdf$/i, "");
1817
2212
  const wantJdfx = !options.forceJson && shouldUseJdfx(doc);
@@ -1820,11 +2215,11 @@ async function importPdf(inputPath, outputPath, options = {}) {
1820
2215
  console.log(`Output: ${output}`);
1821
2216
  if (output.toLowerCase().endsWith(".jdfx")) {
1822
2217
  const { bytes, manifest } = await packJdfx(doc);
1823
- fs.writeFileSync(output, bytes);
2218
+ fs7.writeFileSync(output, bytes);
1824
2219
  console.log(`
1825
2220
  Done! Created ${doc.pages.length} page(s), ${manifest.assets.length} asset(s) bundled`);
1826
2221
  } else {
1827
- fs.writeFileSync(output, JSON.stringify(doc, null, 2));
2222
+ fs7.writeFileSync(output, JSON.stringify(doc, null, 2));
1828
2223
  console.log(`
1829
2224
  Done! Created ${doc.pages.length} page(s)`);
1830
2225
  }
@@ -1837,23 +2232,23 @@ var ImportJsonError = class extends Error {
1837
2232
  }
1838
2233
  };
1839
2234
  async function importJson(inputPath, outputPath, options = {}) {
1840
- const input = path2.resolve(inputPath);
1841
- if (!fs.existsSync(input)) {
2235
+ const input = path7.resolve(inputPath);
2236
+ if (!fs7.existsSync(input)) {
1842
2237
  throw new ImportJsonError(`File not found: ${input}`);
1843
2238
  }
1844
2239
  console.log(`Importing: ${input}`);
1845
- const raw = fs.readFileSync(input, "utf-8");
2240
+ const raw = fs7.readFileSync(input, "utf-8");
1846
2241
  let parsed;
1847
2242
  try {
1848
2243
  parsed = JSON.parse(raw);
1849
2244
  } catch (e) {
1850
2245
  throw new ImportJsonError(`Not valid JSON: ${e.message}`);
1851
2246
  }
1852
- const title = path2.basename(input, path2.extname(input));
2247
+ const title = path7.basename(input, path7.extname(input));
1853
2248
  const doc = normaliseToJdf(parsed, title);
1854
2249
  let output;
1855
2250
  if (outputPath) {
1856
- output = path2.resolve(outputPath);
2251
+ output = path7.resolve(outputPath);
1857
2252
  } else {
1858
2253
  const stem = input.replace(/\.json$/i, "");
1859
2254
  const wantJdfx = !options.forceJson && shouldUseJdfx(doc);
@@ -1862,11 +2257,11 @@ async function importJson(inputPath, outputPath, options = {}) {
1862
2257
  console.log(`Output: ${output}`);
1863
2258
  if (output.toLowerCase().endsWith(".jdfx")) {
1864
2259
  const { bytes, manifest } = await packJdfx(doc);
1865
- fs.writeFileSync(output, bytes);
2260
+ fs7.writeFileSync(output, bytes);
1866
2261
  console.log(`
1867
2262
  Done! Created ${doc.pages.length} page(s), ${manifest.assets.length} asset(s) bundled`);
1868
2263
  } else {
1869
- fs.writeFileSync(output, JSON.stringify(doc, null, 2));
2264
+ fs7.writeFileSync(output, JSON.stringify(doc, null, 2));
1870
2265
  console.log(`
1871
2266
  Done! Created ${doc.pages.length} page(s)`);
1872
2267
  }
@@ -1942,6 +2337,47 @@ function wrapElements(elements, title, meta) {
1942
2337
  ]
1943
2338
  };
1944
2339
  }
2340
+ var DEFAULT_TRANSCRIPT_WINDOW = 45;
2341
+ var fmtTime = (sec) => {
2342
+ const s = Math.max(0, Math.round(sec));
2343
+ const h = Math.floor(s / 3600), m = Math.floor(s % 3600 / 60), r = s % 60;
2344
+ return h ? `${h}:${String(m).padStart(2, "0")}:${String(r).padStart(2, "0")}` : `${String(m).padStart(2, "0")}:${String(r).padStart(2, "0")}`;
2345
+ };
2346
+ function transcriptChunks(el, elementId2, page, crumb, windowSec, maxTokens) {
2347
+ const segs = Array.isArray(el?.transcript?.segments) ? el.transcript.segments : [];
2348
+ if (!segs.length) return [];
2349
+ const chapters = Array.isArray(el?.chapters) ? [...el.chapters].sort((a, b) => a.t - b.t) : [];
2350
+ const chapterAt = (t) => {
2351
+ let cur = null;
2352
+ for (const c of chapters) {
2353
+ if (c.t <= t + 1e-6) cur = c;
2354
+ else break;
2355
+ }
2356
+ return cur;
2357
+ };
2358
+ const out = [];
2359
+ let win = [];
2360
+ const flush = () => {
2361
+ if (!win.length) return;
2362
+ const t0 = win[0].t0, t1 = win[win.length - 1].t1;
2363
+ const body = win.map((sg) => sg.speaker ? `${sg.speaker}: ${sg.text}` : sg.text).join(" ").replace(/\s+/g, " ").trim();
2364
+ const text = `[${fmtTime(t0)}\u2013${fmtTime(t1)}] ${body}`;
2365
+ const chapter = chapterAt(t0);
2366
+ const path9 = [...crumb, ...el.title ? [String(el.title)] : [], ...chapter ? [String(chapter.title)] : []];
2367
+ out.push({ id: `${elementId2}@${Math.round(t0)}`, text, path: path9, page, types: ["video"], tokens: estimateTokens(text), hash: hashText(text), media: { element: elementId2, t0, t1 } });
2368
+ win = [];
2369
+ };
2370
+ for (const sg of segs) {
2371
+ if (typeof sg?.text !== "string" || !sg.text.trim()) continue;
2372
+ const startsNewChapter = win.length && chapterAt(sg.t0) !== chapterAt(win[0].t0);
2373
+ const spansWindow = win.length && sg.t1 - win[0].t0 > windowSec;
2374
+ const overBudget = win.length && estimateTokens(win.map((w) => w.text).join(" ") + sg.text) > maxTokens;
2375
+ if (startsNewChapter || spansWindow || overBudget) flush();
2376
+ win.push(sg);
2377
+ }
2378
+ flush();
2379
+ return out;
2380
+ }
1945
2381
  var DEFAULT_MAX_TOKENS = 512;
1946
2382
  function estimateTokens(text) {
1947
2383
  return Math.ceil(text.length / 4);
@@ -1964,11 +2400,11 @@ function serializeElement(el) {
1964
2400
  case "richtext":
1965
2401
  return (e.runs || []).map((r) => r.text ?? "").join("").trim();
1966
2402
  case "list": {
1967
- const walk = (items, depth = 0) => (items || []).flatMap((it) => {
2403
+ const walk2 = (items, depth = 0) => (items || []).flatMap((it) => {
1968
2404
  const line = " ".repeat(depth) + "- " + String(it.content ?? "").trim();
1969
- return it.children?.length ? [line, ...walk(it.children, depth + 1)] : [line];
2405
+ return it.children?.length ? [line, ...walk2(it.children, depth + 1)] : [line];
1970
2406
  });
1971
- return walk(e.items).join("\n");
2407
+ return walk2(e.items).join("\n");
1972
2408
  }
1973
2409
  case "table": {
1974
2410
  const headers = e.headers ?? e.columns?.map((c) => c.header || "").filter((h) => h) ?? [];
@@ -1999,6 +2435,8 @@ function serializeElement(el) {
1999
2435
  return `${e.checked ? "[x]" : "[ ]"} ${e.label ?? ""}`.trim();
2000
2436
  case "image":
2001
2437
  return e.alt ? `[image: ${e.alt}]` : "";
2438
+ case "video":
2439
+ return e.title ? `[video: ${e.title}]` : "";
2002
2440
  case "toc":
2003
2441
  case "shape":
2004
2442
  case "signature":
@@ -2038,8 +2476,15 @@ function makeChunk(group, breadcrumb) {
2038
2476
  function chunkDocument(doc, options = {}) {
2039
2477
  const strategy = options.strategy ?? "section";
2040
2478
  const maxTokens = options.maxTokens ?? DEFAULT_MAX_TOKENS;
2479
+ const windowSec = options.transcriptWindowSec ?? DEFAULT_TRANSCRIPT_WINDOW;
2041
2480
  const flat = flatten(doc);
2042
2481
  const chunks = [];
2482
+ const withTranscripts = (group, crumb2, c) => {
2483
+ if (c) chunks.push(c);
2484
+ for (const f of group) {
2485
+ if (f.el.type === "video") chunks.push(...transcriptChunks(f.el, f.id, f.page, crumb2, windowSec, maxTokens));
2486
+ }
2487
+ };
2043
2488
  if (strategy === "element") {
2044
2489
  const crumb2 = [];
2045
2490
  for (const f of flat) {
@@ -2048,8 +2493,7 @@ function chunkDocument(doc, options = {}) {
2048
2493
  crumb2.length = Math.max(0, lvl - 1);
2049
2494
  crumb2[lvl - 1] = serializeElement(f.el);
2050
2495
  }
2051
- const c = makeChunk([f], crumb2);
2052
- if (c) chunks.push(c);
2496
+ withTranscripts([f], crumb2, makeChunk([f], crumb2));
2053
2497
  }
2054
2498
  return chunks;
2055
2499
  }
@@ -2058,8 +2502,7 @@ function chunkDocument(doc, options = {}) {
2058
2502
  let buf2 = [];
2059
2503
  let bufTokens = 0;
2060
2504
  const flush = () => {
2061
- const c = makeChunk(buf2, crumb2);
2062
- if (c) chunks.push(c);
2505
+ withTranscripts(buf2, crumb2, makeChunk(buf2, crumb2));
2063
2506
  buf2 = [];
2064
2507
  bufTokens = 0;
2065
2508
  };
@@ -2086,22 +2529,21 @@ function chunkDocument(doc, options = {}) {
2086
2529
  for (const f of buf) {
2087
2530
  const t = estimateTokens(serializeElement(f.el));
2088
2531
  if (subTokens + t > maxTokens && sub.length > 0) {
2089
- const c2 = makeChunk(sub, crumb);
2090
- if (c2) chunks.push(c2);
2532
+ withTranscripts(sub, crumb, makeChunk(sub, crumb));
2091
2533
  sub = [];
2092
2534
  subTokens = 0;
2093
2535
  }
2094
2536
  sub.push(f);
2095
2537
  subTokens += t;
2096
2538
  }
2097
- const c = makeChunk(sub, crumb);
2098
- if (c) chunks.push(c);
2539
+ withTranscripts(sub, crumb, makeChunk(sub, crumb));
2099
2540
  buf = [];
2100
2541
  };
2101
2542
  for (const f of flat) {
2102
2543
  const lvl = headingLevel(f.el);
2103
2544
  if (lvl != null) {
2104
- flushSection();
2545
+ const onlyHeadings = buf.length > 0 && buf.every((b) => headingLevel(b.el) != null);
2546
+ if (!onlyHeadings) flushSection();
2105
2547
  crumb.length = Math.max(0, lvl - 1);
2106
2548
  crumb[lvl - 1] = serializeElement(f.el);
2107
2549
  }
@@ -2112,34 +2554,34 @@ function chunkDocument(doc, options = {}) {
2112
2554
  }
2113
2555
  async function loadJdf(filePath) {
2114
2556
  if (filePath.toLowerCase().endsWith(".jdfx")) {
2115
- const zip = await JSZip.loadAsync(fs.readFileSync(filePath));
2557
+ const zip = await JSZip.loadAsync(fs7.readFileSync(filePath));
2116
2558
  const docFile = zip.file(JDFX_DOCUMENT_PATH);
2117
2559
  if (!docFile) throw new Error(`Bundle missing ${JDFX_DOCUMENT_PATH}`);
2118
2560
  return JSON.parse(await docFile.async("string"));
2119
2561
  }
2120
- return JSON.parse(fs.readFileSync(filePath, "utf-8"));
2562
+ return JSON.parse(fs7.readFileSync(filePath, "utf-8"));
2121
2563
  }
2122
2564
  async function chunkFile(inputPath, opts = {}) {
2123
- const input = path2.resolve(inputPath);
2124
- if (!fs.existsSync(input)) throw new Error(`File not found: ${input}`);
2565
+ const input = path7.resolve(inputPath);
2566
+ if (!fs7.existsSync(input)) throw new Error(`File not found: ${input}`);
2125
2567
  const doc = await loadJdf(input);
2126
2568
  const strategy = opts.strategy ?? "section";
2127
- const chunks = chunkDocument(doc, { strategy, maxTokens: opts.maxTokens });
2569
+ const chunks = chunkDocument(doc, { strategy, maxTokens: opts.maxTokens, transcriptWindowSec: opts.transcriptWindowSec });
2128
2570
  const format = opts.format ?? "jsonl";
2129
2571
  console.log(`Chunking: ${input}`);
2130
2572
  console.log(`Strategy: ${strategy}${opts.maxTokens ? ` (max ${opts.maxTokens} tokens)` : ""}`);
2131
2573
  if (format === "inline") {
2132
- const out = opts.output ? path2.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".jdf");
2574
+ const out = opts.output ? path7.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".jdf");
2133
2575
  const withIndex = { ...doc, index: { chunker: `jdf-${strategy}-v1`, chunks } };
2134
- fs.writeFileSync(out, JSON.stringify(withIndex, null, 2));
2576
+ fs7.writeFileSync(out, JSON.stringify(withIndex, null, 2));
2135
2577
  console.log(`Output: ${out} (${chunks.length} chunks in "index" block)`);
2136
2578
  } else if (format === "json") {
2137
- const out = opts.output ? path2.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".chunks.json");
2138
- fs.writeFileSync(out, JSON.stringify(chunks, null, 2));
2579
+ const out = opts.output ? path7.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".chunks.json");
2580
+ fs7.writeFileSync(out, JSON.stringify(chunks, null, 2));
2139
2581
  console.log(`Output: ${out} (${chunks.length} chunks)`);
2140
2582
  } else {
2141
- const out = opts.output ? path2.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".chunks.jsonl");
2142
- fs.writeFileSync(out, chunks.map((c) => JSON.stringify(c)).join("\n") + "\n");
2583
+ const out = opts.output ? path7.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".chunks.jsonl");
2584
+ fs7.writeFileSync(out, chunks.map((c) => JSON.stringify(c)).join("\n") + "\n");
2143
2585
  console.log(`Output: ${out} (${chunks.length} chunks)`);
2144
2586
  }
2145
2587
  const totalTokens = chunks.reduce((a, c) => a + c.tokens, 0);
@@ -2311,17 +2753,17 @@ async function embedOpenAI(model, inputs) {
2311
2753
  }
2312
2754
  async function loadJdf2(filePath) {
2313
2755
  if (filePath.toLowerCase().endsWith(".jdfx")) {
2314
- const zip = await JSZip.loadAsync(fs.readFileSync(filePath));
2756
+ const zip = await JSZip.loadAsync(fs7.readFileSync(filePath));
2315
2757
  const docFile = zip.file(JDFX_DOCUMENT_PATH);
2316
2758
  if (!docFile) throw new Error(`Bundle missing ${JDFX_DOCUMENT_PATH}`);
2317
2759
  return JSON.parse(await docFile.async("string"));
2318
2760
  }
2319
- return JSON.parse(fs.readFileSync(filePath, "utf-8"));
2761
+ return JSON.parse(fs7.readFileSync(filePath, "utf-8"));
2320
2762
  }
2321
2763
  function loadCache(cachePath) {
2322
2764
  try {
2323
- if (!fs.existsSync(cachePath)) return null;
2324
- return JSON.parse(fs.readFileSync(cachePath, "utf-8"));
2765
+ if (!fs7.existsSync(cachePath)) return null;
2766
+ return JSON.parse(fs7.readFileSync(cachePath, "utf-8"));
2325
2767
  } catch {
2326
2768
  return null;
2327
2769
  }
@@ -2332,15 +2774,15 @@ function batched(items, size) {
2332
2774
  return out;
2333
2775
  }
2334
2776
  async function embedFile(inputPath, opts = {}) {
2335
- const input = path2.resolve(inputPath);
2336
- if (!fs.existsSync(input)) throw new Error(`File not found: ${input}`);
2777
+ const input = path7.resolve(inputPath);
2778
+ if (!fs7.existsSync(input)) throw new Error(`File not found: ${input}`);
2337
2779
  const provider = opts.provider ?? "ollama";
2338
2780
  const model = opts.model ?? DEFAULT_MODEL[provider];
2339
2781
  const strategy = opts.strategy ?? "section";
2340
2782
  const doc = await loadJdf2(input);
2341
- const chunks = chunkDocument(doc, { strategy, maxTokens: opts.maxTokens });
2342
- const output = opts.output ? path2.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".embeddings.json");
2343
- const cachePath = opts.cache ? path2.resolve(opts.cache) : output;
2783
+ const chunks = chunkDocument(doc, { strategy, maxTokens: opts.maxTokens, transcriptWindowSec: opts.transcriptWindowSec });
2784
+ const output = opts.output ? path7.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".embeddings.json");
2785
+ const cachePath = opts.cache ? path7.resolve(opts.cache) : output;
2344
2786
  console.log(`Embedding: ${input}`);
2345
2787
  console.log(`Provider: ${provider} / ${model}${provider === "ollama" ? " (local \u2014 no data leaves this machine)" : " (remote API)"}`);
2346
2788
  console.log(`Strategy: ${strategy} \u2192 ${chunks.length} chunks`);
@@ -2379,11 +2821,295 @@ async function embedFile(inputPath, opts = {}) {
2379
2821
  chunker: `jdf-${strategy}-v1`,
2380
2822
  vectors
2381
2823
  };
2382
- fs.writeFileSync(output, JSON.stringify(sidecar));
2824
+ fs7.writeFileSync(output, JSON.stringify(sidecar));
2383
2825
  console.log(`
2384
2826
  Done! ${Object.keys(vectors).length} vectors (${dims}-dim) \u2192 ${output}`);
2385
2827
  return sidecar;
2386
2828
  }
2829
+ var toSec = (ts) => {
2830
+ const m = ts.trim().replace(",", ".").match(/^(?:(\d+):)?(\d{1,2}):(\d{2}(?:\.\d+)?)$/);
2831
+ if (!m) throw new Error(`bad timestamp "${ts}"`);
2832
+ return (m[1] ? Number(m[1]) * 3600 : 0) + Number(m[2]) * 60 + Number(m[3]);
2833
+ };
2834
+ function parseSubtitles(text, filename = "") {
2835
+ const trimmed = text.replace(/^/, "").trim();
2836
+ if (trimmed.startsWith("{") || trimmed.startsWith("[")) {
2837
+ const j = JSON.parse(trimmed);
2838
+ const arr = Array.isArray(j) ? j : Array.isArray(j.segments) ? j.segments : [];
2839
+ return arr.map((sg) => ({
2840
+ t0: Number(sg.t0 ?? sg.start ?? sg.from ?? 0),
2841
+ t1: Number(sg.t1 ?? sg.end ?? sg.to ?? 0),
2842
+ text: String(sg.text ?? "").trim(),
2843
+ ...sg.speaker ? { speaker: String(sg.speaker) } : {}
2844
+ })).filter((sg) => sg.text);
2845
+ }
2846
+ const segs = [];
2847
+ for (const block of trimmed.split(/\r?\n\r?\n+/)) {
2848
+ const lines = block.split(/\r?\n/).filter((l) => l.trim() !== "" && l.trim() !== "WEBVTT");
2849
+ const ti = lines.findIndex((l) => l.includes("-->"));
2850
+ if (ti < 0) continue;
2851
+ const [a, b] = lines[ti].split("-->").map((x) => x.trim().split(/\s+/)[0]);
2852
+ const body = lines.slice(ti + 1).join(" ").replace(/<[^>]+>/g, "").replace(/\s+/g, " ").trim();
2853
+ if (!body) continue;
2854
+ segs.push({ t0: toSec(a), t1: toSec(b), text: body });
2855
+ }
2856
+ if (!segs.length) throw new Error(`no cues found in ${filename || "subtitle input"} (expected SRT, WebVTT or JSON segments)`);
2857
+ return segs;
2858
+ }
2859
+ function parseChapters(text) {
2860
+ const t = text.trim();
2861
+ if (t.startsWith("[")) return JSON.parse(t).map((c) => ({ t: Number(c.t ?? c.start ?? 0), title: String(c.title ?? "") }));
2862
+ return t.split(/\r?\n/).map((l) => l.trim()).filter(Boolean).map((l) => {
2863
+ const m = l.match(/^(\S+)\s+(.+)$/);
2864
+ if (!m) throw new Error(`bad chapter line "${l}" (expected "mm:ss Title")`);
2865
+ return { t: toSec(m[1]), title: m[2].trim() };
2866
+ });
2867
+ }
2868
+ async function loadDoc(file) {
2869
+ if (file.toLowerCase().endsWith(".jdfx")) {
2870
+ const zip = await JSZip.loadAsync(fs7.readFileSync(file));
2871
+ const f = zip.file(JDFX_DOCUMENT_PATH);
2872
+ if (!f) throw new Error(`Bundle missing ${JDFX_DOCUMENT_PATH}`);
2873
+ const doc = JSON.parse(await f.async("string"));
2874
+ const manifest = zip.file("manifest.json") ? JSON.parse(await zip.file("manifest.json").async("string")) : { assets: [] };
2875
+ for (const a of manifest.assets ?? []) {
2876
+ const af = zip.file(a.path);
2877
+ if (!af) continue;
2878
+ const data = (await af.async("nodebuffer")).toString("base64");
2879
+ const res = { src: "embedded", mimeType: a.mimeType, data };
2880
+ doc.resources ??= {};
2881
+ if (/^video\//i.test(a.mimeType || "")) (doc.resources.videos ??= {})[a.id] = res;
2882
+ else (doc.resources.images ??= {})[a.id] = res;
2883
+ }
2884
+ return { doc, bundle: true, zip };
2885
+ }
2886
+ return { doc: JSON.parse(fs7.readFileSync(file, "utf-8")), bundle: false };
2887
+ }
2888
+ function findVideos(doc) {
2889
+ const out = [];
2890
+ const walk2 = (els, page) => {
2891
+ for (const el of els ?? []) {
2892
+ if (el?.type === "video") out.push({ el, page, index: out.length });
2893
+ if (el?.elements) walk2(el.elements, page);
2894
+ }
2895
+ };
2896
+ doc.pages.forEach((p, i) => walk2(p.elements, i + 1));
2897
+ return out;
2898
+ }
2899
+ async function clipToTempFile(doc, el, docDir) {
2900
+ const tmp = path7.join(fs7.mkdtempSync(path7.join(__require("os").tmpdir(), "jdf-transcribe-")), "clip.mp4");
2901
+ const res = el.resource ? doc.resources?.videos?.[el.resource] ?? doc.resources?.images?.[el.resource] : void 0;
2902
+ if (res?.data) {
2903
+ fs7.writeFileSync(tmp, Buffer.from(res.data.replace(/^data:[^,]*,/, ""), "base64"));
2904
+ return tmp;
2905
+ }
2906
+ if (res?.path) return path7.resolve(docDir, res.path);
2907
+ const src = el.src;
2908
+ if (!src) return null;
2909
+ if (src.startsWith("data:")) {
2910
+ fs7.writeFileSync(tmp, Buffer.from(src.replace(/^data:[^,]*,/, ""), "base64"));
2911
+ return tmp;
2912
+ }
2913
+ if (/^https?:\/\//i.test(src)) {
2914
+ const r = await fetch(src);
2915
+ if (!r.ok) throw new Error(`download failed ${r.status}: ${src}`);
2916
+ fs7.writeFileSync(tmp, Buffer.from(await r.arrayBuffer()));
2917
+ return tmp;
2918
+ }
2919
+ const local = path7.resolve(docDir, src);
2920
+ return fs7.existsSync(local) ? local : null;
2921
+ }
2922
+ function whisperCli(clip, model, language, prompt2) {
2923
+ const ffmpeg = spawnSync("ffmpeg", ["-version"]);
2924
+ if (ffmpeg.error) throw new Error("ffmpeg not found \u2014 needed to extract audio for whisper-cli (brew install ffmpeg)");
2925
+ const wav = clip.replace(/\.[^.]+$/, "") + ".16k.wav";
2926
+ const ex = spawnSync("ffmpeg", ["-y", "-i", clip, "-vn", "-ac", "1", "-ar", "16000", "-f", "wav", wav], { encoding: "utf-8" });
2927
+ if (ex.status !== 0) throw new Error(`ffmpeg failed: ${ex.stderr.slice(-400)}`);
2928
+ const args = ["-f", wav, "-oj", "-of", wav.replace(/\.wav$/, "")];
2929
+ if (model) args.push("-m", model);
2930
+ if (language) args.push("-l", language);
2931
+ if (prompt2) args.push("--prompt", prompt2);
2932
+ const run = spawnSync("whisper-cli", args, { encoding: "utf-8" });
2933
+ if (run.error) throw new Error("whisper-cli not found \u2014 install whisper.cpp (brew install whisper-cpp) or use --from / --provider openai");
2934
+ if (run.status !== 0) throw new Error(`whisper-cli failed: ${run.stderr.slice(-400)}`);
2935
+ const j = JSON.parse(fs7.readFileSync(wav.replace(/\.wav$/, "") + ".json", "utf-8"));
2936
+ const segs = j.transcription ?? j.segments ?? [];
2937
+ const ms = (x) => typeof x === "number" ? x / 1e3 : toSec(String(x).replace(",", "."));
2938
+ return segs.map((sg) => ({ t0: ms(sg.offsets?.from ?? sg.start), t1: ms(sg.offsets?.to ?? sg.end), text: String(sg.text ?? "").trim() })).filter((sg) => sg.text);
2939
+ }
2940
+ async function openaiTranscribe(clip, model, language, prompt2) {
2941
+ const key = process.env.OPENAI_API_KEY;
2942
+ if (!key) throw new Error("OPENAI_API_KEY is not set");
2943
+ const base = process.env.OPENAI_BASE_URL || "https://api.openai.com/v1";
2944
+ const form = new FormData();
2945
+ form.append("file", new Blob([fs7.readFileSync(clip)]), path7.basename(clip));
2946
+ form.append("model", model || "whisper-1");
2947
+ form.append("response_format", "verbose_json");
2948
+ form.append("timestamp_granularities[]", "segment");
2949
+ if (language) form.append("language", language);
2950
+ if (prompt2) form.append("prompt", prompt2);
2951
+ const r = await fetch(`${base}/audio/transcriptions`, { method: "POST", headers: { Authorization: `Bearer ${key}` }, body: form });
2952
+ if (!r.ok) throw new Error(`OpenAI transcription failed ${r.status}: ${(await r.text()).slice(0, 300)}`);
2953
+ const j = await r.json();
2954
+ return (j.segments ?? []).map((sg) => ({ t0: Number(sg.start), t1: Number(sg.end), text: String(sg.text).trim() })).filter((sg) => sg.text);
2955
+ }
2956
+ async function transcribeFile(inputPath, opts = {}) {
2957
+ const input = path7.resolve(inputPath);
2958
+ if (!fs7.existsSync(input)) throw new Error(`File not found: ${input}`);
2959
+ const { doc, bundle } = await loadDoc(input);
2960
+ const videos = findVideos(doc);
2961
+ if (!videos.length) throw new Error("document has no video element");
2962
+ let target = videos[0];
2963
+ if (opts.element != null) {
2964
+ const byId = videos.find((v) => v.el.id === opts.element);
2965
+ const byIdx = /^\d+$/.test(opts.element) ? videos[Number(opts.element)] : void 0;
2966
+ target = byId ?? byIdx ?? (() => {
2967
+ throw new Error(`no video element "${opts.element}" (have: ${videos.map((v) => v.el.id ?? `#${v.index}`).join(", ")})`);
2968
+ })();
2969
+ } else if (videos.length > 1) {
2970
+ throw new Error(`document has ${videos.length} videos \u2014 pick one with --element <id|index>`);
2971
+ }
2972
+ let segments;
2973
+ let source;
2974
+ if (opts.from) {
2975
+ segments = parseSubtitles(fs7.readFileSync(path7.resolve(opts.from), "utf-8"), opts.from);
2976
+ source = `${path7.extname(opts.from).slice(1).toLowerCase() || "file"}-import`;
2977
+ } else {
2978
+ const provider = opts.provider ?? "whisper-cli";
2979
+ const clip = await clipToTempFile(doc, target.el, path7.dirname(input));
2980
+ if (!clip) throw new Error("could not locate the clip bytes (no bundled asset, data URL, local path or http URL) \u2014 use --from to import subtitles instead");
2981
+ segments = provider === "openai" ? await openaiTranscribe(clip, opts.model, opts.language, opts.prompt) : whisperCli(clip, opts.model, opts.language, opts.prompt);
2982
+ source = provider === "openai" ? `openai:${opts.model || "whisper-1"}` : `whisper-cli${opts.model ? ":" + path7.basename(opts.model) : ""}`;
2983
+ }
2984
+ segments.sort((a, b) => a.t0 - b.t0);
2985
+ const transcript = { ...opts.language ? { language: opts.language } : {}, source, created: (/* @__PURE__ */ new Date()).toISOString(), segments };
2986
+ target.el.transcript = transcript;
2987
+ if (opts.chapters) target.el.chapters = parseChapters(fs7.readFileSync(path7.resolve(opts.chapters), "utf-8"));
2988
+ if (!target.el.id) target.el.id = `video-${target.index + 1}`;
2989
+ const output = opts.output ? path7.resolve(opts.output) : input;
2990
+ if (output.toLowerCase().endsWith(".jdfx") || bundle && !opts.output) {
2991
+ const { bytes } = await packJdfx(doc);
2992
+ fs7.writeFileSync(output, bytes);
2993
+ } else {
2994
+ fs7.writeFileSync(output, JSON.stringify(doc, null, 2));
2995
+ }
2996
+ const dur = segments.length ? segments[segments.length - 1].t1 : 0;
2997
+ console.log(`Transcribed: ${path7.basename(input)} \u2192 element "${target.el.id}" (${segments.length} segments, ${Math.round(dur)} s, source ${source})`);
2998
+ if (target.el.chapters) console.log(`Chapters: ${target.el.chapters.length}`);
2999
+ console.log(`Output: ${output}
3000
+ Next: jdf chunk ${path7.basename(output)} # transcript \u2192 time-windowed chunks with media.t0/t1`);
3001
+ return transcript;
3002
+ }
3003
+ var CONFIG_NAME = "jdf.rag.json";
3004
+ var OUT_DIR = ".jdf-rag";
3005
+ function walk(dir, acc = []) {
3006
+ for (const ent of fs7.readdirSync(dir, { withFileTypes: true })) {
3007
+ if (ent.name === "node_modules" || ent.name === OUT_DIR || ent.name.startsWith(".")) continue;
3008
+ const p = path7.join(dir, ent.name);
3009
+ if (ent.isDirectory()) walk(p, acc);
3010
+ else if (/\.(jdf|jdfx)$/i.test(ent.name)) acc.push(p);
3011
+ }
3012
+ return acc.sort();
3013
+ }
3014
+ async function readDoc(file) {
3015
+ if (file.toLowerCase().endsWith(".jdfx")) {
3016
+ const zip = await JSZip.loadAsync(fs7.readFileSync(file));
3017
+ return JSON.parse(await zip.file(JDFX_DOCUMENT_PATH).async("string"));
3018
+ }
3019
+ return JSON.parse(fs7.readFileSync(file, "utf-8"));
3020
+ }
3021
+ function videosIn(doc) {
3022
+ const out = [];
3023
+ const w = (els) => {
3024
+ for (const el of els ?? []) {
3025
+ if (el?.type === "video") out.push({ id: el.id, hasTranscript: !!el.transcript?.segments?.length });
3026
+ if (el?.elements) w(el.elements);
3027
+ }
3028
+ };
3029
+ for (const p of doc.pages ?? []) w(p.elements);
3030
+ return out;
3031
+ }
3032
+ async function ragFolder(dirPath, cli = {}) {
3033
+ const dir = path7.resolve(dirPath);
3034
+ if (!fs7.existsSync(dir) || !fs7.statSync(dir).isDirectory()) throw new Error(`Not a directory: ${dir}`);
3035
+ const cfgPath = path7.join(dir, CONFIG_NAME);
3036
+ const cfg = fs7.existsSync(cfgPath) ? JSON.parse(fs7.readFileSync(cfgPath, "utf-8")) : {};
3037
+ const opts = { ...cfg, ...Object.fromEntries(Object.entries(cli).filter(([, v]) => v !== void 0)) };
3038
+ const provider = opts.provider ?? "ollama";
3039
+ const transcribe = opts.transcribe ?? "none";
3040
+ const outDir = path7.resolve(opts.out ?? path7.join(dir, OUT_DIR));
3041
+ const files = walk(dir);
3042
+ console.log(`jdf rag: ${dir}
3043
+ files: ${files.length} (.jdf/.jdfx)${fs7.existsSync(cfgPath) ? `
3044
+ config: ${CONFIG_NAME}` : ""}
3045
+ embeddings: ${opts.noEmbed ? "skipped (--no-embed)" : `${provider}${opts.model ? " / " + opts.model : ""}`}
3046
+ transcribe: ${transcribe}${opts.dryRun ? "\n DRY RUN \u2014 nothing written" : ""}
3047
+ `);
3048
+ if (!files.length) {
3049
+ console.log("Nothing to do.");
3050
+ return;
3051
+ }
3052
+ const manifest = { dir, created: (/* @__PURE__ */ new Date()).toISOString(), provider: opts.noEmbed ? null : provider, model: opts.model ?? null, strategy: opts.strategy ?? "section", transcribe, files: [], totals: { files: files.length, chunks: 0, videoChunks: 0, videos: 0, transcribed: 0, untranscribed: 0 } };
3053
+ const indexLines = [];
3054
+ for (const file of files) {
3055
+ const rel = path7.relative(dir, file);
3056
+ const doc = await readDoc(file);
3057
+ const vids = videosIn(doc);
3058
+ let transcribedHere = 0;
3059
+ for (const v of vids) {
3060
+ if (v.hasTranscript) continue;
3061
+ if (transcribe === "none") {
3062
+ manifest.totals.untranscribed++;
3063
+ continue;
3064
+ }
3065
+ if (opts.dryRun) {
3066
+ transcribedHere++;
3067
+ continue;
3068
+ }
3069
+ try {
3070
+ await transcribeFile(file, { provider: transcribe, element: v.id, model: opts.transcribeModel, language: opts.language, prompt: opts.prompt });
3071
+ transcribedHere++;
3072
+ } catch (e) {
3073
+ console.warn(` ! ${rel}: transcription failed for video ${v.id ?? "#?"}: ${e.message}`);
3074
+ manifest.totals.untranscribed++;
3075
+ }
3076
+ }
3077
+ manifest.totals.videos += vids.length;
3078
+ manifest.totals.transcribed += transcribedHere;
3079
+ if (opts.dryRun) {
3080
+ manifest.files.push({ file: rel, videos: vids.length, wouldTranscribe: transcribedHere });
3081
+ continue;
3082
+ }
3083
+ const chunkOpts = { strategy: opts.strategy, maxTokens: opts.maxTokens, transcriptWindowSec: opts.transcriptWindowSec };
3084
+ const chunkOut = path7.join(outDir, "chunks", rel.replace(/\.(jdf|jdfx)$/i, ".chunks.jsonl"));
3085
+ fs7.mkdirSync(path7.dirname(chunkOut), { recursive: true });
3086
+ let chunks;
3087
+ if (opts.noEmbed) {
3088
+ chunks = await chunkFile(file, { ...chunkOpts, format: "jsonl", output: chunkOut });
3089
+ } else {
3090
+ const side = await embedFile(file, { ...chunkOpts, provider, model: opts.model, incremental: true });
3091
+ chunks = await chunkFile(file, { ...chunkOpts, format: "jsonl", output: chunkOut });
3092
+ manifest.files.push({ file: rel, chunks: chunks.length, vectors: Object.keys(side.vectors).length, sidecar: path7.relative(dir, file.replace(/\.(jdf|jdfx)$/i, ".embeddings.json")), videos: vids.length, transcribed: transcribedHere });
3093
+ }
3094
+ if (opts.noEmbed) manifest.files.push({ file: rel, chunks: chunks.length, videos: vids.length, transcribed: transcribedHere });
3095
+ for (const c of chunks) {
3096
+ indexLines.push(JSON.stringify({ file: rel, ...c }));
3097
+ manifest.totals.chunks++;
3098
+ if (c.media) manifest.totals.videoChunks++;
3099
+ }
3100
+ }
3101
+ if (!opts.dryRun) {
3102
+ fs7.mkdirSync(outDir, { recursive: true });
3103
+ fs7.writeFileSync(path7.join(outDir, "index.jsonl"), indexLines.join("\n") + (indexLines.length ? "\n" : ""));
3104
+ fs7.writeFileSync(path7.join(outDir, "manifest.json"), JSON.stringify(manifest, null, 2) + "\n");
3105
+ }
3106
+ const t = manifest.totals;
3107
+ console.log(`
3108
+ Done. ${t.files} files \u2192 ${t.chunks} chunks (${t.videoChunks} from video transcripts); videos ${t.videos}, transcribed now ${t.transcribed}, still without transcript ${t.untranscribed}${t.untranscribed && transcribe === "none" ? " (pass --transcribe whisper-cli|openai, or jdf transcribe --from subs.srt)" : ""}.`);
3109
+ if (!opts.dryRun) console.log(`Index: ${path7.join(outDir, "index.jsonl")}
3110
+ Report: ${path7.join(outDir, "manifest.json")}${opts.noEmbed ? "" : `
3111
+ Vectors: one <file>.embeddings.json next to each document (incremental \u2014 re-run any time)`}`);
3112
+ }
2387
3113
 
2388
3114
  // src/index.ts
2389
3115
  var HELP = `jdf \u2014 JSON Document Format CLI
@@ -2395,12 +3121,18 @@ The CLI exists for these workflows:
2395
3121
  into a validated .jdf (or .jdfx) you can ship.
2396
3122
  \u2022 JDF \u2192 chunks turn a document into retrieval-ready chunks (RAG).
2397
3123
  \u2022 JDF \u2192 vectors embed those chunks, incrementally, for a vector store.
3124
+ \u2022 video \u2192 text attach a time-stamped transcript to a video element so
3125
+ RAG retrieves "video at 02:13", not just "a video".
3126
+ \u2022 folder \u2192 index one command over a directory of .jdf/.jdfx: transcribe,
3127
+ chunk, embed incrementally, write .jdf-rag/index.jsonl.
2398
3128
 
2399
3129
  Usage:
2400
3130
  jdf validate <file.jdf>
2401
3131
  jdf convert <file.{pdf,json,md}> [-o output.{jdf,jdfx}] [--json] [--password PW] [--drop-invisible-text]
2402
3132
  jdf chunk <file.{jdf,jdfx}> [--strategy section|element|fixed] [--format jsonl|json|inline] [--max-tokens N] [-o out]
2403
3133
  jdf embed <file.{jdf,jdfx}> [--provider ollama|openai] [--model NAME] [--strategy \u2026] [--incremental] [-o out]
3134
+ jdf transcribe <file.{jdf,jdfx}> [--from subs.srt|.vtt|.json] [--provider whisper-cli|openai] [--element ID] [--chapters FILE] [-o out]
3135
+ jdf rag <dir> [--provider ollama|openai] [--model NAME] [--transcribe none|whisper-cli|openai] [--no-embed] [--dry-run] [--out DIR]
2404
3136
  jdf --help
2405
3137
 
2406
3138
  Commands:
@@ -2408,6 +3140,8 @@ Commands:
2408
3140
  convert Convert a PDF, JSON, or Markdown file into JDF (alias: import)
2409
3141
  chunk Split a JDF document into retrieval-ready chunks (offline, deterministic)
2410
3142
  embed Compute embeddings for the chunks (local via Ollama by default)
3143
+ transcribe Store time-stamped text on a video element (import SRT/VTT/JSON, or run Whisper)
3144
+ rag Make a whole folder retrieval-ready (finds .jdf/.jdfx, transcribes, chunks, embeds, indexes)
2411
3145
 
2412
3146
  Flags:
2413
3147
  -o, --output <path> Explicit output path
@@ -2424,7 +3158,20 @@ Flags:
2424
3158
  --provider <p> embed: ollama (default, local) | openai (remote API)
2425
3159
  --model <name> embed: model id (default: nomic-embed-text / text-embedding-3-small)
2426
3160
  --incremental embed: skip chunks whose content hash is unchanged
3161
+ --cache <path> embed: sidecar to reuse vectors from (default: the
3162
+ output path itself)
2427
3163
  --no-auto-start embed(ollama): don't auto-launch Ollama via Docker
3164
+ --from <file> transcribe: import subtitles (.srt / .vtt / JSON segments) \u2014 offline, no model
3165
+ --element <id|n> transcribe: which video element (id, or 0-based index); default the only one
3166
+ --chapters <file> transcribe: JSON [{t,title}] or "mm:ss Title" lines \u2192 chapter breadcrumbs
3167
+ --language <tag> transcribe: BCP-47 language hint for Whisper
3168
+ --prompt <text> transcribe: Whisper vocabulary hint (names, acronyms) \u2014 not an instruction
3169
+ --window <sec> chunk/embed/rag: transcript window per video chunk (default 45)
3170
+ --transcribe <p> rag: none (default) | whisper-cli | openai \u2014 for videos that have no transcript yet
3171
+ --no-embed rag: chunk + index only
3172
+ --dry-run rag: list what would happen, write nothing
3173
+ --out <dir> rag: index folder (default <dir>/.jdf-rag)
3174
+ rag reads defaults from <dir>/jdf.rag.json (same keys as the flags; flags win)
2428
3175
 
2429
3176
  Environment (embed):
2430
3177
  ollama: OLLAMA_HOST (default http://localhost:11434)
@@ -2438,8 +3185,11 @@ Examples:
2438
3185
  jdf chunk report.jdf --format inline # embed the chunk index into the .jdf
2439
3186
  jdf embed report.jdf # local embeddings via Ollama (auto-setup)
2440
3187
  jdf embed report.jdf --provider openai --incremental
3188
+ jdf transcribe talk.jdfx --from talk.srt --chapters chapters.txt # then: jdf chunk talk.jdfx
3189
+ jdf transcribe talk.jdfx --provider openai --language en --prompt "JDF, jdfx, Ollama"
3190
+ jdf rag ./knowledge-base --transcribe openai # whole folder \u2192 .jdf-rag/index.jsonl
2441
3191
  `;
2442
- var BOOLEAN_FLAGS = /* @__PURE__ */ new Set(["help", "h", "json", "verbose", "skip-validate", "incremental", "no-auto-start", "drop-invisible-text"]);
3192
+ var BOOLEAN_FLAGS = /* @__PURE__ */ new Set(["help", "h", "json", "verbose", "skip-validate", "incremental", "no-auto-start", "drop-invisible-text", "no-embed", "dry-run"]);
2443
3193
  function parseArgs(argv) {
2444
3194
  const positional = [];
2445
3195
  const flags = {};
@@ -2543,14 +3293,55 @@ async function main() {
2543
3293
  strategy: typeof flags.strategy === "string" ? flags.strategy : void 0,
2544
3294
  format: typeof flags.format === "string" ? flags.format : void 0,
2545
3295
  maxTokens: typeof flags["max-tokens"] === "string" ? parseInt(flags["max-tokens"], 10) : void 0,
3296
+ transcriptWindowSec: typeof flags.window === "string" ? parseInt(flags.window, 10) : void 0,
2546
3297
  output: typeof flags.output === "string" ? flags.output : void 0
2547
3298
  });
2548
3299
  process.exit(0);
2549
3300
  }
3301
+ case "transcribe": {
3302
+ const input = positional[0];
3303
+ if (!input) {
3304
+ console.error("Usage: jdf transcribe <file.{jdf,jdfx}> [--from subs.srt|.vtt|.json] [--provider whisper-cli|openai] [--model M] [--language tag] [--element id|n] [--chapters file] [-o out]");
3305
+ process.exit(1);
3306
+ }
3307
+ await transcribeFile(input, {
3308
+ from: typeof flags.from === "string" ? flags.from : void 0,
3309
+ provider: typeof flags.provider === "string" ? flags.provider : void 0,
3310
+ model: typeof flags.model === "string" ? flags.model : void 0,
3311
+ language: typeof flags.language === "string" ? flags.language : void 0,
3312
+ prompt: typeof flags.prompt === "string" ? flags.prompt : void 0,
3313
+ element: typeof flags.element === "string" ? flags.element : void 0,
3314
+ chapters: typeof flags.chapters === "string" ? flags.chapters : void 0,
3315
+ output: typeof flags.output === "string" ? flags.output : void 0
3316
+ });
3317
+ process.exit(0);
3318
+ }
3319
+ case "rag": {
3320
+ const input = positional[0];
3321
+ if (!input) {
3322
+ console.error("Usage: jdf rag <dir> [--provider ollama|openai] [--model NAME] [--strategy \u2026] [--window sec] [--transcribe none|whisper-cli|openai] [--language tag] [--prompt text] [--no-embed] [--dry-run] [--out DIR]");
3323
+ process.exit(1);
3324
+ }
3325
+ await ragFolder(input, {
3326
+ provider: typeof flags.provider === "string" ? flags.provider : void 0,
3327
+ model: typeof flags.model === "string" ? flags.model : void 0,
3328
+ strategy: typeof flags.strategy === "string" ? flags.strategy : void 0,
3329
+ maxTokens: typeof flags["max-tokens"] === "string" ? parseInt(flags["max-tokens"], 10) : void 0,
3330
+ transcriptWindowSec: typeof flags.window === "string" ? parseInt(flags.window, 10) : void 0,
3331
+ transcribe: typeof flags.transcribe === "string" ? flags.transcribe : void 0,
3332
+ transcribeModel: typeof flags["transcribe-model"] === "string" ? flags["transcribe-model"] : void 0,
3333
+ language: typeof flags.language === "string" ? flags.language : void 0,
3334
+ prompt: typeof flags.prompt === "string" ? flags.prompt : void 0,
3335
+ noEmbed: flags["no-embed"] === true,
3336
+ dryRun: flags["dry-run"] === true,
3337
+ out: typeof flags.out === "string" ? flags.out : void 0
3338
+ });
3339
+ process.exit(0);
3340
+ }
2550
3341
  case "embed": {
2551
3342
  const input = positional[0];
2552
3343
  if (!input) {
2553
- console.error("Usage: jdf embed <file.{jdf,jdfx}> [--provider ollama|openai] [--model NAME] [--strategy \u2026] [--incremental] [--no-auto-start] [-o out]");
3344
+ console.error("Usage: jdf embed <file.{jdf,jdfx}> [--provider ollama|openai] [--model NAME] [--strategy \u2026] [--incremental] [--cache prev.embeddings.json] [--no-auto-start] [-o out]");
2554
3345
  process.exit(1);
2555
3346
  }
2556
3347
  await embedFile(input, {
@@ -2559,8 +3350,10 @@ async function main() {
2559
3350
  strategy: typeof flags.strategy === "string" ? flags.strategy : void 0,
2560
3351
  maxTokens: typeof flags["max-tokens"] === "string" ? parseInt(flags["max-tokens"], 10) : void 0,
2561
3352
  incremental: flags.incremental === true,
3353
+ transcriptWindowSec: typeof flags.window === "string" ? parseInt(flags.window, 10) : void 0,
2562
3354
  autoStart: flags["no-auto-start"] !== true,
2563
- output: typeof flags.output === "string" ? flags.output : void 0
3355
+ output: typeof flags.output === "string" ? flags.output : void 0,
3356
+ cache: typeof flags.cache === "string" ? flags.cache : void 0
2564
3357
  });
2565
3358
  process.exit(0);
2566
3359
  }