@uurtech/jdf-cli 0.2.0 → 0.2.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -1,12 +1,13 @@
1
1
  #!/usr/bin/env node
2
- import fs7 from 'fs';
3
- import path7 from 'path';
2
+ import fs8 from 'fs';
3
+ import path8 from 'path';
4
4
  import { fileURLToPath } from 'url';
5
5
  import Ajv from 'ajv';
6
6
  import addFormats from 'ajv-formats';
7
7
  import JSZip from 'jszip';
8
8
  import crypto, { createHash } from 'crypto';
9
9
  import { readFile } from 'fs/promises';
10
+ import os from 'os';
10
11
  import { execFileSync, spawnSync } from 'child_process';
11
12
 
12
13
  var __require = /* @__PURE__ */ ((x) => typeof require !== "undefined" ? require : typeof Proxy !== "undefined" ? new Proxy(x, {
@@ -23,17 +24,17 @@ var JDFX_MANIFEST_PATH = "manifest.json";
23
24
  var JDFX_ASSET_DIR = "assets";
24
25
 
25
26
  // src/commands/validate.ts
26
- var __dirname$1 = path7.dirname(fileURLToPath(import.meta.url));
27
+ var __dirname$1 = path8.dirname(fileURLToPath(import.meta.url));
27
28
  function resolveSchemaPath() {
28
- const bundled = path7.resolve(__dirname$1, "jdf-schema.json");
29
- if (fs7.existsSync(bundled)) return bundled;
30
- const dev = path7.resolve(__dirname$1, "../../../../spec/jdf-schema.json");
29
+ const bundled = path8.resolve(__dirname$1, "jdf-schema.json");
30
+ if (fs8.existsSync(bundled)) return bundled;
31
+ const dev = path8.resolve(__dirname$1, "../../../../spec/jdf-schema.json");
31
32
  return dev;
32
33
  }
33
34
  var SCHEMA_PATH = resolveSchemaPath();
34
35
  async function loadDocument(filePath) {
35
36
  if (filePath.toLowerCase().endsWith(".jdfx")) {
36
- const zip = await JSZip.loadAsync(fs7.readFileSync(filePath));
37
+ const zip = await JSZip.loadAsync(fs8.readFileSync(filePath));
37
38
  const docFile = zip.file(JDFX_DOCUMENT_PATH);
38
39
  if (!docFile) {
39
40
  console.error(`\u2717 Bundle missing ${JDFX_DOCUMENT_PATH}`);
@@ -58,11 +59,11 @@ async function loadDocument(filePath) {
58
59
  }
59
60
  return { doc, bundle: { manifest, assetCount } };
60
61
  }
61
- return { doc: JSON.parse(fs7.readFileSync(filePath, "utf-8")) };
62
+ return { doc: JSON.parse(fs8.readFileSync(filePath, "utf-8")) };
62
63
  }
63
64
  async function validate(file) {
64
- const filePath = path7.resolve(file);
65
- if (!fs7.existsSync(filePath)) {
65
+ const filePath = path8.resolve(file);
66
+ if (!fs8.existsSync(filePath)) {
66
67
  console.error(`File not found: ${filePath}`);
67
68
  return false;
68
69
  }
@@ -75,11 +76,11 @@ async function validate(file) {
75
76
  }
76
77
  if (!loaded) return false;
77
78
  const { doc, bundle } = loaded;
78
- if (!fs7.existsSync(SCHEMA_PATH)) {
79
+ if (!fs8.existsSync(SCHEMA_PATH)) {
79
80
  console.error(`Schema not found at ${SCHEMA_PATH}`);
80
81
  return false;
81
82
  }
82
- const schema = JSON.parse(fs7.readFileSync(SCHEMA_PATH, "utf-8"));
83
+ const schema = JSON.parse(fs8.readFileSync(SCHEMA_PATH, "utf-8"));
83
84
  const ajv = new Ajv({ allErrors: true, strict: false });
84
85
  addFormats(ajv);
85
86
  const validateFn = ajv.compile(schema);
@@ -88,7 +89,7 @@ async function validate(file) {
88
89
  const d = doc;
89
90
  const pageCount = Array.isArray(d.pages) ? d.pages.length : 0;
90
91
  const elCount = Array.isArray(d.pages) ? d.pages.reduce((acc, p) => acc + (Array.isArray(p?.elements) ? p.elements.length : 0), 0) : 0;
91
- console.log(`\u2713 Valid: ${path7.basename(filePath)}`);
92
+ console.log(`\u2713 Valid: ${path8.basename(filePath)}`);
92
93
  console.log(` Format: ${d.$jdf}${bundle ? " (jdfx bundle)" : ""}`);
93
94
  console.log(` Title: ${d.meta?.title}`);
94
95
  console.log(` Pages: ${pageCount}`);
@@ -101,7 +102,7 @@ async function validate(file) {
101
102
  }
102
103
  return true;
103
104
  }
104
- console.error(`\u2717 Invalid: ${path7.basename(filePath)}`);
105
+ console.error(`\u2717 Invalid: ${path8.basename(filePath)}`);
105
106
  for (const err of validateFn.errors || []) {
106
107
  const loc = err.instancePath || "(root)";
107
108
  console.error(` ${loc} \u2014 ${err.message}`);
@@ -304,13 +305,13 @@ function stripInline(text) {
304
305
  return parseInline(text).map((r) => r.text).join("");
305
306
  }
306
307
  async function importMarkdown(inputPath, outputPath) {
307
- const input = path7.resolve(inputPath);
308
+ const input = path8.resolve(inputPath);
308
309
  console.log(`Importing: ${input}`);
309
- const content = fs7.readFileSync(input, "utf-8");
310
- const doc = convertMarkdownToJdf(content, path7.basename(input, path7.extname(input)), path7.dirname(input));
310
+ const content = fs8.readFileSync(input, "utf-8");
311
+ const doc = convertMarkdownToJdf(content, path8.basename(input, path8.extname(input)), path8.dirname(input));
311
312
  let output;
312
313
  if (outputPath) {
313
- output = path7.resolve(outputPath);
314
+ output = path8.resolve(outputPath);
314
315
  } else {
315
316
  const stem = input.replace(/\.(md|markdown)$/i, "");
316
317
  output = stem + (shouldUseJdfx(doc) ? ".jdfx" : ".jdf");
@@ -318,11 +319,11 @@ async function importMarkdown(inputPath, outputPath) {
318
319
  console.log(`Output: ${output}`);
319
320
  if (output.toLowerCase().endsWith(".jdfx")) {
320
321
  const { bytes, manifest } = await packJdfx(doc);
321
- fs7.writeFileSync(output, bytes);
322
+ fs8.writeFileSync(output, bytes);
322
323
  console.log(`
323
324
  Done! Created ${doc.pages.length} page(s), ${manifest.assets.length} asset(s) bundled`);
324
325
  } else {
325
- fs7.writeFileSync(output, JSON.stringify(doc, null, 2));
326
+ fs8.writeFileSync(output, JSON.stringify(doc, null, 2));
326
327
  console.log(`
327
328
  Done! Created ${doc.pages.length} page(s)`);
328
329
  }
@@ -339,10 +340,10 @@ var MIME_BY_EXT2 = {
339
340
  };
340
341
  function resolveImageSrc(src, baseDir) {
341
342
  if (/^(https?:|data:|file:)/i.test(src)) return src;
342
- const abs = path7.isAbsolute(src) ? src : path7.resolve(baseDir, src);
343
+ const abs = path8.isAbsolute(src) ? src : path8.resolve(baseDir, src);
343
344
  try {
344
- const bytes = fs7.readFileSync(abs);
345
- const ext = path7.extname(abs).slice(1).toLowerCase();
345
+ const bytes = fs8.readFileSync(abs);
346
+ const ext = path8.extname(abs).slice(1).toLowerCase();
346
347
  const mime = MIME_BY_EXT2[ext] || "application/octet-stream";
347
348
  return `data:${mime};base64,${bytes.toString("base64")}`;
348
349
  } catch {
@@ -1699,6 +1700,10 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
1699
1700
  if (arr) arr.push(op);
1700
1701
  else opBins.set(key, [op]);
1701
1702
  }
1703
+ const sizePenalty = (op, fontSize) => {
1704
+ if (!op.fontSize || !fontSize) return 0;
1705
+ return Math.abs(Math.log(op.fontSize / fontSize)) * 6;
1706
+ };
1702
1707
  const findOp = (x, y, fontSize) => {
1703
1708
  let best = null;
1704
1709
  let bestD = Infinity;
@@ -1708,7 +1713,7 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
1708
1713
  const arr = opBins.get(`${bx + dx},${by + dy}`);
1709
1714
  if (!arr) continue;
1710
1715
  for (const op of arr) {
1711
- const d = Math.hypot(op.x - x, op.y - y);
1716
+ const d = Math.hypot(op.x - x, op.y - y) + sizePenalty(op, fontSize);
1712
1717
  if (d < bestD) {
1713
1718
  bestD = d;
1714
1719
  best = op;
@@ -1718,8 +1723,9 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
1718
1723
  }
1719
1724
  if (best) return best;
1720
1725
  const tol = Math.max(2, fontSize * 0.6);
1726
+ const sizeOk = (op) => !op.fontSize || !fontSize || op.fontSize / fontSize > 0.6 && op.fontSize / fontSize < 1.7;
1721
1727
  for (const op of ops.textOps) {
1722
- if (Math.abs(op.y - y) > tol) continue;
1728
+ if (!sizeOk(op) || Math.abs(op.y - y) > tol) continue;
1723
1729
  const d = Math.abs(op.x - x) + Math.abs(op.y - y) * 4;
1724
1730
  if (d < bestD) {
1725
1731
  bestD = d;
@@ -1728,6 +1734,7 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
1728
1734
  }
1729
1735
  if (best) return best;
1730
1736
  for (const op of ops.textOps) {
1737
+ if (!sizeOk(op)) continue;
1731
1738
  const d = Math.hypot(op.x - x, op.y - y);
1732
1739
  if (d < bestD) {
1733
1740
  bestD = d;
@@ -1746,6 +1753,7 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
1746
1753
  const conv = viewport.convertToViewportPoint(baseX, baseY);
1747
1754
  const vx = safeNum(conv?.[0], 0);
1748
1755
  const vy = safeNum(conv?.[1], 0);
1756
+ if (fontSize < 1.5) return;
1749
1757
  const op = findOp(vx, vy, fontSize);
1750
1758
  const mode = op?.mode ?? 0;
1751
1759
  if (mode === 7) return;
@@ -1869,10 +1877,92 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
1869
1877
  sizeChars.set(k, (sizeChars.get(k) ?? 0) + l.text.length);
1870
1878
  }
1871
1879
  const bodyFontSize = [...sizeChars.entries()].sort((a, b) => b[1] - a[1])[0]?.[0] ?? 0;
1880
+ const rowOf = /* @__PURE__ */ new Map();
1881
+ const rowStartOf = /* @__PURE__ */ new Map();
1882
+ const nextOnRow = /* @__PURE__ */ new Map();
1883
+ {
1884
+ const order = lines.map((_, i) => i).filter((i) => !consumedLines.has(i));
1885
+ for (let a = 0; a < order.length; a++) {
1886
+ const i = order[a], li = lines[i];
1887
+ const tolY = Math.max(0.6, li.fontSize * PT_TO_MM2 * 0.35);
1888
+ let bestNext = -1, bestX = Infinity;
1889
+ for (let b = 0; b < order.length; b++) {
1890
+ const j = order[b], lj = lines[j];
1891
+ if (j === i || Math.abs(lj.y - li.y) > tolY || lj.x <= li.x) continue;
1892
+ if (lj.x < bestX) {
1893
+ bestX = lj.x;
1894
+ bestNext = j;
1895
+ }
1896
+ }
1897
+ if (bestNext >= 0) nextOnRow.set(i, bestNext);
1898
+ }
1899
+ const seen = /* @__PURE__ */ new Set();
1900
+ for (const i of order) {
1901
+ if (seen.has(i)) continue;
1902
+ const row = [i];
1903
+ seen.add(i);
1904
+ let cur = i;
1905
+ while (nextOnRow.has(cur)) {
1906
+ const j = nextOnRow.get(cur), lc = lines[cur], lj = lines[j];
1907
+ const em = Math.min(lc.fontSize, lj.fontSize) * PT_TO_MM2;
1908
+ const gap = lj.x - (lc.x + lc.width);
1909
+ if (gap < -em * 0.3 || gap > em * 0.6) break;
1910
+ row.push(j);
1911
+ seen.add(j);
1912
+ cur = j;
1913
+ }
1914
+ rowOf.set(i, row);
1915
+ for (const j of row) rowStartOf.set(j, i);
1916
+ }
1917
+ }
1918
+ const runStyle = (l) => {
1919
+ const cls = fontMap.get(l.fontName) || classifyFont(l.fontName || "");
1920
+ return { cls, bold: cls.weight === "bold", italic: cls.style === "italic" };
1921
+ };
1872
1922
  lines.forEach((l, lineIdx) => {
1873
1923
  const tableEl = tableAtLine.get(lineIdx);
1874
1924
  if (tableEl) elements.push(tableEl);
1875
1925
  if (consumedLines.has(lineIdx)) return;
1926
+ const row = rowOf.get(lineIdx);
1927
+ if (!row) return;
1928
+ if (row.length > 1) {
1929
+ const first = lines[row[0]], last = lines[row[row.length - 1]];
1930
+ const base = runStyle(first);
1931
+ const rowEnd = last.x + last.width;
1932
+ const measuredW = Math.max((rowEnd - first.x) * 1.2 + first.fontSize * PT_TO_MM2 * 0.4, first.fontSize * PT_TO_MM2);
1933
+ const nextIdx2 = nextOnRow.get(row[row.length - 1]);
1934
+ const cap2 = nextIdx2 != null ? lines[nextIdx2].x - first.x - first.fontSize * PT_TO_MM2 * 0.3 : pageWmm - first.x;
1935
+ const runs2 = [];
1936
+ row.forEach((idx, k) => {
1937
+ const r = lines[idx];
1938
+ const st = runStyle(r);
1939
+ let text2 = r.text;
1940
+ if (k > 0) {
1941
+ const prev2 = lines[row[k - 1]];
1942
+ const gap = r.x - (prev2.x + prev2.width);
1943
+ if (gap > r.fontSize * PT_TO_MM2 * 0.08 && !/\s$/.test(prev2.text) && !/^\s/.test(text2)) text2 = " " + text2;
1944
+ }
1945
+ const run = { text: text2 };
1946
+ if (st.bold) run.bold = true;
1947
+ if (st.italic) run.italic = true;
1948
+ if (r.color !== "#000000") run.color = r.color;
1949
+ if (Math.abs(r.fontSize - first.fontSize) >= 0.5) run.fontSize = Math.round(r.fontSize * 10) / 10;
1950
+ if (st.cls.family !== base.cls.family) run.fontFamily = st.cls.family;
1951
+ const lk = findLinkForRun2(r);
1952
+ if (lk) run.link = lk.url ? lk.url : lk.destPage != null ? { type: "internal", target: `#page-${lk.destPage + 1}` } : void 0;
1953
+ runs2.push(run);
1954
+ });
1955
+ const style2 = { fontSize: Math.round(first.fontSize * 10) / 10, fontFamily: base.cls.family };
1956
+ if (first.opacity < 0.999) style2.opacity = Math.round(first.opacity * 100) / 100;
1957
+ elements.push({
1958
+ type: "richtext",
1959
+ runs: runs2,
1960
+ position: { x: Math.max(0, Math.round(first.x * 100) / 100), y: Math.max(0, Math.round(Math.min(...row.map((i) => lines[i].y)) * 100) / 100) },
1961
+ width: Math.max(2, Math.round(Math.max(first.fontSize * PT_TO_MM2, Math.min(measuredW, cap2)) * 100) / 100),
1962
+ style: style2
1963
+ });
1964
+ return;
1965
+ }
1876
1966
  const cls = fontMap.get(l.fontName) || classifyFont(l.fontName || "");
1877
1967
  const style = {
1878
1968
  fontSize: Math.round(l.fontSize * 10) / 10,
@@ -1883,9 +1973,11 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
1883
1973
  if (l.color !== "#000000") style.color = l.color;
1884
1974
  if (l.opacity < 0.999) style.opacity = Math.round(l.opacity * 100) / 100;
1885
1975
  const link = findLinkForRun2(l);
1886
- const measured = Math.max(l.width + l.fontSize * PT_TO_MM2 * 0.4, l.fontSize * PT_TO_MM2);
1976
+ const measured = Math.max(l.width * 1.2 + l.fontSize * PT_TO_MM2 * 0.4, l.fontSize * PT_TO_MM2);
1887
1977
  const remaining = Math.max(measured, pageWmm - l.x);
1888
- const elWidth = Math.min(measured, remaining);
1978
+ const nextIdx = nextOnRow.get(lineIdx);
1979
+ const cap = nextIdx != null ? Math.max(l.fontSize * PT_TO_MM2, lines[nextIdx].x - l.x - l.fontSize * PT_TO_MM2 * 0.3) : Infinity;
1980
+ const elWidth = Math.min(measured, remaining, cap);
1889
1981
  const text = {
1890
1982
  type: "text",
1891
1983
  content: l.text,
@@ -2097,25 +2189,209 @@ async function importPdfToJdf2(source, title, options = {}) {
2097
2189
  };
2098
2190
  return importPdfToJdf(source, title, runtime, options);
2099
2191
  }
2192
+ var DEFAULT_OLLAMA_MODEL = "qwen2.5vl:3b";
2193
+ var DEFAULT_OPENAI_MODEL = "gpt-4o-mini";
2194
+ var OLLAMA_HOST = process.env.OLLAMA_HOST || "http://localhost:11434";
2195
+ async function loadDoc(file) {
2196
+ if (file.toLowerCase().endsWith(".jdfx")) {
2197
+ const zip = await JSZip.loadAsync(fs8.readFileSync(file));
2198
+ const f = zip.file(JDFX_DOCUMENT_PATH);
2199
+ if (!f) throw new Error(`Bundle missing ${JDFX_DOCUMENT_PATH}`);
2200
+ const doc = JSON.parse(await f.async("string"));
2201
+ const manifest = zip.file("manifest.json") ? JSON.parse(await zip.file("manifest.json").async("string")) : { assets: [] };
2202
+ for (const a of manifest.assets ?? []) {
2203
+ const af = zip.file(a.path);
2204
+ if (!af) continue;
2205
+ const data = (await af.async("nodebuffer")).toString("base64");
2206
+ const res = { src: "embedded", mimeType: a.mimeType, data };
2207
+ doc.resources ??= {};
2208
+ if (/^video\//i.test(a.mimeType || "")) (doc.resources.videos ??= {})[a.id] = res;
2209
+ else (doc.resources.images ??= {})[a.id] = res;
2210
+ }
2211
+ return { doc, bundle: true };
2212
+ }
2213
+ return { doc: JSON.parse(fs8.readFileSync(file, "utf-8")), bundle: false };
2214
+ }
2215
+ function findImages(doc) {
2216
+ const out = [];
2217
+ const walk2 = (els, page) => {
2218
+ for (const el of els ?? []) {
2219
+ if (el?.type === "image") out.push({ el, page, index: out.length });
2220
+ if (el?.elements) walk2(el.elements, page);
2221
+ }
2222
+ };
2223
+ doc.pages.forEach((p, i) => walk2(p.elements, i + 1));
2224
+ return out;
2225
+ }
2226
+ async function imageBytes(doc, el, docDir) {
2227
+ const fromData = (d, fallback) => {
2228
+ const m = d.match(/^data:([^;,]+)?[^,]*,(.*)$/s);
2229
+ return m ? { bytes: Buffer.from(m[2], "base64"), mime: m[1] || fallback } : { bytes: Buffer.from(d, "base64"), mime: fallback };
2230
+ };
2231
+ const res = el.resource ? doc.resources?.images?.[el.resource] ?? doc.resources?.[el.resource] : void 0;
2232
+ if (res?.data) return fromData(String(res.data), res.mimeType || "image/png");
2233
+ if (res?.path) {
2234
+ const p = path8.resolve(docDir, res.path);
2235
+ return fs8.existsSync(p) ? { bytes: fs8.readFileSync(p), mime: res.mimeType || "image/png" } : null;
2236
+ }
2237
+ const src = el.src;
2238
+ if (!src) return null;
2239
+ if (src.startsWith("data:")) return fromData(src, "image/png");
2240
+ if (/^https?:\/\//i.test(src)) {
2241
+ const r = await fetch(src);
2242
+ if (!r.ok) throw new Error(`download failed ${r.status}: ${src}`);
2243
+ return { bytes: Buffer.from(await r.arrayBuffer()), mime: r.headers.get("content-type") || "image/png" };
2244
+ }
2245
+ const local = path8.resolve(docDir, src);
2246
+ return fs8.existsSync(local) ? { bytes: fs8.readFileSync(local), mime: "image/png" } : null;
2247
+ }
2248
+ async function ocrTesseract(bytes, lang) {
2249
+ const { createWorker } = await import('tesseract.js');
2250
+ const cachePath = path8.join(os.homedir(), ".cache", "jdf", "tesseract");
2251
+ fs8.mkdirSync(cachePath, { recursive: true });
2252
+ const worker = await createWorker(lang, 1, { cachePath, logger: () => {
2253
+ } });
2254
+ try {
2255
+ const { data } = await worker.recognize(bytes, {}, { text: true, blocks: true });
2256
+ let w = 1, h = 1;
2257
+ try {
2258
+ const { loadImage } = await import('@napi-rs/canvas');
2259
+ const im = await loadImage(bytes);
2260
+ w = im.width || 1;
2261
+ h = im.height || 1;
2262
+ } catch {
2263
+ }
2264
+ const lines = data.blocks?.flatMap((b) => b.paragraphs?.flatMap((p) => p.lines ?? []) ?? []) ?? data.lines ?? [];
2265
+ const blocks = lines.map((ln) => ({ text: String(ln.text ?? "").replace(/\s+/g, " ").trim(), confidence: ln.confidence != null ? Math.round(ln.confidence) / 100 : void 0, bbox: ln.bbox ? { x: +(ln.bbox.x0 / w).toFixed(4), y: +(ln.bbox.y0 / h).toFixed(4), w: +((ln.bbox.x1 - ln.bbox.x0) / w).toFixed(4), h: +((ln.bbox.y1 - ln.bbox.y0) / h).toFixed(4) } : void 0 })).filter((b) => b.text.length > 0 && (b.confidence == null || b.confidence >= 0.3));
2266
+ if (!blocks.length && String(data.text ?? "").trim()) blocks.push({ text: String(data.text).replace(/\s+/g, " ").trim() });
2267
+ return { blocks, source: `tesseract.js:${lang}` };
2268
+ } finally {
2269
+ await worker.terminate();
2270
+ }
2271
+ }
2272
+ async function openaiVision(bytes, mime, model, prompt2) {
2273
+ const key = process.env.OPENAI_API_KEY;
2274
+ if (!key) throw new Error("OPENAI_API_KEY is not set");
2275
+ const base = process.env.OPENAI_BASE_URL || "https://api.openai.com/v1";
2276
+ const r = await fetch(`${base}/chat/completions`, { method: "POST", headers: { Authorization: `Bearer ${key}`, "content-type": "application/json" }, body: JSON.stringify({ model, messages: [{ role: "user", content: [{ type: "text", text: prompt2 }, { type: "image_url", image_url: { url: `data:${mime};base64,${bytes.toString("base64")}` } }] }], max_tokens: 800 }) });
2277
+ if (!r.ok) throw new Error(`OpenAI vision failed ${r.status}: ${(await r.text()).slice(0, 300)}`);
2278
+ const j = await r.json();
2279
+ return String(j.choices?.[0]?.message?.content ?? "").trim();
2280
+ }
2281
+ async function ollamaVision(bytes, model, prompt2) {
2282
+ const r = await fetch(`${OLLAMA_HOST}/api/generate`, { method: "POST", body: JSON.stringify({ model, prompt: prompt2, images: [bytes.toString("base64")], stream: false, options: { temperature: 0 } }) });
2283
+ if (!r.ok) throw new Error(`Ollama vision failed ${r.status}: ${(await r.text()).slice(0, 300)} \u2014 is the model pulled? (ollama pull ${model})`);
2284
+ const j = await r.json();
2285
+ return String(j.response ?? "").trim();
2286
+ }
2287
+ var CAPTION_PROMPT = "Describe this image for a search index in one or two factual sentences: what it shows, any chart type, axes, trends, labels, names and numbers you can read. No preamble.";
2288
+ var OCR_PROMPT = "Transcribe all text visible in this image exactly, line by line, top to bottom, left to right. Output only the text.";
2289
+ async function describeDocument(doc, docDir, opts = {}) {
2290
+ const ocrP = opts.ocr ?? "tesseract";
2291
+ const capP = opts.caption ?? "ollama";
2292
+ const lang = opts.ocrLanguage ?? "eng";
2293
+ const images = findImages(doc);
2294
+ const targets = opts.element != null ? images.filter((im) => im.el.id === opts.element || String(im.index) === opts.element) : images;
2295
+ if (opts.element != null && !targets.length) throw new Error(`no image element "${opts.element}" (have: ${images.map((i) => i.el.id ?? `#${i.index}`).join(", ") || "none"})`);
2296
+ const stats = { ocr: 0, captions: 0, skipped: 0, failed: [] };
2297
+ for (const im of targets) {
2298
+ const el = im.el;
2299
+ if (!el.id) el.id = `image-${im.index + 1}`;
2300
+ const needOcr = ocrP !== "none" && (opts.force || !el.ocr?.blocks?.length);
2301
+ const needCap = capP !== "none" && (opts.force || !el.caption);
2302
+ if (!needOcr && !needCap) {
2303
+ stats.skipped++;
2304
+ continue;
2305
+ }
2306
+ const img = await imageBytes(doc, el, docDir);
2307
+ if (!img) {
2308
+ stats.failed.push(`${el.id}: image bytes not reachable`);
2309
+ continue;
2310
+ }
2311
+ if (needOcr) {
2312
+ try {
2313
+ if (ocrP === "tesseract") {
2314
+ const r = await ocrTesseract(img.bytes, lang);
2315
+ el.ocr = { language: lang, source: r.source, created: (/* @__PURE__ */ new Date()).toISOString(), blocks: r.blocks };
2316
+ } else {
2317
+ const text = await openaiVision(img.bytes, img.mime, opts.captionModel ?? DEFAULT_OPENAI_MODEL, OCR_PROMPT);
2318
+ el.ocr = { source: `openai:${opts.captionModel ?? DEFAULT_OPENAI_MODEL}`, created: (/* @__PURE__ */ new Date()).toISOString(), blocks: text.split(/\r?\n/).map((t) => t.trim()).filter(Boolean).map((t) => ({ text: t })) };
2319
+ }
2320
+ stats.ocr++;
2321
+ } catch (e) {
2322
+ stats.failed.push(`${el.id} ocr: ${e.message}`);
2323
+ }
2324
+ }
2325
+ if (needCap) {
2326
+ try {
2327
+ const model = opts.captionModel ?? (capP === "ollama" ? DEFAULT_OLLAMA_MODEL : DEFAULT_OPENAI_MODEL);
2328
+ const text = capP === "ollama" ? await ollamaVision(img.bytes, model, CAPTION_PROMPT) : await openaiVision(img.bytes, img.mime, model, CAPTION_PROMPT);
2329
+ if (text) {
2330
+ el.caption = text;
2331
+ el.captionSource = `${capP}:${model}`;
2332
+ stats.captions++;
2333
+ }
2334
+ } catch (e) {
2335
+ stats.failed.push(`${el.id} caption: ${e.message}`);
2336
+ }
2337
+ }
2338
+ if (!opts.quiet) console.log(` \xB7 ${el.id} (page ${im.page}): ${needOcr ? `ocr ${el.ocr?.blocks?.length ?? 0} block(s)` : "ocr kept"}${needCap ? ` \xB7 caption ${el.caption ? `"${String(el.caption).slice(0, 70)}${String(el.caption).length > 70 ? "\u2026" : ""}"` : "\u2014"}` : ""}`);
2339
+ }
2340
+ return stats;
2341
+ }
2342
+ async function describeFile(inputPath, opts = {}) {
2343
+ const input = path8.resolve(inputPath);
2344
+ if (!fs8.existsSync(input)) throw new Error(`File not found: ${input}`);
2345
+ const { doc, bundle } = await loadDoc(input);
2346
+ const images = findImages(doc);
2347
+ if (!images.length) {
2348
+ console.log(`No image elements in ${path8.basename(input)} \u2014 nothing to describe.`);
2349
+ return;
2350
+ }
2351
+ console.log(`Describing: ${path8.basename(input)} \u2014 ${images.length} image(s); ocr=${opts.ocr ?? "tesseract"} caption=${opts.caption ?? "ollama"}${(opts.caption ?? "ollama") === "ollama" ? ` (${opts.captionModel ?? DEFAULT_OLLAMA_MODEL}, local)` : ""}`);
2352
+ const stats = await describeDocument(doc, path8.dirname(input), opts);
2353
+ const output = opts.output ? path8.resolve(opts.output) : input;
2354
+ if (output.toLowerCase().endsWith(".jdfx") || bundle && !opts.output) fs8.writeFileSync(output, (await packJdfx(doc)).bytes);
2355
+ else fs8.writeFileSync(output, JSON.stringify(doc, null, 2));
2356
+ console.log(`Done: ${stats.ocr} OCR, ${stats.captions} caption(s), ${stats.skipped} already had text${stats.failed.length ? `, ${stats.failed.length} failed` : ""}.`);
2357
+ for (const f of stats.failed) console.warn(` ! ${f}`);
2358
+ console.log(`Output: ${output}
2359
+ Next: jdf chunk ${path8.basename(output)} # image text is now part of the chunks`);
2360
+ if (stats.failed.length && stats.ocr + stats.captions === 0) process.exitCode = 1;
2361
+ }
2100
2362
 
2101
2363
  // src/commands/import-pdf.ts
2102
2364
  async function importPdf(inputPath, outputPath, options = {}) {
2103
- const input = path7.resolve(inputPath);
2104
- if (!fs7.existsSync(input)) {
2365
+ const input = path8.resolve(inputPath);
2366
+ if (!fs8.existsSync(input)) {
2105
2367
  console.error(`File not found: ${input}`);
2106
2368
  process.exit(1);
2107
2369
  }
2108
2370
  console.log(`Importing: ${input}`);
2109
- const title = path7.basename(input, path7.extname(input));
2371
+ const title = path8.basename(input, path8.extname(input));
2110
2372
  const t0 = Date.now();
2111
2373
  const doc = await importPdfToJdf2(input, title, {
2112
2374
  password: options.password,
2113
2375
  invisibleText: options.dropInvisibleText ? "drop" : "keep"
2114
2376
  });
2115
2377
  console.log(`Parsed in ${((Date.now() - t0) / 1e3).toFixed(1)}s \u2014 ${doc.pages.length} page(s)`);
2378
+ const scanned = doc.pages.filter((p) => !p.elements.some((e) => e.type === "text" || e.type === "richtext" || e.type === "table") && p.elements.some((e) => e.type === "image"));
2379
+ if (scanned.length) {
2380
+ if (options.ocr && options.ocr !== "none") {
2381
+ console.log(`OCR: ${scanned.length} scanned page(s) \u2192 ${options.ocr}`);
2382
+ for (const p of scanned) for (const e of p.elements) if (e.type === "image" && !e.id) e.id = `scan-${doc.pages.indexOf(p) + 1}`;
2383
+ const ids = scanned.flatMap((p) => p.elements.filter((e) => e.type === "image").map((e) => e.id));
2384
+ for (const id of ids) {
2385
+ const st = await describeDocument(doc, path8.dirname(input), { element: id, ocr: options.ocr, caption: "none", quiet: true });
2386
+ if (st.failed.length) console.warn(` ! ${st.failed.join("; ")}`);
2387
+ }
2388
+ } else {
2389
+ console.warn(`! ${scanned.length} page(s) have no text layer (scanned). RAG will skip them \u2014 re-run with --ocr tesseract (local) or --ocr openai.`);
2390
+ }
2391
+ }
2116
2392
  let output;
2117
2393
  if (outputPath) {
2118
- output = path7.resolve(outputPath);
2394
+ output = path8.resolve(outputPath);
2119
2395
  } else {
2120
2396
  const stem = input.replace(/\.pdf$/i, "");
2121
2397
  const wantJdfx = !options.forceJson && shouldUseJdfx(doc);
@@ -2124,11 +2400,11 @@ async function importPdf(inputPath, outputPath, options = {}) {
2124
2400
  console.log(`Output: ${output}`);
2125
2401
  if (output.toLowerCase().endsWith(".jdfx")) {
2126
2402
  const { bytes, manifest } = await packJdfx(doc);
2127
- fs7.writeFileSync(output, bytes);
2403
+ fs8.writeFileSync(output, bytes);
2128
2404
  console.log(`
2129
2405
  Done! Created ${doc.pages.length} page(s), ${manifest.assets.length} asset(s) bundled`);
2130
2406
  } else {
2131
- fs7.writeFileSync(output, JSON.stringify(doc, null, 2));
2407
+ fs8.writeFileSync(output, JSON.stringify(doc, null, 2));
2132
2408
  console.log(`
2133
2409
  Done! Created ${doc.pages.length} page(s)`);
2134
2410
  }
@@ -2141,23 +2417,23 @@ var ImportJsonError = class extends Error {
2141
2417
  }
2142
2418
  };
2143
2419
  async function importJson(inputPath, outputPath, options = {}) {
2144
- const input = path7.resolve(inputPath);
2145
- if (!fs7.existsSync(input)) {
2420
+ const input = path8.resolve(inputPath);
2421
+ if (!fs8.existsSync(input)) {
2146
2422
  throw new ImportJsonError(`File not found: ${input}`);
2147
2423
  }
2148
2424
  console.log(`Importing: ${input}`);
2149
- const raw = fs7.readFileSync(input, "utf-8");
2425
+ const raw = fs8.readFileSync(input, "utf-8");
2150
2426
  let parsed;
2151
2427
  try {
2152
2428
  parsed = JSON.parse(raw);
2153
2429
  } catch (e) {
2154
2430
  throw new ImportJsonError(`Not valid JSON: ${e.message}`);
2155
2431
  }
2156
- const title = path7.basename(input, path7.extname(input));
2432
+ const title = path8.basename(input, path8.extname(input));
2157
2433
  const doc = normaliseToJdf(parsed, title);
2158
2434
  let output;
2159
2435
  if (outputPath) {
2160
- output = path7.resolve(outputPath);
2436
+ output = path8.resolve(outputPath);
2161
2437
  } else {
2162
2438
  const stem = input.replace(/\.json$/i, "");
2163
2439
  const wantJdfx = !options.forceJson && shouldUseJdfx(doc);
@@ -2166,11 +2442,11 @@ async function importJson(inputPath, outputPath, options = {}) {
2166
2442
  console.log(`Output: ${output}`);
2167
2443
  if (output.toLowerCase().endsWith(".jdfx")) {
2168
2444
  const { bytes, manifest } = await packJdfx(doc);
2169
- fs7.writeFileSync(output, bytes);
2445
+ fs8.writeFileSync(output, bytes);
2170
2446
  console.log(`
2171
2447
  Done! Created ${doc.pages.length} page(s), ${manifest.assets.length} asset(s) bundled`);
2172
2448
  } else {
2173
- fs7.writeFileSync(output, JSON.stringify(doc, null, 2));
2449
+ fs8.writeFileSync(output, JSON.stringify(doc, null, 2));
2174
2450
  console.log(`
2175
2451
  Done! Created ${doc.pages.length} page(s)`);
2176
2452
  }
@@ -2272,8 +2548,8 @@ function transcriptChunks(el, elementId2, page, crumb, windowSec, maxTokens) {
2272
2548
  const body = win.map((sg) => sg.speaker ? `${sg.speaker}: ${sg.text}` : sg.text).join(" ").replace(/\s+/g, " ").trim();
2273
2549
  const text = `[${fmtTime(t0)}\u2013${fmtTime(t1)}] ${body}`;
2274
2550
  const chapter = chapterAt(t0);
2275
- const path9 = [...crumb, ...el.title ? [String(el.title)] : [], ...chapter ? [String(chapter.title)] : []];
2276
- out.push({ id: `${elementId2}@${Math.round(t0)}`, text, path: path9, page, types: ["video"], tokens: estimateTokens(text), hash: hashText(text), media: { element: elementId2, t0, t1 } });
2551
+ const path10 = [...crumb, ...el.title ? [String(el.title)] : [], ...chapter ? [String(chapter.title)] : []];
2552
+ out.push({ id: `${elementId2}@${Math.round(t0)}`, text, path: path10, page, types: ["video"], tokens: estimateTokens(text), hash: hashText(text), media: { element: elementId2, t0, t1 } });
2277
2553
  win = [];
2278
2554
  };
2279
2555
  for (const sg of segs) {
@@ -2342,8 +2618,14 @@ function serializeElement(el) {
2342
2618
  }
2343
2619
  case "checkbox":
2344
2620
  return `${e.checked ? "[x]" : "[ ]"} ${e.label ?? ""}`.trim();
2345
- case "image":
2346
- return e.alt ? `[image: ${e.alt}]` : "";
2621
+ case "image": {
2622
+ const parts = [];
2623
+ if (e.alt) parts.push(`[image: ${e.alt}]`);
2624
+ if (e.caption) parts.push(String(e.caption).trim());
2625
+ const ocr = (e.ocr?.blocks || []).map((b) => String(b.text ?? "").trim()).filter(Boolean).join("\n");
2626
+ if (ocr) parts.push(ocr);
2627
+ return parts.join("\n");
2628
+ }
2347
2629
  case "video":
2348
2630
  return e.title ? `[video: ${e.title}]` : "";
2349
2631
  case "toc":
@@ -2463,34 +2745,62 @@ function chunkDocument(doc, options = {}) {
2463
2745
  }
2464
2746
  async function loadJdf(filePath) {
2465
2747
  if (filePath.toLowerCase().endsWith(".jdfx")) {
2466
- const zip = await JSZip.loadAsync(fs7.readFileSync(filePath));
2748
+ const zip = await JSZip.loadAsync(fs8.readFileSync(filePath));
2467
2749
  const docFile = zip.file(JDFX_DOCUMENT_PATH);
2468
2750
  if (!docFile) throw new Error(`Bundle missing ${JDFX_DOCUMENT_PATH}`);
2469
2751
  return JSON.parse(await docFile.async("string"));
2470
2752
  }
2471
- return JSON.parse(fs7.readFileSync(filePath, "utf-8"));
2753
+ return JSON.parse(fs8.readFileSync(filePath, "utf-8"));
2754
+ }
2755
+ function mediaCoverage(doc) {
2756
+ const cov = { images: { total: 0, covered: 0, missing: [] }, videos: { total: 0, covered: 0, missing: [] } };
2757
+ const walk2 = (els, page) => {
2758
+ for (const el of els ?? []) {
2759
+ if (el?.type === "image") {
2760
+ cov.images.total++;
2761
+ const has = !!(el.caption && String(el.caption).trim()) || !!el.ocr?.blocks?.some((b) => String(b.text ?? "").trim());
2762
+ if (has) cov.images.covered++;
2763
+ else cov.images.missing.push({ id: el.id, page, alt: el.alt });
2764
+ } else if (el?.type === "video") {
2765
+ cov.videos.total++;
2766
+ if (el.transcript?.segments?.length) cov.videos.covered++;
2767
+ else cov.videos.missing.push({ id: el.id, page, title: el.title });
2768
+ }
2769
+ if (el?.elements) walk2(el.elements, page);
2770
+ }
2771
+ };
2772
+ doc.pages.forEach((p, i) => walk2(p.elements, i + 1));
2773
+ return cov;
2774
+ }
2775
+ function coverageSummary(cov) {
2776
+ const parts = [];
2777
+ if (cov.images.missing.length) parts.push(`${cov.images.missing.length} of ${cov.images.total} image(s) have no caption/OCR text \u2192 jdf describe`);
2778
+ if (cov.videos.missing.length) parts.push(`${cov.videos.missing.length} of ${cov.videos.total} video(s) have no transcript \u2192 jdf transcribe`);
2779
+ return parts.length ? parts.join("; ") : null;
2472
2780
  }
2473
2781
  async function chunkFile(inputPath, opts = {}) {
2474
- const input = path7.resolve(inputPath);
2475
- if (!fs7.existsSync(input)) throw new Error(`File not found: ${input}`);
2782
+ const input = path8.resolve(inputPath);
2783
+ if (!fs8.existsSync(input)) throw new Error(`File not found: ${input}`);
2476
2784
  const doc = await loadJdf(input);
2477
2785
  const strategy = opts.strategy ?? "section";
2478
2786
  const chunks = chunkDocument(doc, { strategy, maxTokens: opts.maxTokens, transcriptWindowSec: opts.transcriptWindowSec });
2787
+ const gap = coverageSummary(mediaCoverage(doc));
2788
+ if (gap) console.warn(` ! media without text (skipped by retrieval): ${gap}`);
2479
2789
  const format = opts.format ?? "jsonl";
2480
2790
  console.log(`Chunking: ${input}`);
2481
2791
  console.log(`Strategy: ${strategy}${opts.maxTokens ? ` (max ${opts.maxTokens} tokens)` : ""}`);
2482
2792
  if (format === "inline") {
2483
- const out = opts.output ? path7.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".jdf");
2793
+ const out = opts.output ? path8.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".jdf");
2484
2794
  const withIndex = { ...doc, index: { chunker: `jdf-${strategy}-v1`, chunks } };
2485
- fs7.writeFileSync(out, JSON.stringify(withIndex, null, 2));
2795
+ fs8.writeFileSync(out, JSON.stringify(withIndex, null, 2));
2486
2796
  console.log(`Output: ${out} (${chunks.length} chunks in "index" block)`);
2487
2797
  } else if (format === "json") {
2488
- const out = opts.output ? path7.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".chunks.json");
2489
- fs7.writeFileSync(out, JSON.stringify(chunks, null, 2));
2798
+ const out = opts.output ? path8.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".chunks.json");
2799
+ fs8.writeFileSync(out, JSON.stringify(chunks, null, 2));
2490
2800
  console.log(`Output: ${out} (${chunks.length} chunks)`);
2491
2801
  } else {
2492
- const out = opts.output ? path7.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".chunks.jsonl");
2493
- fs7.writeFileSync(out, chunks.map((c) => JSON.stringify(c)).join("\n") + "\n");
2802
+ const out = opts.output ? path8.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".chunks.jsonl");
2803
+ fs8.writeFileSync(out, chunks.map((c) => JSON.stringify(c)).join("\n") + "\n");
2494
2804
  console.log(`Output: ${out} (${chunks.length} chunks)`);
2495
2805
  }
2496
2806
  const totalTokens = chunks.reduce((a, c) => a + c.tokens, 0);
@@ -2513,11 +2823,11 @@ async function embedBatch(provider, model, inputs) {
2513
2823
  throw new Error(`Unknown embedding provider: ${provider}`);
2514
2824
  }
2515
2825
  }
2516
- var OLLAMA_HOST = process.env.OLLAMA_HOST || "http://localhost:11434";
2826
+ var OLLAMA_HOST2 = process.env.OLLAMA_HOST || "http://localhost:11434";
2517
2827
  var OLLAMA_CONTAINER = "jdf-ollama";
2518
2828
  async function ollamaUp() {
2519
2829
  try {
2520
- const res = await fetch(`${OLLAMA_HOST}/api/tags`, { signal: AbortSignal.timeout(1500) });
2830
+ const res = await fetch(`${OLLAMA_HOST2}/api/tags`, { signal: AbortSignal.timeout(1500) });
2521
2831
  return res.ok;
2522
2832
  } catch {
2523
2833
  return false;
@@ -2542,12 +2852,12 @@ async function ensureOllama(model, autoStart) {
2542
2852
  docker run -d --name ${OLLAMA_CONTAINER} -p 11434:11434 -v jdf-ollama:/root/.ollama ollama/ollama
2543
2853
  docker exec ${OLLAMA_CONTAINER} ollama pull ${model}`;
2544
2854
  if (!autoStart) {
2545
- throw new Error(`Ollama isn't running at ${OLLAMA_HOST} and --no-auto-start was given.
2855
+ throw new Error(`Ollama isn't running at ${OLLAMA_HOST2} and --no-auto-start was given.
2546
2856
  ${manualHint}`);
2547
2857
  }
2548
2858
  if (!dockerReady()) {
2549
2859
  throw new Error(
2550
- `Ollama isn't running at ${OLLAMA_HOST}, and the Docker daemon isn't available to auto-start it.
2860
+ `Ollama isn't running at ${OLLAMA_HOST2}, and the Docker daemon isn't available to auto-start it.
2551
2861
  ${manualHint}`
2552
2862
  );
2553
2863
  }
@@ -2584,7 +2894,7 @@ ${manualHint}`);
2584
2894
  }
2585
2895
  async function ollamaPull(model) {
2586
2896
  try {
2587
- const show = await fetch(`${OLLAMA_HOST}/api/show`, {
2897
+ const show = await fetch(`${OLLAMA_HOST2}/api/show`, {
2588
2898
  method: "POST",
2589
2899
  headers: { "Content-Type": "application/json" },
2590
2900
  body: JSON.stringify({ name: model })
@@ -2593,7 +2903,7 @@ async function ollamaPull(model) {
2593
2903
  } catch {
2594
2904
  }
2595
2905
  console.log(`Pulling embedding model "${model}" into Ollama (first run only)\u2026`);
2596
- const res = await fetch(`${OLLAMA_HOST}/api/pull`, {
2906
+ const res = await fetch(`${OLLAMA_HOST2}/api/pull`, {
2597
2907
  method: "POST",
2598
2908
  headers: { "Content-Type": "application/json" },
2599
2909
  body: JSON.stringify({ name: model, stream: false })
@@ -2604,7 +2914,7 @@ async function ollamaPull(model) {
2604
2914
  async function embedOllama(model, inputs) {
2605
2915
  const out = [];
2606
2916
  for (const text of inputs) {
2607
- const res = await fetch(`${OLLAMA_HOST}/api/embeddings`, {
2917
+ const res = await fetch(`${OLLAMA_HOST2}/api/embeddings`, {
2608
2918
  method: "POST",
2609
2919
  headers: { "Content-Type": "application/json" },
2610
2920
  body: JSON.stringify({ model, prompt: text })
@@ -2662,17 +2972,17 @@ async function embedOpenAI(model, inputs) {
2662
2972
  }
2663
2973
  async function loadJdf2(filePath) {
2664
2974
  if (filePath.toLowerCase().endsWith(".jdfx")) {
2665
- const zip = await JSZip.loadAsync(fs7.readFileSync(filePath));
2975
+ const zip = await JSZip.loadAsync(fs8.readFileSync(filePath));
2666
2976
  const docFile = zip.file(JDFX_DOCUMENT_PATH);
2667
2977
  if (!docFile) throw new Error(`Bundle missing ${JDFX_DOCUMENT_PATH}`);
2668
2978
  return JSON.parse(await docFile.async("string"));
2669
2979
  }
2670
- return JSON.parse(fs7.readFileSync(filePath, "utf-8"));
2980
+ return JSON.parse(fs8.readFileSync(filePath, "utf-8"));
2671
2981
  }
2672
2982
  function loadCache(cachePath) {
2673
2983
  try {
2674
- if (!fs7.existsSync(cachePath)) return null;
2675
- return JSON.parse(fs7.readFileSync(cachePath, "utf-8"));
2984
+ if (!fs8.existsSync(cachePath)) return null;
2985
+ return JSON.parse(fs8.readFileSync(cachePath, "utf-8"));
2676
2986
  } catch {
2677
2987
  return null;
2678
2988
  }
@@ -2683,15 +2993,15 @@ function batched(items, size) {
2683
2993
  return out;
2684
2994
  }
2685
2995
  async function embedFile(inputPath, opts = {}) {
2686
- const input = path7.resolve(inputPath);
2687
- if (!fs7.existsSync(input)) throw new Error(`File not found: ${input}`);
2996
+ const input = path8.resolve(inputPath);
2997
+ if (!fs8.existsSync(input)) throw new Error(`File not found: ${input}`);
2688
2998
  const provider = opts.provider ?? "ollama";
2689
2999
  const model = opts.model ?? DEFAULT_MODEL[provider];
2690
3000
  const strategy = opts.strategy ?? "section";
2691
3001
  const doc = await loadJdf2(input);
2692
3002
  const chunks = chunkDocument(doc, { strategy, maxTokens: opts.maxTokens, transcriptWindowSec: opts.transcriptWindowSec });
2693
- const output = opts.output ? path7.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".embeddings.json");
2694
- const cachePath = opts.cache ? path7.resolve(opts.cache) : output;
3003
+ const output = opts.output ? path8.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".embeddings.json");
3004
+ const cachePath = opts.cache ? path8.resolve(opts.cache) : output;
2695
3005
  console.log(`Embedding: ${input}`);
2696
3006
  console.log(`Provider: ${provider} / ${model}${provider === "ollama" ? " (local \u2014 no data leaves this machine)" : " (remote API)"}`);
2697
3007
  console.log(`Strategy: ${strategy} \u2192 ${chunks.length} chunks`);
@@ -2730,7 +3040,7 @@ async function embedFile(inputPath, opts = {}) {
2730
3040
  chunker: `jdf-${strategy}-v1`,
2731
3041
  vectors
2732
3042
  };
2733
- fs7.writeFileSync(output, JSON.stringify(sidecar));
3043
+ fs8.writeFileSync(output, JSON.stringify(sidecar));
2734
3044
  console.log(`
2735
3045
  Done! ${Object.keys(vectors).length} vectors (${dims}-dim) \u2192 ${output}`);
2736
3046
  return sidecar;
@@ -2774,9 +3084,9 @@ function parseChapters(text) {
2774
3084
  return { t: toSec(m[1]), title: m[2].trim() };
2775
3085
  });
2776
3086
  }
2777
- async function loadDoc(file) {
3087
+ async function loadDoc2(file) {
2778
3088
  if (file.toLowerCase().endsWith(".jdfx")) {
2779
- const zip = await JSZip.loadAsync(fs7.readFileSync(file));
3089
+ const zip = await JSZip.loadAsync(fs8.readFileSync(file));
2780
3090
  const f = zip.file(JDFX_DOCUMENT_PATH);
2781
3091
  if (!f) throw new Error(`Bundle missing ${JDFX_DOCUMENT_PATH}`);
2782
3092
  const doc = JSON.parse(await f.async("string"));
@@ -2792,7 +3102,7 @@ async function loadDoc(file) {
2792
3102
  }
2793
3103
  return { doc, bundle: true, zip };
2794
3104
  }
2795
- return { doc: JSON.parse(fs7.readFileSync(file, "utf-8")), bundle: false };
3105
+ return { doc: JSON.parse(fs8.readFileSync(file, "utf-8")), bundle: false };
2796
3106
  }
2797
3107
  function findVideos(doc) {
2798
3108
  const out = [];
@@ -2806,27 +3116,27 @@ function findVideos(doc) {
2806
3116
  return out;
2807
3117
  }
2808
3118
  async function clipToTempFile(doc, el, docDir) {
2809
- const tmp = path7.join(fs7.mkdtempSync(path7.join(__require("os").tmpdir(), "jdf-transcribe-")), "clip.mp4");
3119
+ const tmp = path8.join(fs8.mkdtempSync(path8.join(__require("os").tmpdir(), "jdf-transcribe-")), "clip.mp4");
2810
3120
  const res = el.resource ? doc.resources?.videos?.[el.resource] ?? doc.resources?.images?.[el.resource] : void 0;
2811
3121
  if (res?.data) {
2812
- fs7.writeFileSync(tmp, Buffer.from(res.data.replace(/^data:[^,]*,/, ""), "base64"));
3122
+ fs8.writeFileSync(tmp, Buffer.from(res.data.replace(/^data:[^,]*,/, ""), "base64"));
2813
3123
  return tmp;
2814
3124
  }
2815
- if (res?.path) return path7.resolve(docDir, res.path);
3125
+ if (res?.path) return path8.resolve(docDir, res.path);
2816
3126
  const src = el.src;
2817
3127
  if (!src) return null;
2818
3128
  if (src.startsWith("data:")) {
2819
- fs7.writeFileSync(tmp, Buffer.from(src.replace(/^data:[^,]*,/, ""), "base64"));
3129
+ fs8.writeFileSync(tmp, Buffer.from(src.replace(/^data:[^,]*,/, ""), "base64"));
2820
3130
  return tmp;
2821
3131
  }
2822
3132
  if (/^https?:\/\//i.test(src)) {
2823
3133
  const r = await fetch(src);
2824
3134
  if (!r.ok) throw new Error(`download failed ${r.status}: ${src}`);
2825
- fs7.writeFileSync(tmp, Buffer.from(await r.arrayBuffer()));
3135
+ fs8.writeFileSync(tmp, Buffer.from(await r.arrayBuffer()));
2826
3136
  return tmp;
2827
3137
  }
2828
- const local = path7.resolve(docDir, src);
2829
- return fs7.existsSync(local) ? local : null;
3138
+ const local = path8.resolve(docDir, src);
3139
+ return fs8.existsSync(local) ? local : null;
2830
3140
  }
2831
3141
  function whisperCli(clip, model, language, prompt2) {
2832
3142
  const ffmpeg = spawnSync("ffmpeg", ["-version"]);
@@ -2841,7 +3151,7 @@ function whisperCli(clip, model, language, prompt2) {
2841
3151
  const run = spawnSync("whisper-cli", args, { encoding: "utf-8" });
2842
3152
  if (run.error) throw new Error("whisper-cli not found \u2014 install whisper.cpp (brew install whisper-cpp) or use --from / --provider openai");
2843
3153
  if (run.status !== 0) throw new Error(`whisper-cli failed: ${run.stderr.slice(-400)}`);
2844
- const j = JSON.parse(fs7.readFileSync(wav.replace(/\.wav$/, "") + ".json", "utf-8"));
3154
+ const j = JSON.parse(fs8.readFileSync(wav.replace(/\.wav$/, "") + ".json", "utf-8"));
2845
3155
  const segs = j.transcription ?? j.segments ?? [];
2846
3156
  const ms = (x) => typeof x === "number" ? x / 1e3 : toSec(String(x).replace(",", "."));
2847
3157
  return segs.map((sg) => ({ t0: ms(sg.offsets?.from ?? sg.start), t1: ms(sg.offsets?.to ?? sg.end), text: String(sg.text ?? "").trim() })).filter((sg) => sg.text);
@@ -2851,7 +3161,7 @@ async function openaiTranscribe(clip, model, language, prompt2) {
2851
3161
  if (!key) throw new Error("OPENAI_API_KEY is not set");
2852
3162
  const base = process.env.OPENAI_BASE_URL || "https://api.openai.com/v1";
2853
3163
  const form = new FormData();
2854
- form.append("file", new Blob([fs7.readFileSync(clip)]), path7.basename(clip));
3164
+ form.append("file", new Blob([fs8.readFileSync(clip)]), path8.basename(clip));
2855
3165
  form.append("model", model || "whisper-1");
2856
3166
  form.append("response_format", "verbose_json");
2857
3167
  form.append("timestamp_granularities[]", "segment");
@@ -2863,9 +3173,9 @@ async function openaiTranscribe(clip, model, language, prompt2) {
2863
3173
  return (j.segments ?? []).map((sg) => ({ t0: Number(sg.start), t1: Number(sg.end), text: String(sg.text).trim() })).filter((sg) => sg.text);
2864
3174
  }
2865
3175
  async function transcribeFile(inputPath, opts = {}) {
2866
- const input = path7.resolve(inputPath);
2867
- if (!fs7.existsSync(input)) throw new Error(`File not found: ${input}`);
2868
- const { doc, bundle } = await loadDoc(input);
3176
+ const input = path8.resolve(inputPath);
3177
+ if (!fs8.existsSync(input)) throw new Error(`File not found: ${input}`);
3178
+ const { doc, bundle } = await loadDoc2(input);
2869
3179
  const videos = findVideos(doc);
2870
3180
  if (!videos.length) throw new Error("document has no video element");
2871
3181
  let target = videos[0];
@@ -2881,40 +3191,40 @@ async function transcribeFile(inputPath, opts = {}) {
2881
3191
  let segments;
2882
3192
  let source;
2883
3193
  if (opts.from) {
2884
- segments = parseSubtitles(fs7.readFileSync(path7.resolve(opts.from), "utf-8"), opts.from);
2885
- source = `${path7.extname(opts.from).slice(1).toLowerCase() || "file"}-import`;
3194
+ segments = parseSubtitles(fs8.readFileSync(path8.resolve(opts.from), "utf-8"), opts.from);
3195
+ source = `${path8.extname(opts.from).slice(1).toLowerCase() || "file"}-import`;
2886
3196
  } else {
2887
3197
  const provider = opts.provider ?? "whisper-cli";
2888
- const clip = await clipToTempFile(doc, target.el, path7.dirname(input));
3198
+ const clip = await clipToTempFile(doc, target.el, path8.dirname(input));
2889
3199
  if (!clip) throw new Error("could not locate the clip bytes (no bundled asset, data URL, local path or http URL) \u2014 use --from to import subtitles instead");
2890
3200
  segments = provider === "openai" ? await openaiTranscribe(clip, opts.model, opts.language, opts.prompt) : whisperCli(clip, opts.model, opts.language, opts.prompt);
2891
- source = provider === "openai" ? `openai:${opts.model || "whisper-1"}` : `whisper-cli${opts.model ? ":" + path7.basename(opts.model) : ""}`;
3201
+ source = provider === "openai" ? `openai:${opts.model || "whisper-1"}` : `whisper-cli${opts.model ? ":" + path8.basename(opts.model) : ""}`;
2892
3202
  }
2893
3203
  segments.sort((a, b) => a.t0 - b.t0);
2894
3204
  const transcript = { ...opts.language ? { language: opts.language } : {}, source, created: (/* @__PURE__ */ new Date()).toISOString(), segments };
2895
3205
  target.el.transcript = transcript;
2896
- if (opts.chapters) target.el.chapters = parseChapters(fs7.readFileSync(path7.resolve(opts.chapters), "utf-8"));
3206
+ if (opts.chapters) target.el.chapters = parseChapters(fs8.readFileSync(path8.resolve(opts.chapters), "utf-8"));
2897
3207
  if (!target.el.id) target.el.id = `video-${target.index + 1}`;
2898
- const output = opts.output ? path7.resolve(opts.output) : input;
3208
+ const output = opts.output ? path8.resolve(opts.output) : input;
2899
3209
  if (output.toLowerCase().endsWith(".jdfx") || bundle && !opts.output) {
2900
3210
  const { bytes } = await packJdfx(doc);
2901
- fs7.writeFileSync(output, bytes);
3211
+ fs8.writeFileSync(output, bytes);
2902
3212
  } else {
2903
- fs7.writeFileSync(output, JSON.stringify(doc, null, 2));
3213
+ fs8.writeFileSync(output, JSON.stringify(doc, null, 2));
2904
3214
  }
2905
3215
  const dur = segments.length ? segments[segments.length - 1].t1 : 0;
2906
- console.log(`Transcribed: ${path7.basename(input)} \u2192 element "${target.el.id}" (${segments.length} segments, ${Math.round(dur)} s, source ${source})`);
3216
+ console.log(`Transcribed: ${path8.basename(input)} \u2192 element "${target.el.id}" (${segments.length} segments, ${Math.round(dur)} s, source ${source})`);
2907
3217
  if (target.el.chapters) console.log(`Chapters: ${target.el.chapters.length}`);
2908
3218
  console.log(`Output: ${output}
2909
- Next: jdf chunk ${path7.basename(output)} # transcript \u2192 time-windowed chunks with media.t0/t1`);
3219
+ Next: jdf chunk ${path8.basename(output)} # transcript \u2192 time-windowed chunks with media.t0/t1`);
2910
3220
  return transcript;
2911
3221
  }
2912
3222
  var CONFIG_NAME = "jdf.rag.json";
2913
3223
  var OUT_DIR = ".jdf-rag";
2914
3224
  function walk(dir, acc = []) {
2915
- for (const ent of fs7.readdirSync(dir, { withFileTypes: true })) {
3225
+ for (const ent of fs8.readdirSync(dir, { withFileTypes: true })) {
2916
3226
  if (ent.name === "node_modules" || ent.name === OUT_DIR || ent.name.startsWith(".")) continue;
2917
- const p = path7.join(dir, ent.name);
3227
+ const p = path8.join(dir, ent.name);
2918
3228
  if (ent.isDirectory()) walk(p, acc);
2919
3229
  else if (/\.(jdf|jdfx)$/i.test(ent.name)) acc.push(p);
2920
3230
  }
@@ -2922,10 +3232,10 @@ function walk(dir, acc = []) {
2922
3232
  }
2923
3233
  async function readDoc(file) {
2924
3234
  if (file.toLowerCase().endsWith(".jdfx")) {
2925
- const zip = await JSZip.loadAsync(fs7.readFileSync(file));
3235
+ const zip = await JSZip.loadAsync(fs8.readFileSync(file));
2926
3236
  return JSON.parse(await zip.file(JDFX_DOCUMENT_PATH).async("string"));
2927
3237
  }
2928
- return JSON.parse(fs7.readFileSync(file, "utf-8"));
3238
+ return JSON.parse(fs8.readFileSync(file, "utf-8"));
2929
3239
  }
2930
3240
  function videosIn(doc) {
2931
3241
  const out = [];
@@ -2939,29 +3249,44 @@ function videosIn(doc) {
2939
3249
  return out;
2940
3250
  }
2941
3251
  async function ragFolder(dirPath, cli = {}) {
2942
- const dir = path7.resolve(dirPath);
2943
- if (!fs7.existsSync(dir) || !fs7.statSync(dir).isDirectory()) throw new Error(`Not a directory: ${dir}`);
2944
- const cfgPath = path7.join(dir, CONFIG_NAME);
2945
- const cfg = fs7.existsSync(cfgPath) ? JSON.parse(fs7.readFileSync(cfgPath, "utf-8")) : {};
3252
+ const dir = path8.resolve(dirPath);
3253
+ if (!fs8.existsSync(dir) || !fs8.statSync(dir).isDirectory()) throw new Error(`Not a directory: ${dir}`);
3254
+ const cfgPath = path8.join(dir, CONFIG_NAME);
3255
+ const cfg = fs8.existsSync(cfgPath) ? JSON.parse(fs8.readFileSync(cfgPath, "utf-8")) : {};
2946
3256
  const opts = { ...cfg, ...Object.fromEntries(Object.entries(cli).filter(([, v]) => v !== void 0)) };
2947
3257
  const provider = opts.provider ?? "ollama";
2948
3258
  const transcribe = opts.transcribe ?? "none";
2949
- const outDir = path7.resolve(opts.out ?? path7.join(dir, OUT_DIR));
3259
+ const outDir = path8.resolve(opts.out ?? path8.join(dir, OUT_DIR));
3260
+ const ocr = opts.ocr ?? "none";
3261
+ const caption = opts.caption ?? "none";
2950
3262
  const files = walk(dir);
2951
3263
  console.log(`jdf rag: ${dir}
2952
- files: ${files.length} (.jdf/.jdfx)${fs7.existsSync(cfgPath) ? `
3264
+ files: ${files.length} (.jdf/.jdfx)${fs8.existsSync(cfgPath) ? `
2953
3265
  config: ${CONFIG_NAME}` : ""}
2954
3266
  embeddings: ${opts.noEmbed ? "skipped (--no-embed)" : `${provider}${opts.model ? " / " + opts.model : ""}`}
2955
- transcribe: ${transcribe}${opts.dryRun ? "\n DRY RUN \u2014 nothing written" : ""}
3267
+ transcribe: ${transcribe} ocr: ${ocr} caption: ${caption}${opts.dryRun ? "\n DRY RUN \u2014 nothing written" : ""}
2956
3268
  `);
2957
3269
  if (!files.length) {
2958
3270
  console.log("Nothing to do.");
2959
3271
  return;
2960
3272
  }
2961
- const manifest = { dir, created: (/* @__PURE__ */ new Date()).toISOString(), provider: opts.noEmbed ? null : provider, model: opts.model ?? null, strategy: opts.strategy ?? "section", transcribe, files: [], totals: { files: files.length, chunks: 0, videoChunks: 0, videos: 0, transcribed: 0, untranscribed: 0 } };
3273
+ const manifest = {
3274
+ dir,
3275
+ created: (/* @__PURE__ */ new Date()).toISOString(),
3276
+ provider: opts.noEmbed ? null : provider,
3277
+ model: opts.model ?? null,
3278
+ strategy: opts.strategy ?? "section",
3279
+ transcribe,
3280
+ ocr,
3281
+ caption,
3282
+ files: [],
3283
+ totals: { files: files.length, chunks: 0, videoChunks: 0, videos: 0, transcribed: 0, untranscribed: 0, images: 0, described: 0, imagesWithoutText: 0 },
3284
+ /** Every media element that retrieval would still skip, by file — the thing to fix before shipping an index. */
3285
+ mediaWithoutText: []
3286
+ };
2962
3287
  const indexLines = [];
2963
3288
  for (const file of files) {
2964
- const rel = path7.relative(dir, file);
3289
+ const rel = path8.relative(dir, file);
2965
3290
  const doc = await readDoc(file);
2966
3291
  const vids = videosIn(doc);
2967
3292
  let transcribedHere = 0;
@@ -2985,20 +3310,36 @@ async function ragFolder(dirPath, cli = {}) {
2985
3310
  }
2986
3311
  manifest.totals.videos += vids.length;
2987
3312
  manifest.totals.transcribed += transcribedHere;
3313
+ const covBefore = mediaCoverage(doc);
3314
+ let describedHere = 0;
3315
+ if (covBefore.images.missing.length && (ocr !== "none" || caption !== "none") && !opts.dryRun) {
3316
+ try {
3317
+ await describeFile(file, { ocr, caption, captionModel: opts.captionModel, quiet: true });
3318
+ describedHere = covBefore.images.missing.length;
3319
+ } catch (e) {
3320
+ console.warn(` ! ${rel}: describe failed: ${e.message}`);
3321
+ }
3322
+ }
3323
+ const covAfter = transcribedHere || describedHere ? mediaCoverage(await readDoc(file)) : covBefore;
3324
+ manifest.totals.images += covAfter.images.total;
3325
+ manifest.totals.described += describedHere;
3326
+ manifest.totals.imagesWithoutText += covAfter.images.missing.length;
3327
+ for (const m of covAfter.images.missing) manifest.mediaWithoutText.push({ file: rel, type: "image", ...m });
3328
+ for (const m of covAfter.videos.missing) manifest.mediaWithoutText.push({ file: rel, type: "video", ...m });
2988
3329
  if (opts.dryRun) {
2989
3330
  manifest.files.push({ file: rel, videos: vids.length, wouldTranscribe: transcribedHere });
2990
3331
  continue;
2991
3332
  }
2992
3333
  const chunkOpts = { strategy: opts.strategy, maxTokens: opts.maxTokens, transcriptWindowSec: opts.transcriptWindowSec };
2993
- const chunkOut = path7.join(outDir, "chunks", rel.replace(/\.(jdf|jdfx)$/i, ".chunks.jsonl"));
2994
- fs7.mkdirSync(path7.dirname(chunkOut), { recursive: true });
3334
+ const chunkOut = path8.join(outDir, "chunks", rel.replace(/\.(jdf|jdfx)$/i, ".chunks.jsonl"));
3335
+ fs8.mkdirSync(path8.dirname(chunkOut), { recursive: true });
2995
3336
  let chunks;
2996
3337
  if (opts.noEmbed) {
2997
3338
  chunks = await chunkFile(file, { ...chunkOpts, format: "jsonl", output: chunkOut });
2998
3339
  } else {
2999
3340
  const side = await embedFile(file, { ...chunkOpts, provider, model: opts.model, incremental: true });
3000
3341
  chunks = await chunkFile(file, { ...chunkOpts, format: "jsonl", output: chunkOut });
3001
- manifest.files.push({ file: rel, chunks: chunks.length, vectors: Object.keys(side.vectors).length, sidecar: path7.relative(dir, file.replace(/\.(jdf|jdfx)$/i, ".embeddings.json")), videos: vids.length, transcribed: transcribedHere });
3342
+ manifest.files.push({ file: rel, chunks: chunks.length, vectors: Object.keys(side.vectors).length, sidecar: path8.relative(dir, file.replace(/\.(jdf|jdfx)$/i, ".embeddings.json")), videos: vids.length, transcribed: transcribedHere });
3002
3343
  }
3003
3344
  if (opts.noEmbed) manifest.files.push({ file: rel, chunks: chunks.length, videos: vids.length, transcribed: transcribedHere });
3004
3345
  for (const c of chunks) {
@@ -3008,15 +3349,27 @@ async function ragFolder(dirPath, cli = {}) {
3008
3349
  }
3009
3350
  }
3010
3351
  if (!opts.dryRun) {
3011
- fs7.mkdirSync(outDir, { recursive: true });
3012
- fs7.writeFileSync(path7.join(outDir, "index.jsonl"), indexLines.join("\n") + (indexLines.length ? "\n" : ""));
3013
- fs7.writeFileSync(path7.join(outDir, "manifest.json"), JSON.stringify(manifest, null, 2) + "\n");
3352
+ fs8.mkdirSync(outDir, { recursive: true });
3353
+ fs8.writeFileSync(path8.join(outDir, "index.jsonl"), indexLines.join("\n") + (indexLines.length ? "\n" : ""));
3354
+ fs8.writeFileSync(path8.join(outDir, "manifest.json"), JSON.stringify(manifest, null, 2) + "\n");
3014
3355
  }
3015
3356
  const t = manifest.totals;
3016
3357
  console.log(`
3017
- Done. ${t.files} files \u2192 ${t.chunks} chunks (${t.videoChunks} from video transcripts); videos ${t.videos}, transcribed now ${t.transcribed}, still without transcript ${t.untranscribed}${t.untranscribed && transcribe === "none" ? " (pass --transcribe whisper-cli|openai, or jdf transcribe --from subs.srt)" : ""}.`);
3018
- if (!opts.dryRun) console.log(`Index: ${path7.join(outDir, "index.jsonl")}
3019
- Report: ${path7.join(outDir, "manifest.json")}${opts.noEmbed ? "" : `
3358
+ Done. ${t.files} files \u2192 ${t.chunks} chunks (${t.videoChunks} from video transcripts).`);
3359
+ console.log(`Media coverage: videos ${t.videos - t.untranscribed}/${t.videos} with transcript (transcribed now ${t.transcribed}), images ${t.images - t.imagesWithoutText}/${t.images} with caption/OCR (described now ${t.described}).`);
3360
+ if (manifest.mediaWithoutText.length) {
3361
+ console.log(`
3362
+ ! ${manifest.mediaWithoutText.length} media element(s) still have NO text \u2014 retrieval will skip them:`);
3363
+ for (const m of manifest.mediaWithoutText.slice(0, 12)) console.log(` ${m.file} \xB7 ${m.type} ${m.id ?? ""} (page ${m.page})${m.title ? ` "${m.title}"` : m.alt ? ` alt="${m.alt}"` : ""}`);
3364
+ if (manifest.mediaWithoutText.length > 12) console.log(` \u2026 ${manifest.mediaWithoutText.length - 12} more in manifest.json`);
3365
+ console.log(` fix: jdf rag <dir> --transcribe whisper-cli|openai --ocr tesseract --caption ollama (or jdf transcribe / jdf describe per file)`);
3366
+ if (opts.strict) {
3367
+ console.error(`--strict: failing because media without text remains.`);
3368
+ process.exitCode = 1;
3369
+ }
3370
+ }
3371
+ if (!opts.dryRun) console.log(`Index: ${path8.join(outDir, "index.jsonl")}
3372
+ Report: ${path8.join(outDir, "manifest.json")}${opts.noEmbed ? "" : `
3020
3373
  Vectors: one <file>.embeddings.json next to each document (incremental \u2014 re-run any time)`}`);
3021
3374
  }
3022
3375
 
@@ -3032,8 +3385,11 @@ The CLI exists for these workflows:
3032
3385
  \u2022 JDF \u2192 vectors embed those chunks, incrementally, for a vector store.
3033
3386
  \u2022 video \u2192 text attach a time-stamped transcript to a video element so
3034
3387
  RAG retrieves "video at 02:13", not just "a video".
3388
+ \u2022 image \u2192 text OCR + a vision caption for every image so charts and
3389
+ scanned pages are retrievable, not skipped.
3035
3390
  \u2022 folder \u2192 index one command over a directory of .jdf/.jdfx: transcribe,
3036
- chunk, embed incrementally, write .jdf-rag/index.jsonl.
3391
+ describe, chunk, embed incrementally, write .jdf-rag/index.jsonl.
3392
+ Reports media coverage; --strict fails when anything has no text.
3037
3393
 
3038
3394
  Usage:
3039
3395
  jdf validate <file.jdf>
@@ -3041,7 +3397,8 @@ Usage:
3041
3397
  jdf chunk <file.{jdf,jdfx}> [--strategy section|element|fixed] [--format jsonl|json|inline] [--max-tokens N] [-o out]
3042
3398
  jdf embed <file.{jdf,jdfx}> [--provider ollama|openai] [--model NAME] [--strategy \u2026] [--incremental] [-o out]
3043
3399
  jdf transcribe <file.{jdf,jdfx}> [--from subs.srt|.vtt|.json] [--provider whisper-cli|openai] [--element ID] [--chapters FILE] [-o out]
3044
- jdf rag <dir> [--provider ollama|openai] [--model NAME] [--transcribe none|whisper-cli|openai] [--no-embed] [--dry-run] [--out DIR]
3400
+ jdf describe <file.{jdf,jdfx}> [--ocr tesseract|openai|none] [--caption ollama|openai|none] [--caption-model M] [--ocr-language eng] [--element ID] [--force] [-o out]
3401
+ jdf rag <dir> [--provider ollama|openai] [--model NAME] [--transcribe none|whisper-cli|openai] [--ocr none|tesseract|openai] [--caption none|ollama|openai] [--strict] [--no-embed] [--dry-run] [--out DIR]
3045
3402
  jdf --help
3046
3403
 
3047
3404
  Commands:
@@ -3050,7 +3407,8 @@ Commands:
3050
3407
  chunk Split a JDF document into retrieval-ready chunks (offline, deterministic)
3051
3408
  embed Compute embeddings for the chunks (local via Ollama by default)
3052
3409
  transcribe Store time-stamped text on a video element (import SRT/VTT/JSON, or run Whisper)
3053
- rag Make a whole folder retrieval-ready (finds .jdf/.jdfx, transcribes, chunks, embeds, indexes)
3410
+ describe Give images text: OCR blocks (tesseract.js, local) + a caption (Ollama vision model, local)
3411
+ rag Make a whole folder retrieval-ready (finds .jdf/.jdfx, transcribes, describes, chunks, embeds, indexes)
3054
3412
 
3055
3413
  Flags:
3056
3414
  -o, --output <path> Explicit output path
@@ -3077,6 +3435,12 @@ Flags:
3077
3435
  --prompt <text> transcribe: Whisper vocabulary hint (names, acronyms) \u2014 not an instruction
3078
3436
  --window <sec> chunk/embed/rag: transcript window per video chunk (default 45)
3079
3437
  --transcribe <p> rag: none (default) | whisper-cli | openai \u2014 for videos that have no transcript yet
3438
+ --ocr <p> describe/rag/convert: tesseract (local WASM) | openai | none
3439
+ --caption <p> describe/rag: ollama (local vision model, default moondream) | openai | none
3440
+ --caption-model describe/rag: vision model name (ollama: qwen2.5vl:3b, llava\u2026; openai: gpt-4o-mini\u2026)
3441
+ --ocr-language describe: tesseract language(s), e.g. eng, tur, eng+tur (default eng)
3442
+ --force describe: redo images that already have text
3443
+ --strict rag: exit 1 if any image/video is still without text after the run
3080
3444
  --no-embed rag: chunk + index only
3081
3445
  --dry-run rag: list what would happen, write nothing
3082
3446
  --out <dir> rag: index folder (default <dir>/.jdf-rag)
@@ -3096,9 +3460,11 @@ Examples:
3096
3460
  jdf embed report.jdf --provider openai --incremental
3097
3461
  jdf transcribe talk.jdfx --from talk.srt --chapters chapters.txt # then: jdf chunk talk.jdfx
3098
3462
  jdf transcribe talk.jdfx --provider openai --language en --prompt "JDF, jdfx, Ollama"
3099
- jdf rag ./knowledge-base --transcribe openai # whole folder \u2192 .jdf-rag/index.jsonl
3463
+ jdf describe report.jdfx # OCR (tesseract) + caption (Ollama qwen2.5vl), local
3464
+ jdf convert scan.pdf --ocr tesseract # scanned pages get OCR text instead of silence
3465
+ jdf rag ./knowledge-base --transcribe openai --ocr tesseract --caption ollama --strict
3100
3466
  `;
3101
- var BOOLEAN_FLAGS = /* @__PURE__ */ new Set(["help", "h", "json", "verbose", "skip-validate", "incremental", "no-auto-start", "drop-invisible-text", "no-embed", "dry-run"]);
3467
+ var BOOLEAN_FLAGS = /* @__PURE__ */ new Set(["help", "h", "json", "verbose", "skip-validate", "incremental", "no-auto-start", "drop-invisible-text", "no-embed", "dry-run", "force", "strict"]);
3102
3468
  function parseArgs(argv) {
3103
3469
  const positional = [];
3104
3470
  const flags = {};
@@ -3181,7 +3547,8 @@ async function main() {
3181
3547
  await importPdf(input, output, {
3182
3548
  forceJson,
3183
3549
  password: typeof flags.password === "string" ? flags.password : void 0,
3184
- dropInvisibleText: flags["drop-invisible-text"] === true
3550
+ dropInvisibleText: flags["drop-invisible-text"] === true,
3551
+ ocr: typeof flags.ocr === "string" ? flags.ocr : void 0
3185
3552
  });
3186
3553
  process.exit(0);
3187
3554
  } else if (lower.endsWith(".json")) {
@@ -3225,6 +3592,23 @@ async function main() {
3225
3592
  });
3226
3593
  process.exit(0);
3227
3594
  }
3595
+ case "describe": {
3596
+ const input = positional[0];
3597
+ if (!input) {
3598
+ console.error("Usage: jdf describe <file.{jdf,jdfx}> [--ocr tesseract|openai|none] [--caption ollama|openai|none] [--caption-model M] [--ocr-language eng] [--element id|n] [--force] [-o out]");
3599
+ process.exit(1);
3600
+ }
3601
+ await describeFile(input, {
3602
+ ocr: typeof flags.ocr === "string" ? flags.ocr : void 0,
3603
+ caption: typeof flags.caption === "string" ? flags.caption : void 0,
3604
+ captionModel: typeof flags["caption-model"] === "string" ? flags["caption-model"] : void 0,
3605
+ ocrLanguage: typeof flags["ocr-language"] === "string" ? flags["ocr-language"] : void 0,
3606
+ element: typeof flags.element === "string" ? flags.element : void 0,
3607
+ force: flags.force === true,
3608
+ output: typeof flags.output === "string" ? flags.output : void 0
3609
+ });
3610
+ process.exit(process.exitCode ?? 0);
3611
+ }
3228
3612
  case "rag": {
3229
3613
  const input = positional[0];
3230
3614
  if (!input) {
@@ -3241,11 +3625,15 @@ async function main() {
3241
3625
  transcribeModel: typeof flags["transcribe-model"] === "string" ? flags["transcribe-model"] : void 0,
3242
3626
  language: typeof flags.language === "string" ? flags.language : void 0,
3243
3627
  prompt: typeof flags.prompt === "string" ? flags.prompt : void 0,
3628
+ ocr: typeof flags.ocr === "string" ? flags.ocr : void 0,
3629
+ caption: typeof flags.caption === "string" ? flags.caption : void 0,
3630
+ captionModel: typeof flags["caption-model"] === "string" ? flags["caption-model"] : void 0,
3631
+ strict: flags.strict === true,
3244
3632
  noEmbed: flags["no-embed"] === true,
3245
3633
  dryRun: flags["dry-run"] === true,
3246
3634
  out: typeof flags.out === "string" ? flags.out : void 0
3247
3635
  });
3248
- process.exit(0);
3636
+ process.exit(process.exitCode ?? 0);
3249
3637
  }
3250
3638
  case "embed": {
3251
3639
  const input = positional[0];
@@ -343,7 +343,17 @@
343
343
  "anyOf": [{ "required": ["resource"] }, { "required": ["src"] }],
344
344
  "properties": {
345
345
  "type": { "const": "image" },
346
+ "id": { "type": "string" },
346
347
  "resource": { "type": "string" }, "src": { "type": "string" }, "alt": { "type": "string" },
348
+ "caption": { "type": "string" }, "captionSource": { "type": "string" },
349
+ "ocr": {
350
+ "type": "object",
351
+ "required": ["blocks"],
352
+ "properties": {
353
+ "language": { "type": "string" }, "source": { "type": "string" }, "created": { "type": "string" },
354
+ "blocks": { "type": "array", "items": { "type": "object", "required": ["text"], "properties": { "text": { "type": "string" }, "confidence": { "type": "number" }, "bbox": { "type": "object", "properties": { "x": { "type": "number" }, "y": { "type": "number" }, "w": { "type": "number" }, "h": { "type": "number" } } } } } }
355
+ }
356
+ },
347
357
  "position": { "$ref": "#/definitions/Position" },
348
358
  "width": { "type": "number" }, "height": { "type": "number" },
349
359
  "fit": { "type": "string", "enum": ["contain","cover","fill","none"] },
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@uurtech/jdf-cli",
3
- "version": "0.2.0",
3
+ "version": "0.2.2",
4
4
  "description": "Command-line tool for the JDF (JSON Document Format) — validate and convert documents.",
5
5
  "license": "MIT",
6
6
  "author": "Ugur Kazdal",
@@ -52,7 +52,8 @@
52
52
  "ajv": "^8.17.1",
53
53
  "ajv-formats": "^3.0.1",
54
54
  "jszip": "^3.10.1",
55
- "pdfjs-dist": "^4.10.38"
55
+ "pdfjs-dist": "^4.10.38",
56
+ "tesseract.js": "^7.0.0"
56
57
  },
57
58
  "devDependencies": {
58
59
  "@jdf/core": "workspace:*",