@uurtech/jdf-cli 0.2.1 → 0.2.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -1,12 +1,13 @@
1
1
  #!/usr/bin/env node
2
- import fs7 from 'fs';
3
- import path7 from 'path';
2
+ import fs8 from 'fs';
3
+ import path8 from 'path';
4
4
  import { fileURLToPath } from 'url';
5
5
  import Ajv from 'ajv';
6
6
  import addFormats from 'ajv-formats';
7
7
  import JSZip from 'jszip';
8
8
  import crypto, { createHash } from 'crypto';
9
9
  import { readFile } from 'fs/promises';
10
+ import os from 'os';
10
11
  import { execFileSync, spawnSync } from 'child_process';
11
12
 
12
13
  var __require = /* @__PURE__ */ ((x) => typeof require !== "undefined" ? require : typeof Proxy !== "undefined" ? new Proxy(x, {
@@ -23,17 +24,17 @@ var JDFX_MANIFEST_PATH = "manifest.json";
23
24
  var JDFX_ASSET_DIR = "assets";
24
25
 
25
26
  // src/commands/validate.ts
26
- var __dirname$1 = path7.dirname(fileURLToPath(import.meta.url));
27
+ var __dirname$1 = path8.dirname(fileURLToPath(import.meta.url));
27
28
  function resolveSchemaPath() {
28
- const bundled = path7.resolve(__dirname$1, "jdf-schema.json");
29
- if (fs7.existsSync(bundled)) return bundled;
30
- const dev = path7.resolve(__dirname$1, "../../../../spec/jdf-schema.json");
29
+ const bundled = path8.resolve(__dirname$1, "jdf-schema.json");
30
+ if (fs8.existsSync(bundled)) return bundled;
31
+ const dev = path8.resolve(__dirname$1, "../../../../spec/jdf-schema.json");
31
32
  return dev;
32
33
  }
33
34
  var SCHEMA_PATH = resolveSchemaPath();
34
35
  async function loadDocument(filePath) {
35
36
  if (filePath.toLowerCase().endsWith(".jdfx")) {
36
- const zip = await JSZip.loadAsync(fs7.readFileSync(filePath));
37
+ const zip = await JSZip.loadAsync(fs8.readFileSync(filePath));
37
38
  const docFile = zip.file(JDFX_DOCUMENT_PATH);
38
39
  if (!docFile) {
39
40
  console.error(`\u2717 Bundle missing ${JDFX_DOCUMENT_PATH}`);
@@ -58,11 +59,11 @@ async function loadDocument(filePath) {
58
59
  }
59
60
  return { doc, bundle: { manifest, assetCount } };
60
61
  }
61
- return { doc: JSON.parse(fs7.readFileSync(filePath, "utf-8")) };
62
+ return { doc: JSON.parse(fs8.readFileSync(filePath, "utf-8")) };
62
63
  }
63
64
  async function validate(file) {
64
- const filePath = path7.resolve(file);
65
- if (!fs7.existsSync(filePath)) {
65
+ const filePath = path8.resolve(file);
66
+ if (!fs8.existsSync(filePath)) {
66
67
  console.error(`File not found: ${filePath}`);
67
68
  return false;
68
69
  }
@@ -75,11 +76,11 @@ async function validate(file) {
75
76
  }
76
77
  if (!loaded) return false;
77
78
  const { doc, bundle } = loaded;
78
- if (!fs7.existsSync(SCHEMA_PATH)) {
79
+ if (!fs8.existsSync(SCHEMA_PATH)) {
79
80
  console.error(`Schema not found at ${SCHEMA_PATH}`);
80
81
  return false;
81
82
  }
82
- const schema = JSON.parse(fs7.readFileSync(SCHEMA_PATH, "utf-8"));
83
+ const schema = JSON.parse(fs8.readFileSync(SCHEMA_PATH, "utf-8"));
83
84
  const ajv = new Ajv({ allErrors: true, strict: false });
84
85
  addFormats(ajv);
85
86
  const validateFn = ajv.compile(schema);
@@ -88,7 +89,7 @@ async function validate(file) {
88
89
  const d = doc;
89
90
  const pageCount = Array.isArray(d.pages) ? d.pages.length : 0;
90
91
  const elCount = Array.isArray(d.pages) ? d.pages.reduce((acc, p) => acc + (Array.isArray(p?.elements) ? p.elements.length : 0), 0) : 0;
91
- console.log(`\u2713 Valid: ${path7.basename(filePath)}`);
92
+ console.log(`\u2713 Valid: ${path8.basename(filePath)}`);
92
93
  console.log(` Format: ${d.$jdf}${bundle ? " (jdfx bundle)" : ""}`);
93
94
  console.log(` Title: ${d.meta?.title}`);
94
95
  console.log(` Pages: ${pageCount}`);
@@ -101,7 +102,7 @@ async function validate(file) {
101
102
  }
102
103
  return true;
103
104
  }
104
- console.error(`\u2717 Invalid: ${path7.basename(filePath)}`);
105
+ console.error(`\u2717 Invalid: ${path8.basename(filePath)}`);
105
106
  for (const err of validateFn.errors || []) {
106
107
  const loc = err.instancePath || "(root)";
107
108
  console.error(` ${loc} \u2014 ${err.message}`);
@@ -304,13 +305,13 @@ function stripInline(text) {
304
305
  return parseInline(text).map((r) => r.text).join("");
305
306
  }
306
307
  async function importMarkdown(inputPath, outputPath) {
307
- const input = path7.resolve(inputPath);
308
+ const input = path8.resolve(inputPath);
308
309
  console.log(`Importing: ${input}`);
309
- const content = fs7.readFileSync(input, "utf-8");
310
- const doc = convertMarkdownToJdf(content, path7.basename(input, path7.extname(input)), path7.dirname(input));
310
+ const content = fs8.readFileSync(input, "utf-8");
311
+ const doc = convertMarkdownToJdf(content, path8.basename(input, path8.extname(input)), path8.dirname(input));
311
312
  let output;
312
313
  if (outputPath) {
313
- output = path7.resolve(outputPath);
314
+ output = path8.resolve(outputPath);
314
315
  } else {
315
316
  const stem = input.replace(/\.(md|markdown)$/i, "");
316
317
  output = stem + (shouldUseJdfx(doc) ? ".jdfx" : ".jdf");
@@ -318,11 +319,11 @@ async function importMarkdown(inputPath, outputPath) {
318
319
  console.log(`Output: ${output}`);
319
320
  if (output.toLowerCase().endsWith(".jdfx")) {
320
321
  const { bytes, manifest } = await packJdfx(doc);
321
- fs7.writeFileSync(output, bytes);
322
+ fs8.writeFileSync(output, bytes);
322
323
  console.log(`
323
324
  Done! Created ${doc.pages.length} page(s), ${manifest.assets.length} asset(s) bundled`);
324
325
  } else {
325
- fs7.writeFileSync(output, JSON.stringify(doc, null, 2));
326
+ fs8.writeFileSync(output, JSON.stringify(doc, null, 2));
326
327
  console.log(`
327
328
  Done! Created ${doc.pages.length} page(s)`);
328
329
  }
@@ -339,10 +340,10 @@ var MIME_BY_EXT2 = {
339
340
  };
340
341
  function resolveImageSrc(src, baseDir) {
341
342
  if (/^(https?:|data:|file:)/i.test(src)) return src;
342
- const abs = path7.isAbsolute(src) ? src : path7.resolve(baseDir, src);
343
+ const abs = path8.isAbsolute(src) ? src : path8.resolve(baseDir, src);
343
344
  try {
344
- const bytes = fs7.readFileSync(abs);
345
- const ext = path7.extname(abs).slice(1).toLowerCase();
345
+ const bytes = fs8.readFileSync(abs);
346
+ const ext = path8.extname(abs).slice(1).toLowerCase();
346
347
  const mime = MIME_BY_EXT2[ext] || "application/octet-stream";
347
348
  return `data:${mime};base64,${bytes.toString("base64")}`;
348
349
  } catch {
@@ -2188,25 +2189,209 @@ async function importPdfToJdf2(source, title, options = {}) {
2188
2189
  };
2189
2190
  return importPdfToJdf(source, title, runtime, options);
2190
2191
  }
2192
+ var DEFAULT_OLLAMA_MODEL = "qwen2.5vl:3b";
2193
+ var DEFAULT_OPENAI_MODEL = "gpt-4o-mini";
2194
+ var OLLAMA_HOST = process.env.OLLAMA_HOST || "http://localhost:11434";
2195
+ async function loadDoc(file) {
2196
+ if (file.toLowerCase().endsWith(".jdfx")) {
2197
+ const zip = await JSZip.loadAsync(fs8.readFileSync(file));
2198
+ const f = zip.file(JDFX_DOCUMENT_PATH);
2199
+ if (!f) throw new Error(`Bundle missing ${JDFX_DOCUMENT_PATH}`);
2200
+ const doc = JSON.parse(await f.async("string"));
2201
+ const manifest = zip.file("manifest.json") ? JSON.parse(await zip.file("manifest.json").async("string")) : { assets: [] };
2202
+ for (const a of manifest.assets ?? []) {
2203
+ const af = zip.file(a.path);
2204
+ if (!af) continue;
2205
+ const data = (await af.async("nodebuffer")).toString("base64");
2206
+ const res = { src: "embedded", mimeType: a.mimeType, data };
2207
+ doc.resources ??= {};
2208
+ if (/^video\//i.test(a.mimeType || "")) (doc.resources.videos ??= {})[a.id] = res;
2209
+ else (doc.resources.images ??= {})[a.id] = res;
2210
+ }
2211
+ return { doc, bundle: true };
2212
+ }
2213
+ return { doc: JSON.parse(fs8.readFileSync(file, "utf-8")), bundle: false };
2214
+ }
2215
+ function findImages(doc) {
2216
+ const out = [];
2217
+ const walk2 = (els, page) => {
2218
+ for (const el of els ?? []) {
2219
+ if (el?.type === "image") out.push({ el, page, index: out.length });
2220
+ if (el?.elements) walk2(el.elements, page);
2221
+ }
2222
+ };
2223
+ doc.pages.forEach((p, i) => walk2(p.elements, i + 1));
2224
+ return out;
2225
+ }
2226
+ async function imageBytes(doc, el, docDir) {
2227
+ const fromData = (d, fallback) => {
2228
+ const m = d.match(/^data:([^;,]+)?[^,]*,(.*)$/s);
2229
+ return m ? { bytes: Buffer.from(m[2], "base64"), mime: m[1] || fallback } : { bytes: Buffer.from(d, "base64"), mime: fallback };
2230
+ };
2231
+ const res = el.resource ? doc.resources?.images?.[el.resource] ?? doc.resources?.[el.resource] : void 0;
2232
+ if (res?.data) return fromData(String(res.data), res.mimeType || "image/png");
2233
+ if (res?.path) {
2234
+ const p = path8.resolve(docDir, res.path);
2235
+ return fs8.existsSync(p) ? { bytes: fs8.readFileSync(p), mime: res.mimeType || "image/png" } : null;
2236
+ }
2237
+ const src = el.src;
2238
+ if (!src) return null;
2239
+ if (src.startsWith("data:")) return fromData(src, "image/png");
2240
+ if (/^https?:\/\//i.test(src)) {
2241
+ const r = await fetch(src);
2242
+ if (!r.ok) throw new Error(`download failed ${r.status}: ${src}`);
2243
+ return { bytes: Buffer.from(await r.arrayBuffer()), mime: r.headers.get("content-type") || "image/png" };
2244
+ }
2245
+ const local = path8.resolve(docDir, src);
2246
+ return fs8.existsSync(local) ? { bytes: fs8.readFileSync(local), mime: "image/png" } : null;
2247
+ }
2248
+ async function ocrTesseract(bytes, lang) {
2249
+ const { createWorker } = await import('tesseract.js');
2250
+ const cachePath = path8.join(os.homedir(), ".cache", "jdf", "tesseract");
2251
+ fs8.mkdirSync(cachePath, { recursive: true });
2252
+ const worker = await createWorker(lang, 1, { cachePath, logger: () => {
2253
+ } });
2254
+ try {
2255
+ const { data } = await worker.recognize(bytes, {}, { text: true, blocks: true });
2256
+ let w = 1, h = 1;
2257
+ try {
2258
+ const { loadImage } = await import('@napi-rs/canvas');
2259
+ const im = await loadImage(bytes);
2260
+ w = im.width || 1;
2261
+ h = im.height || 1;
2262
+ } catch {
2263
+ }
2264
+ const lines = data.blocks?.flatMap((b) => b.paragraphs?.flatMap((p) => p.lines ?? []) ?? []) ?? data.lines ?? [];
2265
+ const blocks = lines.map((ln) => ({ text: String(ln.text ?? "").replace(/\s+/g, " ").trim(), confidence: ln.confidence != null ? Math.round(ln.confidence) / 100 : void 0, bbox: ln.bbox ? { x: +(ln.bbox.x0 / w).toFixed(4), y: +(ln.bbox.y0 / h).toFixed(4), w: +((ln.bbox.x1 - ln.bbox.x0) / w).toFixed(4), h: +((ln.bbox.y1 - ln.bbox.y0) / h).toFixed(4) } : void 0 })).filter((b) => b.text.length > 0 && (b.confidence == null || b.confidence >= 0.3));
2266
+ if (!blocks.length && String(data.text ?? "").trim()) blocks.push({ text: String(data.text).replace(/\s+/g, " ").trim() });
2267
+ return { blocks, source: `tesseract.js:${lang}` };
2268
+ } finally {
2269
+ await worker.terminate();
2270
+ }
2271
+ }
2272
+ async function openaiVision(bytes, mime, model, prompt2) {
2273
+ const key = process.env.OPENAI_API_KEY;
2274
+ if (!key) throw new Error("OPENAI_API_KEY is not set");
2275
+ const base = process.env.OPENAI_BASE_URL || "https://api.openai.com/v1";
2276
+ const r = await fetch(`${base}/chat/completions`, { method: "POST", headers: { Authorization: `Bearer ${key}`, "content-type": "application/json" }, body: JSON.stringify({ model, messages: [{ role: "user", content: [{ type: "text", text: prompt2 }, { type: "image_url", image_url: { url: `data:${mime};base64,${bytes.toString("base64")}` } }] }], max_tokens: 800 }) });
2277
+ if (!r.ok) throw new Error(`OpenAI vision failed ${r.status}: ${(await r.text()).slice(0, 300)}`);
2278
+ const j = await r.json();
2279
+ return String(j.choices?.[0]?.message?.content ?? "").trim();
2280
+ }
2281
+ async function ollamaVision(bytes, model, prompt2) {
2282
+ const r = await fetch(`${OLLAMA_HOST}/api/generate`, { method: "POST", body: JSON.stringify({ model, prompt: prompt2, images: [bytes.toString("base64")], stream: false, options: { temperature: 0 } }) });
2283
+ if (!r.ok) throw new Error(`Ollama vision failed ${r.status}: ${(await r.text()).slice(0, 300)} \u2014 is the model pulled? (ollama pull ${model})`);
2284
+ const j = await r.json();
2285
+ return String(j.response ?? "").trim();
2286
+ }
2287
+ var CAPTION_PROMPT = "Describe this image for a search index in one or two factual sentences: what it shows, any chart type, axes, trends, labels, names and numbers you can read. No preamble.";
2288
+ var OCR_PROMPT = "Transcribe all text visible in this image exactly, line by line, top to bottom, left to right. Output only the text.";
2289
+ async function describeDocument(doc, docDir, opts = {}) {
2290
+ const ocrP = opts.ocr ?? "tesseract";
2291
+ const capP = opts.caption ?? "ollama";
2292
+ const lang = opts.ocrLanguage ?? "eng";
2293
+ const images = findImages(doc);
2294
+ const targets = opts.element != null ? images.filter((im) => im.el.id === opts.element || String(im.index) === opts.element) : images;
2295
+ if (opts.element != null && !targets.length) throw new Error(`no image element "${opts.element}" (have: ${images.map((i) => i.el.id ?? `#${i.index}`).join(", ") || "none"})`);
2296
+ const stats = { ocr: 0, captions: 0, skipped: 0, failed: [] };
2297
+ for (const im of targets) {
2298
+ const el = im.el;
2299
+ if (!el.id) el.id = `image-${im.index + 1}`;
2300
+ const needOcr = ocrP !== "none" && (opts.force || !el.ocr?.blocks?.length);
2301
+ const needCap = capP !== "none" && (opts.force || !el.caption);
2302
+ if (!needOcr && !needCap) {
2303
+ stats.skipped++;
2304
+ continue;
2305
+ }
2306
+ const img = await imageBytes(doc, el, docDir);
2307
+ if (!img) {
2308
+ stats.failed.push(`${el.id}: image bytes not reachable`);
2309
+ continue;
2310
+ }
2311
+ if (needOcr) {
2312
+ try {
2313
+ if (ocrP === "tesseract") {
2314
+ const r = await ocrTesseract(img.bytes, lang);
2315
+ el.ocr = { language: lang, source: r.source, created: (/* @__PURE__ */ new Date()).toISOString(), blocks: r.blocks };
2316
+ } else {
2317
+ const text = await openaiVision(img.bytes, img.mime, opts.captionModel ?? DEFAULT_OPENAI_MODEL, OCR_PROMPT);
2318
+ el.ocr = { source: `openai:${opts.captionModel ?? DEFAULT_OPENAI_MODEL}`, created: (/* @__PURE__ */ new Date()).toISOString(), blocks: text.split(/\r?\n/).map((t) => t.trim()).filter(Boolean).map((t) => ({ text: t })) };
2319
+ }
2320
+ stats.ocr++;
2321
+ } catch (e) {
2322
+ stats.failed.push(`${el.id} ocr: ${e.message}`);
2323
+ }
2324
+ }
2325
+ if (needCap) {
2326
+ try {
2327
+ const model = opts.captionModel ?? (capP === "ollama" ? DEFAULT_OLLAMA_MODEL : DEFAULT_OPENAI_MODEL);
2328
+ const text = capP === "ollama" ? await ollamaVision(img.bytes, model, CAPTION_PROMPT) : await openaiVision(img.bytes, img.mime, model, CAPTION_PROMPT);
2329
+ if (text) {
2330
+ el.caption = text;
2331
+ el.captionSource = `${capP}:${model}`;
2332
+ stats.captions++;
2333
+ }
2334
+ } catch (e) {
2335
+ stats.failed.push(`${el.id} caption: ${e.message}`);
2336
+ }
2337
+ }
2338
+ if (!opts.quiet) console.log(` \xB7 ${el.id} (page ${im.page}): ${needOcr ? `ocr ${el.ocr?.blocks?.length ?? 0} block(s)` : "ocr kept"}${needCap ? ` \xB7 caption ${el.caption ? `"${String(el.caption).slice(0, 70)}${String(el.caption).length > 70 ? "\u2026" : ""}"` : "\u2014"}` : ""}`);
2339
+ }
2340
+ return stats;
2341
+ }
2342
+ async function describeFile(inputPath, opts = {}) {
2343
+ const input = path8.resolve(inputPath);
2344
+ if (!fs8.existsSync(input)) throw new Error(`File not found: ${input}`);
2345
+ const { doc, bundle } = await loadDoc(input);
2346
+ const images = findImages(doc);
2347
+ if (!images.length) {
2348
+ console.log(`No image elements in ${path8.basename(input)} \u2014 nothing to describe.`);
2349
+ return;
2350
+ }
2351
+ console.log(`Describing: ${path8.basename(input)} \u2014 ${images.length} image(s); ocr=${opts.ocr ?? "tesseract"} caption=${opts.caption ?? "ollama"}${(opts.caption ?? "ollama") === "ollama" ? ` (${opts.captionModel ?? DEFAULT_OLLAMA_MODEL}, local)` : ""}`);
2352
+ const stats = await describeDocument(doc, path8.dirname(input), opts);
2353
+ const output = opts.output ? path8.resolve(opts.output) : input;
2354
+ if (output.toLowerCase().endsWith(".jdfx") || bundle && !opts.output) fs8.writeFileSync(output, (await packJdfx(doc)).bytes);
2355
+ else fs8.writeFileSync(output, JSON.stringify(doc, null, 2));
2356
+ console.log(`Done: ${stats.ocr} OCR, ${stats.captions} caption(s), ${stats.skipped} already had text${stats.failed.length ? `, ${stats.failed.length} failed` : ""}.`);
2357
+ for (const f of stats.failed) console.warn(` ! ${f}`);
2358
+ console.log(`Output: ${output}
2359
+ Next: jdf chunk ${path8.basename(output)} # image text is now part of the chunks`);
2360
+ if (stats.failed.length && stats.ocr + stats.captions === 0) process.exitCode = 1;
2361
+ }
2191
2362
 
2192
2363
  // src/commands/import-pdf.ts
2193
2364
  async function importPdf(inputPath, outputPath, options = {}) {
2194
- const input = path7.resolve(inputPath);
2195
- if (!fs7.existsSync(input)) {
2365
+ const input = path8.resolve(inputPath);
2366
+ if (!fs8.existsSync(input)) {
2196
2367
  console.error(`File not found: ${input}`);
2197
2368
  process.exit(1);
2198
2369
  }
2199
2370
  console.log(`Importing: ${input}`);
2200
- const title = path7.basename(input, path7.extname(input));
2371
+ const title = path8.basename(input, path8.extname(input));
2201
2372
  const t0 = Date.now();
2202
2373
  const doc = await importPdfToJdf2(input, title, {
2203
2374
  password: options.password,
2204
2375
  invisibleText: options.dropInvisibleText ? "drop" : "keep"
2205
2376
  });
2206
2377
  console.log(`Parsed in ${((Date.now() - t0) / 1e3).toFixed(1)}s \u2014 ${doc.pages.length} page(s)`);
2378
+ const scanned = doc.pages.filter((p) => !p.elements.some((e) => e.type === "text" || e.type === "richtext" || e.type === "table") && p.elements.some((e) => e.type === "image"));
2379
+ if (scanned.length) {
2380
+ if (options.ocr && options.ocr !== "none") {
2381
+ console.log(`OCR: ${scanned.length} scanned page(s) \u2192 ${options.ocr}`);
2382
+ for (const p of scanned) for (const e of p.elements) if (e.type === "image" && !e.id) e.id = `scan-${doc.pages.indexOf(p) + 1}`;
2383
+ const ids = scanned.flatMap((p) => p.elements.filter((e) => e.type === "image").map((e) => e.id));
2384
+ for (const id of ids) {
2385
+ const st = await describeDocument(doc, path8.dirname(input), { element: id, ocr: options.ocr, caption: "none", quiet: true });
2386
+ if (st.failed.length) console.warn(` ! ${st.failed.join("; ")}`);
2387
+ }
2388
+ } else {
2389
+ console.warn(`! ${scanned.length} page(s) have no text layer (scanned). RAG will skip them \u2014 re-run with --ocr tesseract (local) or --ocr openai.`);
2390
+ }
2391
+ }
2207
2392
  let output;
2208
2393
  if (outputPath) {
2209
- output = path7.resolve(outputPath);
2394
+ output = path8.resolve(outputPath);
2210
2395
  } else {
2211
2396
  const stem = input.replace(/\.pdf$/i, "");
2212
2397
  const wantJdfx = !options.forceJson && shouldUseJdfx(doc);
@@ -2215,11 +2400,11 @@ async function importPdf(inputPath, outputPath, options = {}) {
2215
2400
  console.log(`Output: ${output}`);
2216
2401
  if (output.toLowerCase().endsWith(".jdfx")) {
2217
2402
  const { bytes, manifest } = await packJdfx(doc);
2218
- fs7.writeFileSync(output, bytes);
2403
+ fs8.writeFileSync(output, bytes);
2219
2404
  console.log(`
2220
2405
  Done! Created ${doc.pages.length} page(s), ${manifest.assets.length} asset(s) bundled`);
2221
2406
  } else {
2222
- fs7.writeFileSync(output, JSON.stringify(doc, null, 2));
2407
+ fs8.writeFileSync(output, JSON.stringify(doc, null, 2));
2223
2408
  console.log(`
2224
2409
  Done! Created ${doc.pages.length} page(s)`);
2225
2410
  }
@@ -2232,23 +2417,23 @@ var ImportJsonError = class extends Error {
2232
2417
  }
2233
2418
  };
2234
2419
  async function importJson(inputPath, outputPath, options = {}) {
2235
- const input = path7.resolve(inputPath);
2236
- if (!fs7.existsSync(input)) {
2420
+ const input = path8.resolve(inputPath);
2421
+ if (!fs8.existsSync(input)) {
2237
2422
  throw new ImportJsonError(`File not found: ${input}`);
2238
2423
  }
2239
2424
  console.log(`Importing: ${input}`);
2240
- const raw = fs7.readFileSync(input, "utf-8");
2425
+ const raw = fs8.readFileSync(input, "utf-8");
2241
2426
  let parsed;
2242
2427
  try {
2243
2428
  parsed = JSON.parse(raw);
2244
2429
  } catch (e) {
2245
2430
  throw new ImportJsonError(`Not valid JSON: ${e.message}`);
2246
2431
  }
2247
- const title = path7.basename(input, path7.extname(input));
2432
+ const title = path8.basename(input, path8.extname(input));
2248
2433
  const doc = normaliseToJdf(parsed, title);
2249
2434
  let output;
2250
2435
  if (outputPath) {
2251
- output = path7.resolve(outputPath);
2436
+ output = path8.resolve(outputPath);
2252
2437
  } else {
2253
2438
  const stem = input.replace(/\.json$/i, "");
2254
2439
  const wantJdfx = !options.forceJson && shouldUseJdfx(doc);
@@ -2257,11 +2442,11 @@ async function importJson(inputPath, outputPath, options = {}) {
2257
2442
  console.log(`Output: ${output}`);
2258
2443
  if (output.toLowerCase().endsWith(".jdfx")) {
2259
2444
  const { bytes, manifest } = await packJdfx(doc);
2260
- fs7.writeFileSync(output, bytes);
2445
+ fs8.writeFileSync(output, bytes);
2261
2446
  console.log(`
2262
2447
  Done! Created ${doc.pages.length} page(s), ${manifest.assets.length} asset(s) bundled`);
2263
2448
  } else {
2264
- fs7.writeFileSync(output, JSON.stringify(doc, null, 2));
2449
+ fs8.writeFileSync(output, JSON.stringify(doc, null, 2));
2265
2450
  console.log(`
2266
2451
  Done! Created ${doc.pages.length} page(s)`);
2267
2452
  }
@@ -2363,8 +2548,8 @@ function transcriptChunks(el, elementId2, page, crumb, windowSec, maxTokens) {
2363
2548
  const body = win.map((sg) => sg.speaker ? `${sg.speaker}: ${sg.text}` : sg.text).join(" ").replace(/\s+/g, " ").trim();
2364
2549
  const text = `[${fmtTime(t0)}\u2013${fmtTime(t1)}] ${body}`;
2365
2550
  const chapter = chapterAt(t0);
2366
- const path9 = [...crumb, ...el.title ? [String(el.title)] : [], ...chapter ? [String(chapter.title)] : []];
2367
- out.push({ id: `${elementId2}@${Math.round(t0)}`, text, path: path9, page, types: ["video"], tokens: estimateTokens(text), hash: hashText(text), media: { element: elementId2, t0, t1 } });
2551
+ const path10 = [...crumb, ...el.title ? [String(el.title)] : [], ...chapter ? [String(chapter.title)] : []];
2552
+ out.push({ id: `${elementId2}@${Math.round(t0)}`, text, path: path10, page, types: ["video"], tokens: estimateTokens(text), hash: hashText(text), media: { element: elementId2, t0, t1 } });
2368
2553
  win = [];
2369
2554
  };
2370
2555
  for (const sg of segs) {
@@ -2433,8 +2618,14 @@ function serializeElement(el) {
2433
2618
  }
2434
2619
  case "checkbox":
2435
2620
  return `${e.checked ? "[x]" : "[ ]"} ${e.label ?? ""}`.trim();
2436
- case "image":
2437
- return e.alt ? `[image: ${e.alt}]` : "";
2621
+ case "image": {
2622
+ const parts = [];
2623
+ if (e.alt) parts.push(`[image: ${e.alt}]`);
2624
+ if (e.caption) parts.push(String(e.caption).trim());
2625
+ const ocr = (e.ocr?.blocks || []).map((b) => String(b.text ?? "").trim()).filter(Boolean).join("\n");
2626
+ if (ocr) parts.push(ocr);
2627
+ return parts.join("\n");
2628
+ }
2438
2629
  case "video":
2439
2630
  return e.title ? `[video: ${e.title}]` : "";
2440
2631
  case "toc":
@@ -2554,34 +2745,62 @@ function chunkDocument(doc, options = {}) {
2554
2745
  }
2555
2746
  async function loadJdf(filePath) {
2556
2747
  if (filePath.toLowerCase().endsWith(".jdfx")) {
2557
- const zip = await JSZip.loadAsync(fs7.readFileSync(filePath));
2748
+ const zip = await JSZip.loadAsync(fs8.readFileSync(filePath));
2558
2749
  const docFile = zip.file(JDFX_DOCUMENT_PATH);
2559
2750
  if (!docFile) throw new Error(`Bundle missing ${JDFX_DOCUMENT_PATH}`);
2560
2751
  return JSON.parse(await docFile.async("string"));
2561
2752
  }
2562
- return JSON.parse(fs7.readFileSync(filePath, "utf-8"));
2753
+ return JSON.parse(fs8.readFileSync(filePath, "utf-8"));
2754
+ }
2755
+ function mediaCoverage(doc) {
2756
+ const cov = { images: { total: 0, covered: 0, missing: [] }, videos: { total: 0, covered: 0, missing: [] } };
2757
+ const walk2 = (els, page) => {
2758
+ for (const el of els ?? []) {
2759
+ if (el?.type === "image") {
2760
+ cov.images.total++;
2761
+ const has = !!(el.caption && String(el.caption).trim()) || !!el.ocr?.blocks?.some((b) => String(b.text ?? "").trim());
2762
+ if (has) cov.images.covered++;
2763
+ else cov.images.missing.push({ id: el.id, page, alt: el.alt });
2764
+ } else if (el?.type === "video") {
2765
+ cov.videos.total++;
2766
+ if (el.transcript?.segments?.length) cov.videos.covered++;
2767
+ else cov.videos.missing.push({ id: el.id, page, title: el.title });
2768
+ }
2769
+ if (el?.elements) walk2(el.elements, page);
2770
+ }
2771
+ };
2772
+ doc.pages.forEach((p, i) => walk2(p.elements, i + 1));
2773
+ return cov;
2774
+ }
2775
+ function coverageSummary(cov) {
2776
+ const parts = [];
2777
+ if (cov.images.missing.length) parts.push(`${cov.images.missing.length} of ${cov.images.total} image(s) have no caption/OCR text \u2192 jdf describe`);
2778
+ if (cov.videos.missing.length) parts.push(`${cov.videos.missing.length} of ${cov.videos.total} video(s) have no transcript \u2192 jdf transcribe`);
2779
+ return parts.length ? parts.join("; ") : null;
2563
2780
  }
2564
2781
  async function chunkFile(inputPath, opts = {}) {
2565
- const input = path7.resolve(inputPath);
2566
- if (!fs7.existsSync(input)) throw new Error(`File not found: ${input}`);
2782
+ const input = path8.resolve(inputPath);
2783
+ if (!fs8.existsSync(input)) throw new Error(`File not found: ${input}`);
2567
2784
  const doc = await loadJdf(input);
2568
2785
  const strategy = opts.strategy ?? "section";
2569
2786
  const chunks = chunkDocument(doc, { strategy, maxTokens: opts.maxTokens, transcriptWindowSec: opts.transcriptWindowSec });
2787
+ const gap = coverageSummary(mediaCoverage(doc));
2788
+ if (gap) console.warn(` ! media without text (skipped by retrieval): ${gap}`);
2570
2789
  const format = opts.format ?? "jsonl";
2571
2790
  console.log(`Chunking: ${input}`);
2572
2791
  console.log(`Strategy: ${strategy}${opts.maxTokens ? ` (max ${opts.maxTokens} tokens)` : ""}`);
2573
2792
  if (format === "inline") {
2574
- const out = opts.output ? path7.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".jdf");
2793
+ const out = opts.output ? path8.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".jdf");
2575
2794
  const withIndex = { ...doc, index: { chunker: `jdf-${strategy}-v1`, chunks } };
2576
- fs7.writeFileSync(out, JSON.stringify(withIndex, null, 2));
2795
+ fs8.writeFileSync(out, JSON.stringify(withIndex, null, 2));
2577
2796
  console.log(`Output: ${out} (${chunks.length} chunks in "index" block)`);
2578
2797
  } else if (format === "json") {
2579
- const out = opts.output ? path7.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".chunks.json");
2580
- fs7.writeFileSync(out, JSON.stringify(chunks, null, 2));
2798
+ const out = opts.output ? path8.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".chunks.json");
2799
+ fs8.writeFileSync(out, JSON.stringify(chunks, null, 2));
2581
2800
  console.log(`Output: ${out} (${chunks.length} chunks)`);
2582
2801
  } else {
2583
- const out = opts.output ? path7.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".chunks.jsonl");
2584
- fs7.writeFileSync(out, chunks.map((c) => JSON.stringify(c)).join("\n") + "\n");
2802
+ const out = opts.output ? path8.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".chunks.jsonl");
2803
+ fs8.writeFileSync(out, chunks.map((c) => JSON.stringify(c)).join("\n") + "\n");
2585
2804
  console.log(`Output: ${out} (${chunks.length} chunks)`);
2586
2805
  }
2587
2806
  const totalTokens = chunks.reduce((a, c) => a + c.tokens, 0);
@@ -2604,11 +2823,11 @@ async function embedBatch(provider, model, inputs) {
2604
2823
  throw new Error(`Unknown embedding provider: ${provider}`);
2605
2824
  }
2606
2825
  }
2607
- var OLLAMA_HOST = process.env.OLLAMA_HOST || "http://localhost:11434";
2826
+ var OLLAMA_HOST2 = process.env.OLLAMA_HOST || "http://localhost:11434";
2608
2827
  var OLLAMA_CONTAINER = "jdf-ollama";
2609
2828
  async function ollamaUp() {
2610
2829
  try {
2611
- const res = await fetch(`${OLLAMA_HOST}/api/tags`, { signal: AbortSignal.timeout(1500) });
2830
+ const res = await fetch(`${OLLAMA_HOST2}/api/tags`, { signal: AbortSignal.timeout(1500) });
2612
2831
  return res.ok;
2613
2832
  } catch {
2614
2833
  return false;
@@ -2633,12 +2852,12 @@ async function ensureOllama(model, autoStart) {
2633
2852
  docker run -d --name ${OLLAMA_CONTAINER} -p 11434:11434 -v jdf-ollama:/root/.ollama ollama/ollama
2634
2853
  docker exec ${OLLAMA_CONTAINER} ollama pull ${model}`;
2635
2854
  if (!autoStart) {
2636
- throw new Error(`Ollama isn't running at ${OLLAMA_HOST} and --no-auto-start was given.
2855
+ throw new Error(`Ollama isn't running at ${OLLAMA_HOST2} and --no-auto-start was given.
2637
2856
  ${manualHint}`);
2638
2857
  }
2639
2858
  if (!dockerReady()) {
2640
2859
  throw new Error(
2641
- `Ollama isn't running at ${OLLAMA_HOST}, and the Docker daemon isn't available to auto-start it.
2860
+ `Ollama isn't running at ${OLLAMA_HOST2}, and the Docker daemon isn't available to auto-start it.
2642
2861
  ${manualHint}`
2643
2862
  );
2644
2863
  }
@@ -2675,7 +2894,7 @@ ${manualHint}`);
2675
2894
  }
2676
2895
  async function ollamaPull(model) {
2677
2896
  try {
2678
- const show = await fetch(`${OLLAMA_HOST}/api/show`, {
2897
+ const show = await fetch(`${OLLAMA_HOST2}/api/show`, {
2679
2898
  method: "POST",
2680
2899
  headers: { "Content-Type": "application/json" },
2681
2900
  body: JSON.stringify({ name: model })
@@ -2684,7 +2903,7 @@ async function ollamaPull(model) {
2684
2903
  } catch {
2685
2904
  }
2686
2905
  console.log(`Pulling embedding model "${model}" into Ollama (first run only)\u2026`);
2687
- const res = await fetch(`${OLLAMA_HOST}/api/pull`, {
2906
+ const res = await fetch(`${OLLAMA_HOST2}/api/pull`, {
2688
2907
  method: "POST",
2689
2908
  headers: { "Content-Type": "application/json" },
2690
2909
  body: JSON.stringify({ name: model, stream: false })
@@ -2695,7 +2914,7 @@ async function ollamaPull(model) {
2695
2914
  async function embedOllama(model, inputs) {
2696
2915
  const out = [];
2697
2916
  for (const text of inputs) {
2698
- const res = await fetch(`${OLLAMA_HOST}/api/embeddings`, {
2917
+ const res = await fetch(`${OLLAMA_HOST2}/api/embeddings`, {
2699
2918
  method: "POST",
2700
2919
  headers: { "Content-Type": "application/json" },
2701
2920
  body: JSON.stringify({ model, prompt: text })
@@ -2753,17 +2972,17 @@ async function embedOpenAI(model, inputs) {
2753
2972
  }
2754
2973
  async function loadJdf2(filePath) {
2755
2974
  if (filePath.toLowerCase().endsWith(".jdfx")) {
2756
- const zip = await JSZip.loadAsync(fs7.readFileSync(filePath));
2975
+ const zip = await JSZip.loadAsync(fs8.readFileSync(filePath));
2757
2976
  const docFile = zip.file(JDFX_DOCUMENT_PATH);
2758
2977
  if (!docFile) throw new Error(`Bundle missing ${JDFX_DOCUMENT_PATH}`);
2759
2978
  return JSON.parse(await docFile.async("string"));
2760
2979
  }
2761
- return JSON.parse(fs7.readFileSync(filePath, "utf-8"));
2980
+ return JSON.parse(fs8.readFileSync(filePath, "utf-8"));
2762
2981
  }
2763
2982
  function loadCache(cachePath) {
2764
2983
  try {
2765
- if (!fs7.existsSync(cachePath)) return null;
2766
- return JSON.parse(fs7.readFileSync(cachePath, "utf-8"));
2984
+ if (!fs8.existsSync(cachePath)) return null;
2985
+ return JSON.parse(fs8.readFileSync(cachePath, "utf-8"));
2767
2986
  } catch {
2768
2987
  return null;
2769
2988
  }
@@ -2774,15 +2993,15 @@ function batched(items, size) {
2774
2993
  return out;
2775
2994
  }
2776
2995
  async function embedFile(inputPath, opts = {}) {
2777
- const input = path7.resolve(inputPath);
2778
- if (!fs7.existsSync(input)) throw new Error(`File not found: ${input}`);
2996
+ const input = path8.resolve(inputPath);
2997
+ if (!fs8.existsSync(input)) throw new Error(`File not found: ${input}`);
2779
2998
  const provider = opts.provider ?? "ollama";
2780
2999
  const model = opts.model ?? DEFAULT_MODEL[provider];
2781
3000
  const strategy = opts.strategy ?? "section";
2782
3001
  const doc = await loadJdf2(input);
2783
3002
  const chunks = chunkDocument(doc, { strategy, maxTokens: opts.maxTokens, transcriptWindowSec: opts.transcriptWindowSec });
2784
- const output = opts.output ? path7.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".embeddings.json");
2785
- const cachePath = opts.cache ? path7.resolve(opts.cache) : output;
3003
+ const output = opts.output ? path8.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".embeddings.json");
3004
+ const cachePath = opts.cache ? path8.resolve(opts.cache) : output;
2786
3005
  console.log(`Embedding: ${input}`);
2787
3006
  console.log(`Provider: ${provider} / ${model}${provider === "ollama" ? " (local \u2014 no data leaves this machine)" : " (remote API)"}`);
2788
3007
  console.log(`Strategy: ${strategy} \u2192 ${chunks.length} chunks`);
@@ -2821,7 +3040,7 @@ async function embedFile(inputPath, opts = {}) {
2821
3040
  chunker: `jdf-${strategy}-v1`,
2822
3041
  vectors
2823
3042
  };
2824
- fs7.writeFileSync(output, JSON.stringify(sidecar));
3043
+ fs8.writeFileSync(output, JSON.stringify(sidecar));
2825
3044
  console.log(`
2826
3045
  Done! ${Object.keys(vectors).length} vectors (${dims}-dim) \u2192 ${output}`);
2827
3046
  return sidecar;
@@ -2865,9 +3084,9 @@ function parseChapters(text) {
2865
3084
  return { t: toSec(m[1]), title: m[2].trim() };
2866
3085
  });
2867
3086
  }
2868
- async function loadDoc(file) {
3087
+ async function loadDoc2(file) {
2869
3088
  if (file.toLowerCase().endsWith(".jdfx")) {
2870
- const zip = await JSZip.loadAsync(fs7.readFileSync(file));
3089
+ const zip = await JSZip.loadAsync(fs8.readFileSync(file));
2871
3090
  const f = zip.file(JDFX_DOCUMENT_PATH);
2872
3091
  if (!f) throw new Error(`Bundle missing ${JDFX_DOCUMENT_PATH}`);
2873
3092
  const doc = JSON.parse(await f.async("string"));
@@ -2883,7 +3102,7 @@ async function loadDoc(file) {
2883
3102
  }
2884
3103
  return { doc, bundle: true, zip };
2885
3104
  }
2886
- return { doc: JSON.parse(fs7.readFileSync(file, "utf-8")), bundle: false };
3105
+ return { doc: JSON.parse(fs8.readFileSync(file, "utf-8")), bundle: false };
2887
3106
  }
2888
3107
  function findVideos(doc) {
2889
3108
  const out = [];
@@ -2897,27 +3116,27 @@ function findVideos(doc) {
2897
3116
  return out;
2898
3117
  }
2899
3118
  async function clipToTempFile(doc, el, docDir) {
2900
- const tmp = path7.join(fs7.mkdtempSync(path7.join(__require("os").tmpdir(), "jdf-transcribe-")), "clip.mp4");
3119
+ const tmp = path8.join(fs8.mkdtempSync(path8.join(__require("os").tmpdir(), "jdf-transcribe-")), "clip.mp4");
2901
3120
  const res = el.resource ? doc.resources?.videos?.[el.resource] ?? doc.resources?.images?.[el.resource] : void 0;
2902
3121
  if (res?.data) {
2903
- fs7.writeFileSync(tmp, Buffer.from(res.data.replace(/^data:[^,]*,/, ""), "base64"));
3122
+ fs8.writeFileSync(tmp, Buffer.from(res.data.replace(/^data:[^,]*,/, ""), "base64"));
2904
3123
  return tmp;
2905
3124
  }
2906
- if (res?.path) return path7.resolve(docDir, res.path);
3125
+ if (res?.path) return path8.resolve(docDir, res.path);
2907
3126
  const src = el.src;
2908
3127
  if (!src) return null;
2909
3128
  if (src.startsWith("data:")) {
2910
- fs7.writeFileSync(tmp, Buffer.from(src.replace(/^data:[^,]*,/, ""), "base64"));
3129
+ fs8.writeFileSync(tmp, Buffer.from(src.replace(/^data:[^,]*,/, ""), "base64"));
2911
3130
  return tmp;
2912
3131
  }
2913
3132
  if (/^https?:\/\//i.test(src)) {
2914
3133
  const r = await fetch(src);
2915
3134
  if (!r.ok) throw new Error(`download failed ${r.status}: ${src}`);
2916
- fs7.writeFileSync(tmp, Buffer.from(await r.arrayBuffer()));
3135
+ fs8.writeFileSync(tmp, Buffer.from(await r.arrayBuffer()));
2917
3136
  return tmp;
2918
3137
  }
2919
- const local = path7.resolve(docDir, src);
2920
- return fs7.existsSync(local) ? local : null;
3138
+ const local = path8.resolve(docDir, src);
3139
+ return fs8.existsSync(local) ? local : null;
2921
3140
  }
2922
3141
  function whisperCli(clip, model, language, prompt2) {
2923
3142
  const ffmpeg = spawnSync("ffmpeg", ["-version"]);
@@ -2932,7 +3151,7 @@ function whisperCli(clip, model, language, prompt2) {
2932
3151
  const run = spawnSync("whisper-cli", args, { encoding: "utf-8" });
2933
3152
  if (run.error) throw new Error("whisper-cli not found \u2014 install whisper.cpp (brew install whisper-cpp) or use --from / --provider openai");
2934
3153
  if (run.status !== 0) throw new Error(`whisper-cli failed: ${run.stderr.slice(-400)}`);
2935
- const j = JSON.parse(fs7.readFileSync(wav.replace(/\.wav$/, "") + ".json", "utf-8"));
3154
+ const j = JSON.parse(fs8.readFileSync(wav.replace(/\.wav$/, "") + ".json", "utf-8"));
2936
3155
  const segs = j.transcription ?? j.segments ?? [];
2937
3156
  const ms = (x) => typeof x === "number" ? x / 1e3 : toSec(String(x).replace(",", "."));
2938
3157
  return segs.map((sg) => ({ t0: ms(sg.offsets?.from ?? sg.start), t1: ms(sg.offsets?.to ?? sg.end), text: String(sg.text ?? "").trim() })).filter((sg) => sg.text);
@@ -2942,7 +3161,7 @@ async function openaiTranscribe(clip, model, language, prompt2) {
2942
3161
  if (!key) throw new Error("OPENAI_API_KEY is not set");
2943
3162
  const base = process.env.OPENAI_BASE_URL || "https://api.openai.com/v1";
2944
3163
  const form = new FormData();
2945
- form.append("file", new Blob([fs7.readFileSync(clip)]), path7.basename(clip));
3164
+ form.append("file", new Blob([fs8.readFileSync(clip)]), path8.basename(clip));
2946
3165
  form.append("model", model || "whisper-1");
2947
3166
  form.append("response_format", "verbose_json");
2948
3167
  form.append("timestamp_granularities[]", "segment");
@@ -2954,9 +3173,9 @@ async function openaiTranscribe(clip, model, language, prompt2) {
2954
3173
  return (j.segments ?? []).map((sg) => ({ t0: Number(sg.start), t1: Number(sg.end), text: String(sg.text).trim() })).filter((sg) => sg.text);
2955
3174
  }
2956
3175
  async function transcribeFile(inputPath, opts = {}) {
2957
- const input = path7.resolve(inputPath);
2958
- if (!fs7.existsSync(input)) throw new Error(`File not found: ${input}`);
2959
- const { doc, bundle } = await loadDoc(input);
3176
+ const input = path8.resolve(inputPath);
3177
+ if (!fs8.existsSync(input)) throw new Error(`File not found: ${input}`);
3178
+ const { doc, bundle } = await loadDoc2(input);
2960
3179
  const videos = findVideos(doc);
2961
3180
  if (!videos.length) throw new Error("document has no video element");
2962
3181
  let target = videos[0];
@@ -2972,40 +3191,40 @@ async function transcribeFile(inputPath, opts = {}) {
2972
3191
  let segments;
2973
3192
  let source;
2974
3193
  if (opts.from) {
2975
- segments = parseSubtitles(fs7.readFileSync(path7.resolve(opts.from), "utf-8"), opts.from);
2976
- source = `${path7.extname(opts.from).slice(1).toLowerCase() || "file"}-import`;
3194
+ segments = parseSubtitles(fs8.readFileSync(path8.resolve(opts.from), "utf-8"), opts.from);
3195
+ source = `${path8.extname(opts.from).slice(1).toLowerCase() || "file"}-import`;
2977
3196
  } else {
2978
3197
  const provider = opts.provider ?? "whisper-cli";
2979
- const clip = await clipToTempFile(doc, target.el, path7.dirname(input));
3198
+ const clip = await clipToTempFile(doc, target.el, path8.dirname(input));
2980
3199
  if (!clip) throw new Error("could not locate the clip bytes (no bundled asset, data URL, local path or http URL) \u2014 use --from to import subtitles instead");
2981
3200
  segments = provider === "openai" ? await openaiTranscribe(clip, opts.model, opts.language, opts.prompt) : whisperCli(clip, opts.model, opts.language, opts.prompt);
2982
- source = provider === "openai" ? `openai:${opts.model || "whisper-1"}` : `whisper-cli${opts.model ? ":" + path7.basename(opts.model) : ""}`;
3201
+ source = provider === "openai" ? `openai:${opts.model || "whisper-1"}` : `whisper-cli${opts.model ? ":" + path8.basename(opts.model) : ""}`;
2983
3202
  }
2984
3203
  segments.sort((a, b) => a.t0 - b.t0);
2985
3204
  const transcript = { ...opts.language ? { language: opts.language } : {}, source, created: (/* @__PURE__ */ new Date()).toISOString(), segments };
2986
3205
  target.el.transcript = transcript;
2987
- if (opts.chapters) target.el.chapters = parseChapters(fs7.readFileSync(path7.resolve(opts.chapters), "utf-8"));
3206
+ if (opts.chapters) target.el.chapters = parseChapters(fs8.readFileSync(path8.resolve(opts.chapters), "utf-8"));
2988
3207
  if (!target.el.id) target.el.id = `video-${target.index + 1}`;
2989
- const output = opts.output ? path7.resolve(opts.output) : input;
3208
+ const output = opts.output ? path8.resolve(opts.output) : input;
2990
3209
  if (output.toLowerCase().endsWith(".jdfx") || bundle && !opts.output) {
2991
3210
  const { bytes } = await packJdfx(doc);
2992
- fs7.writeFileSync(output, bytes);
3211
+ fs8.writeFileSync(output, bytes);
2993
3212
  } else {
2994
- fs7.writeFileSync(output, JSON.stringify(doc, null, 2));
3213
+ fs8.writeFileSync(output, JSON.stringify(doc, null, 2));
2995
3214
  }
2996
3215
  const dur = segments.length ? segments[segments.length - 1].t1 : 0;
2997
- console.log(`Transcribed: ${path7.basename(input)} \u2192 element "${target.el.id}" (${segments.length} segments, ${Math.round(dur)} s, source ${source})`);
3216
+ console.log(`Transcribed: ${path8.basename(input)} \u2192 element "${target.el.id}" (${segments.length} segments, ${Math.round(dur)} s, source ${source})`);
2998
3217
  if (target.el.chapters) console.log(`Chapters: ${target.el.chapters.length}`);
2999
3218
  console.log(`Output: ${output}
3000
- Next: jdf chunk ${path7.basename(output)} # transcript \u2192 time-windowed chunks with media.t0/t1`);
3219
+ Next: jdf chunk ${path8.basename(output)} # transcript \u2192 time-windowed chunks with media.t0/t1`);
3001
3220
  return transcript;
3002
3221
  }
3003
3222
  var CONFIG_NAME = "jdf.rag.json";
3004
3223
  var OUT_DIR = ".jdf-rag";
3005
3224
  function walk(dir, acc = []) {
3006
- for (const ent of fs7.readdirSync(dir, { withFileTypes: true })) {
3225
+ for (const ent of fs8.readdirSync(dir, { withFileTypes: true })) {
3007
3226
  if (ent.name === "node_modules" || ent.name === OUT_DIR || ent.name.startsWith(".")) continue;
3008
- const p = path7.join(dir, ent.name);
3227
+ const p = path8.join(dir, ent.name);
3009
3228
  if (ent.isDirectory()) walk(p, acc);
3010
3229
  else if (/\.(jdf|jdfx)$/i.test(ent.name)) acc.push(p);
3011
3230
  }
@@ -3013,10 +3232,10 @@ function walk(dir, acc = []) {
3013
3232
  }
3014
3233
  async function readDoc(file) {
3015
3234
  if (file.toLowerCase().endsWith(".jdfx")) {
3016
- const zip = await JSZip.loadAsync(fs7.readFileSync(file));
3235
+ const zip = await JSZip.loadAsync(fs8.readFileSync(file));
3017
3236
  return JSON.parse(await zip.file(JDFX_DOCUMENT_PATH).async("string"));
3018
3237
  }
3019
- return JSON.parse(fs7.readFileSync(file, "utf-8"));
3238
+ return JSON.parse(fs8.readFileSync(file, "utf-8"));
3020
3239
  }
3021
3240
  function videosIn(doc) {
3022
3241
  const out = [];
@@ -3030,29 +3249,44 @@ function videosIn(doc) {
3030
3249
  return out;
3031
3250
  }
3032
3251
  async function ragFolder(dirPath, cli = {}) {
3033
- const dir = path7.resolve(dirPath);
3034
- if (!fs7.existsSync(dir) || !fs7.statSync(dir).isDirectory()) throw new Error(`Not a directory: ${dir}`);
3035
- const cfgPath = path7.join(dir, CONFIG_NAME);
3036
- const cfg = fs7.existsSync(cfgPath) ? JSON.parse(fs7.readFileSync(cfgPath, "utf-8")) : {};
3252
+ const dir = path8.resolve(dirPath);
3253
+ if (!fs8.existsSync(dir) || !fs8.statSync(dir).isDirectory()) throw new Error(`Not a directory: ${dir}`);
3254
+ const cfgPath = path8.join(dir, CONFIG_NAME);
3255
+ const cfg = fs8.existsSync(cfgPath) ? JSON.parse(fs8.readFileSync(cfgPath, "utf-8")) : {};
3037
3256
  const opts = { ...cfg, ...Object.fromEntries(Object.entries(cli).filter(([, v]) => v !== void 0)) };
3038
3257
  const provider = opts.provider ?? "ollama";
3039
3258
  const transcribe = opts.transcribe ?? "none";
3040
- const outDir = path7.resolve(opts.out ?? path7.join(dir, OUT_DIR));
3259
+ const outDir = path8.resolve(opts.out ?? path8.join(dir, OUT_DIR));
3260
+ const ocr = opts.ocr ?? "none";
3261
+ const caption = opts.caption ?? "none";
3041
3262
  const files = walk(dir);
3042
3263
  console.log(`jdf rag: ${dir}
3043
- files: ${files.length} (.jdf/.jdfx)${fs7.existsSync(cfgPath) ? `
3264
+ files: ${files.length} (.jdf/.jdfx)${fs8.existsSync(cfgPath) ? `
3044
3265
  config: ${CONFIG_NAME}` : ""}
3045
3266
  embeddings: ${opts.noEmbed ? "skipped (--no-embed)" : `${provider}${opts.model ? " / " + opts.model : ""}`}
3046
- transcribe: ${transcribe}${opts.dryRun ? "\n DRY RUN \u2014 nothing written" : ""}
3267
+ transcribe: ${transcribe} ocr: ${ocr} caption: ${caption}${opts.dryRun ? "\n DRY RUN \u2014 nothing written" : ""}
3047
3268
  `);
3048
3269
  if (!files.length) {
3049
3270
  console.log("Nothing to do.");
3050
3271
  return;
3051
3272
  }
3052
- const manifest = { dir, created: (/* @__PURE__ */ new Date()).toISOString(), provider: opts.noEmbed ? null : provider, model: opts.model ?? null, strategy: opts.strategy ?? "section", transcribe, files: [], totals: { files: files.length, chunks: 0, videoChunks: 0, videos: 0, transcribed: 0, untranscribed: 0 } };
3273
+ const manifest = {
3274
+ dir,
3275
+ created: (/* @__PURE__ */ new Date()).toISOString(),
3276
+ provider: opts.noEmbed ? null : provider,
3277
+ model: opts.model ?? null,
3278
+ strategy: opts.strategy ?? "section",
3279
+ transcribe,
3280
+ ocr,
3281
+ caption,
3282
+ files: [],
3283
+ totals: { files: files.length, chunks: 0, videoChunks: 0, videos: 0, transcribed: 0, untranscribed: 0, images: 0, described: 0, imagesWithoutText: 0 },
3284
+ /** Every media element that retrieval would still skip, by file — the thing to fix before shipping an index. */
3285
+ mediaWithoutText: []
3286
+ };
3053
3287
  const indexLines = [];
3054
3288
  for (const file of files) {
3055
- const rel = path7.relative(dir, file);
3289
+ const rel = path8.relative(dir, file);
3056
3290
  const doc = await readDoc(file);
3057
3291
  const vids = videosIn(doc);
3058
3292
  let transcribedHere = 0;
@@ -3076,20 +3310,36 @@ async function ragFolder(dirPath, cli = {}) {
3076
3310
  }
3077
3311
  manifest.totals.videos += vids.length;
3078
3312
  manifest.totals.transcribed += transcribedHere;
3313
+ const covBefore = mediaCoverage(doc);
3314
+ let describedHere = 0;
3315
+ if (covBefore.images.missing.length && (ocr !== "none" || caption !== "none") && !opts.dryRun) {
3316
+ try {
3317
+ await describeFile(file, { ocr, caption, captionModel: opts.captionModel, quiet: true });
3318
+ describedHere = covBefore.images.missing.length;
3319
+ } catch (e) {
3320
+ console.warn(` ! ${rel}: describe failed: ${e.message}`);
3321
+ }
3322
+ }
3323
+ const covAfter = transcribedHere || describedHere ? mediaCoverage(await readDoc(file)) : covBefore;
3324
+ manifest.totals.images += covAfter.images.total;
3325
+ manifest.totals.described += describedHere;
3326
+ manifest.totals.imagesWithoutText += covAfter.images.missing.length;
3327
+ for (const m of covAfter.images.missing) manifest.mediaWithoutText.push({ file: rel, type: "image", ...m });
3328
+ for (const m of covAfter.videos.missing) manifest.mediaWithoutText.push({ file: rel, type: "video", ...m });
3079
3329
  if (opts.dryRun) {
3080
3330
  manifest.files.push({ file: rel, videos: vids.length, wouldTranscribe: transcribedHere });
3081
3331
  continue;
3082
3332
  }
3083
3333
  const chunkOpts = { strategy: opts.strategy, maxTokens: opts.maxTokens, transcriptWindowSec: opts.transcriptWindowSec };
3084
- const chunkOut = path7.join(outDir, "chunks", rel.replace(/\.(jdf|jdfx)$/i, ".chunks.jsonl"));
3085
- fs7.mkdirSync(path7.dirname(chunkOut), { recursive: true });
3334
+ const chunkOut = path8.join(outDir, "chunks", rel.replace(/\.(jdf|jdfx)$/i, ".chunks.jsonl"));
3335
+ fs8.mkdirSync(path8.dirname(chunkOut), { recursive: true });
3086
3336
  let chunks;
3087
3337
  if (opts.noEmbed) {
3088
3338
  chunks = await chunkFile(file, { ...chunkOpts, format: "jsonl", output: chunkOut });
3089
3339
  } else {
3090
3340
  const side = await embedFile(file, { ...chunkOpts, provider, model: opts.model, incremental: true });
3091
3341
  chunks = await chunkFile(file, { ...chunkOpts, format: "jsonl", output: chunkOut });
3092
- manifest.files.push({ file: rel, chunks: chunks.length, vectors: Object.keys(side.vectors).length, sidecar: path7.relative(dir, file.replace(/\.(jdf|jdfx)$/i, ".embeddings.json")), videos: vids.length, transcribed: transcribedHere });
3342
+ manifest.files.push({ file: rel, chunks: chunks.length, vectors: Object.keys(side.vectors).length, sidecar: path8.relative(dir, file.replace(/\.(jdf|jdfx)$/i, ".embeddings.json")), videos: vids.length, transcribed: transcribedHere });
3093
3343
  }
3094
3344
  if (opts.noEmbed) manifest.files.push({ file: rel, chunks: chunks.length, videos: vids.length, transcribed: transcribedHere });
3095
3345
  for (const c of chunks) {
@@ -3099,15 +3349,27 @@ async function ragFolder(dirPath, cli = {}) {
3099
3349
  }
3100
3350
  }
3101
3351
  if (!opts.dryRun) {
3102
- fs7.mkdirSync(outDir, { recursive: true });
3103
- fs7.writeFileSync(path7.join(outDir, "index.jsonl"), indexLines.join("\n") + (indexLines.length ? "\n" : ""));
3104
- fs7.writeFileSync(path7.join(outDir, "manifest.json"), JSON.stringify(manifest, null, 2) + "\n");
3352
+ fs8.mkdirSync(outDir, { recursive: true });
3353
+ fs8.writeFileSync(path8.join(outDir, "index.jsonl"), indexLines.join("\n") + (indexLines.length ? "\n" : ""));
3354
+ fs8.writeFileSync(path8.join(outDir, "manifest.json"), JSON.stringify(manifest, null, 2) + "\n");
3105
3355
  }
3106
3356
  const t = manifest.totals;
3107
3357
  console.log(`
3108
- Done. ${t.files} files \u2192 ${t.chunks} chunks (${t.videoChunks} from video transcripts); videos ${t.videos}, transcribed now ${t.transcribed}, still without transcript ${t.untranscribed}${t.untranscribed && transcribe === "none" ? " (pass --transcribe whisper-cli|openai, or jdf transcribe --from subs.srt)" : ""}.`);
3109
- if (!opts.dryRun) console.log(`Index: ${path7.join(outDir, "index.jsonl")}
3110
- Report: ${path7.join(outDir, "manifest.json")}${opts.noEmbed ? "" : `
3358
+ Done. ${t.files} files \u2192 ${t.chunks} chunks (${t.videoChunks} from video transcripts).`);
3359
+ console.log(`Media coverage: videos ${t.videos - t.untranscribed}/${t.videos} with transcript (transcribed now ${t.transcribed}), images ${t.images - t.imagesWithoutText}/${t.images} with caption/OCR (described now ${t.described}).`);
3360
+ if (manifest.mediaWithoutText.length) {
3361
+ console.log(`
3362
+ ! ${manifest.mediaWithoutText.length} media element(s) still have NO text \u2014 retrieval will skip them:`);
3363
+ for (const m of manifest.mediaWithoutText.slice(0, 12)) console.log(` ${m.file} \xB7 ${m.type} ${m.id ?? ""} (page ${m.page})${m.title ? ` "${m.title}"` : m.alt ? ` alt="${m.alt}"` : ""}`);
3364
+ if (manifest.mediaWithoutText.length > 12) console.log(` \u2026 ${manifest.mediaWithoutText.length - 12} more in manifest.json`);
3365
+ console.log(` fix: jdf rag <dir> --transcribe whisper-cli|openai --ocr tesseract --caption ollama (or jdf transcribe / jdf describe per file)`);
3366
+ if (opts.strict) {
3367
+ console.error(`--strict: failing because media without text remains.`);
3368
+ process.exitCode = 1;
3369
+ }
3370
+ }
3371
+ if (!opts.dryRun) console.log(`Index: ${path8.join(outDir, "index.jsonl")}
3372
+ Report: ${path8.join(outDir, "manifest.json")}${opts.noEmbed ? "" : `
3111
3373
  Vectors: one <file>.embeddings.json next to each document (incremental \u2014 re-run any time)`}`);
3112
3374
  }
3113
3375
 
@@ -3123,8 +3385,11 @@ The CLI exists for these workflows:
3123
3385
  \u2022 JDF \u2192 vectors embed those chunks, incrementally, for a vector store.
3124
3386
  \u2022 video \u2192 text attach a time-stamped transcript to a video element so
3125
3387
  RAG retrieves "video at 02:13", not just "a video".
3388
+ \u2022 image \u2192 text OCR + a vision caption for every image so charts and
3389
+ scanned pages are retrievable, not skipped.
3126
3390
  \u2022 folder \u2192 index one command over a directory of .jdf/.jdfx: transcribe,
3127
- chunk, embed incrementally, write .jdf-rag/index.jsonl.
3391
+ describe, chunk, embed incrementally, write .jdf-rag/index.jsonl.
3392
+ Reports media coverage; --strict fails when anything has no text.
3128
3393
 
3129
3394
  Usage:
3130
3395
  jdf validate <file.jdf>
@@ -3132,7 +3397,8 @@ Usage:
3132
3397
  jdf chunk <file.{jdf,jdfx}> [--strategy section|element|fixed] [--format jsonl|json|inline] [--max-tokens N] [-o out]
3133
3398
  jdf embed <file.{jdf,jdfx}> [--provider ollama|openai] [--model NAME] [--strategy \u2026] [--incremental] [-o out]
3134
3399
  jdf transcribe <file.{jdf,jdfx}> [--from subs.srt|.vtt|.json] [--provider whisper-cli|openai] [--element ID] [--chapters FILE] [-o out]
3135
- jdf rag <dir> [--provider ollama|openai] [--model NAME] [--transcribe none|whisper-cli|openai] [--no-embed] [--dry-run] [--out DIR]
3400
+ jdf describe <file.{jdf,jdfx}> [--ocr tesseract|openai|none] [--caption ollama|openai|none] [--caption-model M] [--ocr-language eng] [--element ID] [--force] [-o out]
3401
+ jdf rag <dir> [--provider ollama|openai] [--model NAME] [--transcribe none|whisper-cli|openai] [--ocr none|tesseract|openai] [--caption none|ollama|openai] [--strict] [--no-embed] [--dry-run] [--out DIR]
3136
3402
  jdf --help
3137
3403
 
3138
3404
  Commands:
@@ -3141,7 +3407,8 @@ Commands:
3141
3407
  chunk Split a JDF document into retrieval-ready chunks (offline, deterministic)
3142
3408
  embed Compute embeddings for the chunks (local via Ollama by default)
3143
3409
  transcribe Store time-stamped text on a video element (import SRT/VTT/JSON, or run Whisper)
3144
- rag Make a whole folder retrieval-ready (finds .jdf/.jdfx, transcribes, chunks, embeds, indexes)
3410
+ describe Give images text: OCR blocks (tesseract.js, local) + a caption (Ollama vision model, local)
3411
+ rag Make a whole folder retrieval-ready (finds .jdf/.jdfx, transcribes, describes, chunks, embeds, indexes)
3145
3412
 
3146
3413
  Flags:
3147
3414
  -o, --output <path> Explicit output path
@@ -3168,6 +3435,12 @@ Flags:
3168
3435
  --prompt <text> transcribe: Whisper vocabulary hint (names, acronyms) \u2014 not an instruction
3169
3436
  --window <sec> chunk/embed/rag: transcript window per video chunk (default 45)
3170
3437
  --transcribe <p> rag: none (default) | whisper-cli | openai \u2014 for videos that have no transcript yet
3438
+ --ocr <p> describe/rag/convert: tesseract (local WASM) | openai | none
3439
+ --caption <p> describe/rag: ollama (local vision model, default moondream) | openai | none
3440
+ --caption-model describe/rag: vision model name (ollama: qwen2.5vl:3b, llava\u2026; openai: gpt-4o-mini\u2026)
3441
+ --ocr-language describe: tesseract language(s), e.g. eng, tur, eng+tur (default eng)
3442
+ --force describe: redo images that already have text
3443
+ --strict rag: exit 1 if any image/video is still without text after the run
3171
3444
  --no-embed rag: chunk + index only
3172
3445
  --dry-run rag: list what would happen, write nothing
3173
3446
  --out <dir> rag: index folder (default <dir>/.jdf-rag)
@@ -3187,9 +3460,11 @@ Examples:
3187
3460
  jdf embed report.jdf --provider openai --incremental
3188
3461
  jdf transcribe talk.jdfx --from talk.srt --chapters chapters.txt # then: jdf chunk talk.jdfx
3189
3462
  jdf transcribe talk.jdfx --provider openai --language en --prompt "JDF, jdfx, Ollama"
3190
- jdf rag ./knowledge-base --transcribe openai # whole folder \u2192 .jdf-rag/index.jsonl
3463
+ jdf describe report.jdfx # OCR (tesseract) + caption (Ollama qwen2.5vl), local
3464
+ jdf convert scan.pdf --ocr tesseract # scanned pages get OCR text instead of silence
3465
+ jdf rag ./knowledge-base --transcribe openai --ocr tesseract --caption ollama --strict
3191
3466
  `;
3192
- var BOOLEAN_FLAGS = /* @__PURE__ */ new Set(["help", "h", "json", "verbose", "skip-validate", "incremental", "no-auto-start", "drop-invisible-text", "no-embed", "dry-run"]);
3467
+ var BOOLEAN_FLAGS = /* @__PURE__ */ new Set(["help", "h", "json", "verbose", "skip-validate", "incremental", "no-auto-start", "drop-invisible-text", "no-embed", "dry-run", "force", "strict"]);
3193
3468
  function parseArgs(argv) {
3194
3469
  const positional = [];
3195
3470
  const flags = {};
@@ -3272,7 +3547,8 @@ async function main() {
3272
3547
  await importPdf(input, output, {
3273
3548
  forceJson,
3274
3549
  password: typeof flags.password === "string" ? flags.password : void 0,
3275
- dropInvisibleText: flags["drop-invisible-text"] === true
3550
+ dropInvisibleText: flags["drop-invisible-text"] === true,
3551
+ ocr: typeof flags.ocr === "string" ? flags.ocr : void 0
3276
3552
  });
3277
3553
  process.exit(0);
3278
3554
  } else if (lower.endsWith(".json")) {
@@ -3316,6 +3592,23 @@ async function main() {
3316
3592
  });
3317
3593
  process.exit(0);
3318
3594
  }
3595
+ case "describe": {
3596
+ const input = positional[0];
3597
+ if (!input) {
3598
+ console.error("Usage: jdf describe <file.{jdf,jdfx}> [--ocr tesseract|openai|none] [--caption ollama|openai|none] [--caption-model M] [--ocr-language eng] [--element id|n] [--force] [-o out]");
3599
+ process.exit(1);
3600
+ }
3601
+ await describeFile(input, {
3602
+ ocr: typeof flags.ocr === "string" ? flags.ocr : void 0,
3603
+ caption: typeof flags.caption === "string" ? flags.caption : void 0,
3604
+ captionModel: typeof flags["caption-model"] === "string" ? flags["caption-model"] : void 0,
3605
+ ocrLanguage: typeof flags["ocr-language"] === "string" ? flags["ocr-language"] : void 0,
3606
+ element: typeof flags.element === "string" ? flags.element : void 0,
3607
+ force: flags.force === true,
3608
+ output: typeof flags.output === "string" ? flags.output : void 0
3609
+ });
3610
+ process.exit(process.exitCode ?? 0);
3611
+ }
3319
3612
  case "rag": {
3320
3613
  const input = positional[0];
3321
3614
  if (!input) {
@@ -3332,11 +3625,15 @@ async function main() {
3332
3625
  transcribeModel: typeof flags["transcribe-model"] === "string" ? flags["transcribe-model"] : void 0,
3333
3626
  language: typeof flags.language === "string" ? flags.language : void 0,
3334
3627
  prompt: typeof flags.prompt === "string" ? flags.prompt : void 0,
3628
+ ocr: typeof flags.ocr === "string" ? flags.ocr : void 0,
3629
+ caption: typeof flags.caption === "string" ? flags.caption : void 0,
3630
+ captionModel: typeof flags["caption-model"] === "string" ? flags["caption-model"] : void 0,
3631
+ strict: flags.strict === true,
3335
3632
  noEmbed: flags["no-embed"] === true,
3336
3633
  dryRun: flags["dry-run"] === true,
3337
3634
  out: typeof flags.out === "string" ? flags.out : void 0
3338
3635
  });
3339
- process.exit(0);
3636
+ process.exit(process.exitCode ?? 0);
3340
3637
  }
3341
3638
  case "embed": {
3342
3639
  const input = positional[0];
@@ -343,7 +343,17 @@
343
343
  "anyOf": [{ "required": ["resource"] }, { "required": ["src"] }],
344
344
  "properties": {
345
345
  "type": { "const": "image" },
346
+ "id": { "type": "string" },
346
347
  "resource": { "type": "string" }, "src": { "type": "string" }, "alt": { "type": "string" },
348
+ "caption": { "type": "string" }, "captionSource": { "type": "string" },
349
+ "ocr": {
350
+ "type": "object",
351
+ "required": ["blocks"],
352
+ "properties": {
353
+ "language": { "type": "string" }, "source": { "type": "string" }, "created": { "type": "string" },
354
+ "blocks": { "type": "array", "items": { "type": "object", "required": ["text"], "properties": { "text": { "type": "string" }, "confidence": { "type": "number" }, "bbox": { "type": "object", "properties": { "x": { "type": "number" }, "y": { "type": "number" }, "w": { "type": "number" }, "h": { "type": "number" } } } } } }
355
+ }
356
+ },
347
357
  "position": { "$ref": "#/definitions/Position" },
348
358
  "width": { "type": "number" }, "height": { "type": "number" },
349
359
  "fit": { "type": "string", "enum": ["contain","cover","fill","none"] },
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@uurtech/jdf-cli",
3
- "version": "0.2.1",
3
+ "version": "0.2.2",
4
4
  "description": "Command-line tool for the JDF (JSON Document Format) — validate and convert documents.",
5
5
  "license": "MIT",
6
6
  "author": "Ugur Kazdal",
@@ -52,7 +52,8 @@
52
52
  "ajv": "^8.17.1",
53
53
  "ajv-formats": "^3.0.1",
54
54
  "jszip": "^3.10.1",
55
- "pdfjs-dist": "^4.10.38"
55
+ "pdfjs-dist": "^4.10.38",
56
+ "tesseract.js": "^7.0.0"
56
57
  },
57
58
  "devDependencies": {
58
59
  "@jdf/core": "workspace:*",