@uurtech/jdf-cli 0.2.1 → 0.2.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.js +417 -120
- package/dist/jdf-schema.json +10 -0
- package/package.json +3 -2
package/dist/index.js
CHANGED
|
@@ -1,12 +1,13 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
|
-
import
|
|
3
|
-
import
|
|
2
|
+
import fs8 from 'fs';
|
|
3
|
+
import path8 from 'path';
|
|
4
4
|
import { fileURLToPath } from 'url';
|
|
5
5
|
import Ajv from 'ajv';
|
|
6
6
|
import addFormats from 'ajv-formats';
|
|
7
7
|
import JSZip from 'jszip';
|
|
8
8
|
import crypto, { createHash } from 'crypto';
|
|
9
9
|
import { readFile } from 'fs/promises';
|
|
10
|
+
import os from 'os';
|
|
10
11
|
import { execFileSync, spawnSync } from 'child_process';
|
|
11
12
|
|
|
12
13
|
var __require = /* @__PURE__ */ ((x) => typeof require !== "undefined" ? require : typeof Proxy !== "undefined" ? new Proxy(x, {
|
|
@@ -23,17 +24,17 @@ var JDFX_MANIFEST_PATH = "manifest.json";
|
|
|
23
24
|
var JDFX_ASSET_DIR = "assets";
|
|
24
25
|
|
|
25
26
|
// src/commands/validate.ts
|
|
26
|
-
var __dirname$1 =
|
|
27
|
+
var __dirname$1 = path8.dirname(fileURLToPath(import.meta.url));
|
|
27
28
|
function resolveSchemaPath() {
|
|
28
|
-
const bundled =
|
|
29
|
-
if (
|
|
30
|
-
const dev =
|
|
29
|
+
const bundled = path8.resolve(__dirname$1, "jdf-schema.json");
|
|
30
|
+
if (fs8.existsSync(bundled)) return bundled;
|
|
31
|
+
const dev = path8.resolve(__dirname$1, "../../../../spec/jdf-schema.json");
|
|
31
32
|
return dev;
|
|
32
33
|
}
|
|
33
34
|
var SCHEMA_PATH = resolveSchemaPath();
|
|
34
35
|
async function loadDocument(filePath) {
|
|
35
36
|
if (filePath.toLowerCase().endsWith(".jdfx")) {
|
|
36
|
-
const zip = await JSZip.loadAsync(
|
|
37
|
+
const zip = await JSZip.loadAsync(fs8.readFileSync(filePath));
|
|
37
38
|
const docFile = zip.file(JDFX_DOCUMENT_PATH);
|
|
38
39
|
if (!docFile) {
|
|
39
40
|
console.error(`\u2717 Bundle missing ${JDFX_DOCUMENT_PATH}`);
|
|
@@ -58,11 +59,11 @@ async function loadDocument(filePath) {
|
|
|
58
59
|
}
|
|
59
60
|
return { doc, bundle: { manifest, assetCount } };
|
|
60
61
|
}
|
|
61
|
-
return { doc: JSON.parse(
|
|
62
|
+
return { doc: JSON.parse(fs8.readFileSync(filePath, "utf-8")) };
|
|
62
63
|
}
|
|
63
64
|
async function validate(file) {
|
|
64
|
-
const filePath =
|
|
65
|
-
if (!
|
|
65
|
+
const filePath = path8.resolve(file);
|
|
66
|
+
if (!fs8.existsSync(filePath)) {
|
|
66
67
|
console.error(`File not found: ${filePath}`);
|
|
67
68
|
return false;
|
|
68
69
|
}
|
|
@@ -75,11 +76,11 @@ async function validate(file) {
|
|
|
75
76
|
}
|
|
76
77
|
if (!loaded) return false;
|
|
77
78
|
const { doc, bundle } = loaded;
|
|
78
|
-
if (!
|
|
79
|
+
if (!fs8.existsSync(SCHEMA_PATH)) {
|
|
79
80
|
console.error(`Schema not found at ${SCHEMA_PATH}`);
|
|
80
81
|
return false;
|
|
81
82
|
}
|
|
82
|
-
const schema = JSON.parse(
|
|
83
|
+
const schema = JSON.parse(fs8.readFileSync(SCHEMA_PATH, "utf-8"));
|
|
83
84
|
const ajv = new Ajv({ allErrors: true, strict: false });
|
|
84
85
|
addFormats(ajv);
|
|
85
86
|
const validateFn = ajv.compile(schema);
|
|
@@ -88,7 +89,7 @@ async function validate(file) {
|
|
|
88
89
|
const d = doc;
|
|
89
90
|
const pageCount = Array.isArray(d.pages) ? d.pages.length : 0;
|
|
90
91
|
const elCount = Array.isArray(d.pages) ? d.pages.reduce((acc, p) => acc + (Array.isArray(p?.elements) ? p.elements.length : 0), 0) : 0;
|
|
91
|
-
console.log(`\u2713 Valid: ${
|
|
92
|
+
console.log(`\u2713 Valid: ${path8.basename(filePath)}`);
|
|
92
93
|
console.log(` Format: ${d.$jdf}${bundle ? " (jdfx bundle)" : ""}`);
|
|
93
94
|
console.log(` Title: ${d.meta?.title}`);
|
|
94
95
|
console.log(` Pages: ${pageCount}`);
|
|
@@ -101,7 +102,7 @@ async function validate(file) {
|
|
|
101
102
|
}
|
|
102
103
|
return true;
|
|
103
104
|
}
|
|
104
|
-
console.error(`\u2717 Invalid: ${
|
|
105
|
+
console.error(`\u2717 Invalid: ${path8.basename(filePath)}`);
|
|
105
106
|
for (const err of validateFn.errors || []) {
|
|
106
107
|
const loc = err.instancePath || "(root)";
|
|
107
108
|
console.error(` ${loc} \u2014 ${err.message}`);
|
|
@@ -304,13 +305,13 @@ function stripInline(text) {
|
|
|
304
305
|
return parseInline(text).map((r) => r.text).join("");
|
|
305
306
|
}
|
|
306
307
|
async function importMarkdown(inputPath, outputPath) {
|
|
307
|
-
const input =
|
|
308
|
+
const input = path8.resolve(inputPath);
|
|
308
309
|
console.log(`Importing: ${input}`);
|
|
309
|
-
const content =
|
|
310
|
-
const doc = convertMarkdownToJdf(content,
|
|
310
|
+
const content = fs8.readFileSync(input, "utf-8");
|
|
311
|
+
const doc = convertMarkdownToJdf(content, path8.basename(input, path8.extname(input)), path8.dirname(input));
|
|
311
312
|
let output;
|
|
312
313
|
if (outputPath) {
|
|
313
|
-
output =
|
|
314
|
+
output = path8.resolve(outputPath);
|
|
314
315
|
} else {
|
|
315
316
|
const stem = input.replace(/\.(md|markdown)$/i, "");
|
|
316
317
|
output = stem + (shouldUseJdfx(doc) ? ".jdfx" : ".jdf");
|
|
@@ -318,11 +319,11 @@ async function importMarkdown(inputPath, outputPath) {
|
|
|
318
319
|
console.log(`Output: ${output}`);
|
|
319
320
|
if (output.toLowerCase().endsWith(".jdfx")) {
|
|
320
321
|
const { bytes, manifest } = await packJdfx(doc);
|
|
321
|
-
|
|
322
|
+
fs8.writeFileSync(output, bytes);
|
|
322
323
|
console.log(`
|
|
323
324
|
Done! Created ${doc.pages.length} page(s), ${manifest.assets.length} asset(s) bundled`);
|
|
324
325
|
} else {
|
|
325
|
-
|
|
326
|
+
fs8.writeFileSync(output, JSON.stringify(doc, null, 2));
|
|
326
327
|
console.log(`
|
|
327
328
|
Done! Created ${doc.pages.length} page(s)`);
|
|
328
329
|
}
|
|
@@ -339,10 +340,10 @@ var MIME_BY_EXT2 = {
|
|
|
339
340
|
};
|
|
340
341
|
function resolveImageSrc(src, baseDir) {
|
|
341
342
|
if (/^(https?:|data:|file:)/i.test(src)) return src;
|
|
342
|
-
const abs =
|
|
343
|
+
const abs = path8.isAbsolute(src) ? src : path8.resolve(baseDir, src);
|
|
343
344
|
try {
|
|
344
|
-
const bytes =
|
|
345
|
-
const ext =
|
|
345
|
+
const bytes = fs8.readFileSync(abs);
|
|
346
|
+
const ext = path8.extname(abs).slice(1).toLowerCase();
|
|
346
347
|
const mime = MIME_BY_EXT2[ext] || "application/octet-stream";
|
|
347
348
|
return `data:${mime};base64,${bytes.toString("base64")}`;
|
|
348
349
|
} catch {
|
|
@@ -2188,25 +2189,209 @@ async function importPdfToJdf2(source, title, options = {}) {
|
|
|
2188
2189
|
};
|
|
2189
2190
|
return importPdfToJdf(source, title, runtime, options);
|
|
2190
2191
|
}
|
|
2192
|
+
var DEFAULT_OLLAMA_MODEL = "qwen2.5vl:3b";
|
|
2193
|
+
var DEFAULT_OPENAI_MODEL = "gpt-4o-mini";
|
|
2194
|
+
var OLLAMA_HOST = process.env.OLLAMA_HOST || "http://localhost:11434";
|
|
2195
|
+
async function loadDoc(file) {
|
|
2196
|
+
if (file.toLowerCase().endsWith(".jdfx")) {
|
|
2197
|
+
const zip = await JSZip.loadAsync(fs8.readFileSync(file));
|
|
2198
|
+
const f = zip.file(JDFX_DOCUMENT_PATH);
|
|
2199
|
+
if (!f) throw new Error(`Bundle missing ${JDFX_DOCUMENT_PATH}`);
|
|
2200
|
+
const doc = JSON.parse(await f.async("string"));
|
|
2201
|
+
const manifest = zip.file("manifest.json") ? JSON.parse(await zip.file("manifest.json").async("string")) : { assets: [] };
|
|
2202
|
+
for (const a of manifest.assets ?? []) {
|
|
2203
|
+
const af = zip.file(a.path);
|
|
2204
|
+
if (!af) continue;
|
|
2205
|
+
const data = (await af.async("nodebuffer")).toString("base64");
|
|
2206
|
+
const res = { src: "embedded", mimeType: a.mimeType, data };
|
|
2207
|
+
doc.resources ??= {};
|
|
2208
|
+
if (/^video\//i.test(a.mimeType || "")) (doc.resources.videos ??= {})[a.id] = res;
|
|
2209
|
+
else (doc.resources.images ??= {})[a.id] = res;
|
|
2210
|
+
}
|
|
2211
|
+
return { doc, bundle: true };
|
|
2212
|
+
}
|
|
2213
|
+
return { doc: JSON.parse(fs8.readFileSync(file, "utf-8")), bundle: false };
|
|
2214
|
+
}
|
|
2215
|
+
function findImages(doc) {
|
|
2216
|
+
const out = [];
|
|
2217
|
+
const walk2 = (els, page) => {
|
|
2218
|
+
for (const el of els ?? []) {
|
|
2219
|
+
if (el?.type === "image") out.push({ el, page, index: out.length });
|
|
2220
|
+
if (el?.elements) walk2(el.elements, page);
|
|
2221
|
+
}
|
|
2222
|
+
};
|
|
2223
|
+
doc.pages.forEach((p, i) => walk2(p.elements, i + 1));
|
|
2224
|
+
return out;
|
|
2225
|
+
}
|
|
2226
|
+
async function imageBytes(doc, el, docDir) {
|
|
2227
|
+
const fromData = (d, fallback) => {
|
|
2228
|
+
const m = d.match(/^data:([^;,]+)?[^,]*,(.*)$/s);
|
|
2229
|
+
return m ? { bytes: Buffer.from(m[2], "base64"), mime: m[1] || fallback } : { bytes: Buffer.from(d, "base64"), mime: fallback };
|
|
2230
|
+
};
|
|
2231
|
+
const res = el.resource ? doc.resources?.images?.[el.resource] ?? doc.resources?.[el.resource] : void 0;
|
|
2232
|
+
if (res?.data) return fromData(String(res.data), res.mimeType || "image/png");
|
|
2233
|
+
if (res?.path) {
|
|
2234
|
+
const p = path8.resolve(docDir, res.path);
|
|
2235
|
+
return fs8.existsSync(p) ? { bytes: fs8.readFileSync(p), mime: res.mimeType || "image/png" } : null;
|
|
2236
|
+
}
|
|
2237
|
+
const src = el.src;
|
|
2238
|
+
if (!src) return null;
|
|
2239
|
+
if (src.startsWith("data:")) return fromData(src, "image/png");
|
|
2240
|
+
if (/^https?:\/\//i.test(src)) {
|
|
2241
|
+
const r = await fetch(src);
|
|
2242
|
+
if (!r.ok) throw new Error(`download failed ${r.status}: ${src}`);
|
|
2243
|
+
return { bytes: Buffer.from(await r.arrayBuffer()), mime: r.headers.get("content-type") || "image/png" };
|
|
2244
|
+
}
|
|
2245
|
+
const local = path8.resolve(docDir, src);
|
|
2246
|
+
return fs8.existsSync(local) ? { bytes: fs8.readFileSync(local), mime: "image/png" } : null;
|
|
2247
|
+
}
|
|
2248
|
+
async function ocrTesseract(bytes, lang) {
|
|
2249
|
+
const { createWorker } = await import('tesseract.js');
|
|
2250
|
+
const cachePath = path8.join(os.homedir(), ".cache", "jdf", "tesseract");
|
|
2251
|
+
fs8.mkdirSync(cachePath, { recursive: true });
|
|
2252
|
+
const worker = await createWorker(lang, 1, { cachePath, logger: () => {
|
|
2253
|
+
} });
|
|
2254
|
+
try {
|
|
2255
|
+
const { data } = await worker.recognize(bytes, {}, { text: true, blocks: true });
|
|
2256
|
+
let w = 1, h = 1;
|
|
2257
|
+
try {
|
|
2258
|
+
const { loadImage } = await import('@napi-rs/canvas');
|
|
2259
|
+
const im = await loadImage(bytes);
|
|
2260
|
+
w = im.width || 1;
|
|
2261
|
+
h = im.height || 1;
|
|
2262
|
+
} catch {
|
|
2263
|
+
}
|
|
2264
|
+
const lines = data.blocks?.flatMap((b) => b.paragraphs?.flatMap((p) => p.lines ?? []) ?? []) ?? data.lines ?? [];
|
|
2265
|
+
const blocks = lines.map((ln) => ({ text: String(ln.text ?? "").replace(/\s+/g, " ").trim(), confidence: ln.confidence != null ? Math.round(ln.confidence) / 100 : void 0, bbox: ln.bbox ? { x: +(ln.bbox.x0 / w).toFixed(4), y: +(ln.bbox.y0 / h).toFixed(4), w: +((ln.bbox.x1 - ln.bbox.x0) / w).toFixed(4), h: +((ln.bbox.y1 - ln.bbox.y0) / h).toFixed(4) } : void 0 })).filter((b) => b.text.length > 0 && (b.confidence == null || b.confidence >= 0.3));
|
|
2266
|
+
if (!blocks.length && String(data.text ?? "").trim()) blocks.push({ text: String(data.text).replace(/\s+/g, " ").trim() });
|
|
2267
|
+
return { blocks, source: `tesseract.js:${lang}` };
|
|
2268
|
+
} finally {
|
|
2269
|
+
await worker.terminate();
|
|
2270
|
+
}
|
|
2271
|
+
}
|
|
2272
|
+
async function openaiVision(bytes, mime, model, prompt2) {
|
|
2273
|
+
const key = process.env.OPENAI_API_KEY;
|
|
2274
|
+
if (!key) throw new Error("OPENAI_API_KEY is not set");
|
|
2275
|
+
const base = process.env.OPENAI_BASE_URL || "https://api.openai.com/v1";
|
|
2276
|
+
const r = await fetch(`${base}/chat/completions`, { method: "POST", headers: { Authorization: `Bearer ${key}`, "content-type": "application/json" }, body: JSON.stringify({ model, messages: [{ role: "user", content: [{ type: "text", text: prompt2 }, { type: "image_url", image_url: { url: `data:${mime};base64,${bytes.toString("base64")}` } }] }], max_tokens: 800 }) });
|
|
2277
|
+
if (!r.ok) throw new Error(`OpenAI vision failed ${r.status}: ${(await r.text()).slice(0, 300)}`);
|
|
2278
|
+
const j = await r.json();
|
|
2279
|
+
return String(j.choices?.[0]?.message?.content ?? "").trim();
|
|
2280
|
+
}
|
|
2281
|
+
async function ollamaVision(bytes, model, prompt2) {
|
|
2282
|
+
const r = await fetch(`${OLLAMA_HOST}/api/generate`, { method: "POST", body: JSON.stringify({ model, prompt: prompt2, images: [bytes.toString("base64")], stream: false, options: { temperature: 0 } }) });
|
|
2283
|
+
if (!r.ok) throw new Error(`Ollama vision failed ${r.status}: ${(await r.text()).slice(0, 300)} \u2014 is the model pulled? (ollama pull ${model})`);
|
|
2284
|
+
const j = await r.json();
|
|
2285
|
+
return String(j.response ?? "").trim();
|
|
2286
|
+
}
|
|
2287
|
+
var CAPTION_PROMPT = "Describe this image for a search index in one or two factual sentences: what it shows, any chart type, axes, trends, labels, names and numbers you can read. No preamble.";
|
|
2288
|
+
var OCR_PROMPT = "Transcribe all text visible in this image exactly, line by line, top to bottom, left to right. Output only the text.";
|
|
2289
|
+
async function describeDocument(doc, docDir, opts = {}) {
|
|
2290
|
+
const ocrP = opts.ocr ?? "tesseract";
|
|
2291
|
+
const capP = opts.caption ?? "ollama";
|
|
2292
|
+
const lang = opts.ocrLanguage ?? "eng";
|
|
2293
|
+
const images = findImages(doc);
|
|
2294
|
+
const targets = opts.element != null ? images.filter((im) => im.el.id === opts.element || String(im.index) === opts.element) : images;
|
|
2295
|
+
if (opts.element != null && !targets.length) throw new Error(`no image element "${opts.element}" (have: ${images.map((i) => i.el.id ?? `#${i.index}`).join(", ") || "none"})`);
|
|
2296
|
+
const stats = { ocr: 0, captions: 0, skipped: 0, failed: [] };
|
|
2297
|
+
for (const im of targets) {
|
|
2298
|
+
const el = im.el;
|
|
2299
|
+
if (!el.id) el.id = `image-${im.index + 1}`;
|
|
2300
|
+
const needOcr = ocrP !== "none" && (opts.force || !el.ocr?.blocks?.length);
|
|
2301
|
+
const needCap = capP !== "none" && (opts.force || !el.caption);
|
|
2302
|
+
if (!needOcr && !needCap) {
|
|
2303
|
+
stats.skipped++;
|
|
2304
|
+
continue;
|
|
2305
|
+
}
|
|
2306
|
+
const img = await imageBytes(doc, el, docDir);
|
|
2307
|
+
if (!img) {
|
|
2308
|
+
stats.failed.push(`${el.id}: image bytes not reachable`);
|
|
2309
|
+
continue;
|
|
2310
|
+
}
|
|
2311
|
+
if (needOcr) {
|
|
2312
|
+
try {
|
|
2313
|
+
if (ocrP === "tesseract") {
|
|
2314
|
+
const r = await ocrTesseract(img.bytes, lang);
|
|
2315
|
+
el.ocr = { language: lang, source: r.source, created: (/* @__PURE__ */ new Date()).toISOString(), blocks: r.blocks };
|
|
2316
|
+
} else {
|
|
2317
|
+
const text = await openaiVision(img.bytes, img.mime, opts.captionModel ?? DEFAULT_OPENAI_MODEL, OCR_PROMPT);
|
|
2318
|
+
el.ocr = { source: `openai:${opts.captionModel ?? DEFAULT_OPENAI_MODEL}`, created: (/* @__PURE__ */ new Date()).toISOString(), blocks: text.split(/\r?\n/).map((t) => t.trim()).filter(Boolean).map((t) => ({ text: t })) };
|
|
2319
|
+
}
|
|
2320
|
+
stats.ocr++;
|
|
2321
|
+
} catch (e) {
|
|
2322
|
+
stats.failed.push(`${el.id} ocr: ${e.message}`);
|
|
2323
|
+
}
|
|
2324
|
+
}
|
|
2325
|
+
if (needCap) {
|
|
2326
|
+
try {
|
|
2327
|
+
const model = opts.captionModel ?? (capP === "ollama" ? DEFAULT_OLLAMA_MODEL : DEFAULT_OPENAI_MODEL);
|
|
2328
|
+
const text = capP === "ollama" ? await ollamaVision(img.bytes, model, CAPTION_PROMPT) : await openaiVision(img.bytes, img.mime, model, CAPTION_PROMPT);
|
|
2329
|
+
if (text) {
|
|
2330
|
+
el.caption = text;
|
|
2331
|
+
el.captionSource = `${capP}:${model}`;
|
|
2332
|
+
stats.captions++;
|
|
2333
|
+
}
|
|
2334
|
+
} catch (e) {
|
|
2335
|
+
stats.failed.push(`${el.id} caption: ${e.message}`);
|
|
2336
|
+
}
|
|
2337
|
+
}
|
|
2338
|
+
if (!opts.quiet) console.log(` \xB7 ${el.id} (page ${im.page}): ${needOcr ? `ocr ${el.ocr?.blocks?.length ?? 0} block(s)` : "ocr kept"}${needCap ? ` \xB7 caption ${el.caption ? `"${String(el.caption).slice(0, 70)}${String(el.caption).length > 70 ? "\u2026" : ""}"` : "\u2014"}` : ""}`);
|
|
2339
|
+
}
|
|
2340
|
+
return stats;
|
|
2341
|
+
}
|
|
2342
|
+
async function describeFile(inputPath, opts = {}) {
|
|
2343
|
+
const input = path8.resolve(inputPath);
|
|
2344
|
+
if (!fs8.existsSync(input)) throw new Error(`File not found: ${input}`);
|
|
2345
|
+
const { doc, bundle } = await loadDoc(input);
|
|
2346
|
+
const images = findImages(doc);
|
|
2347
|
+
if (!images.length) {
|
|
2348
|
+
console.log(`No image elements in ${path8.basename(input)} \u2014 nothing to describe.`);
|
|
2349
|
+
return;
|
|
2350
|
+
}
|
|
2351
|
+
console.log(`Describing: ${path8.basename(input)} \u2014 ${images.length} image(s); ocr=${opts.ocr ?? "tesseract"} caption=${opts.caption ?? "ollama"}${(opts.caption ?? "ollama") === "ollama" ? ` (${opts.captionModel ?? DEFAULT_OLLAMA_MODEL}, local)` : ""}`);
|
|
2352
|
+
const stats = await describeDocument(doc, path8.dirname(input), opts);
|
|
2353
|
+
const output = opts.output ? path8.resolve(opts.output) : input;
|
|
2354
|
+
if (output.toLowerCase().endsWith(".jdfx") || bundle && !opts.output) fs8.writeFileSync(output, (await packJdfx(doc)).bytes);
|
|
2355
|
+
else fs8.writeFileSync(output, JSON.stringify(doc, null, 2));
|
|
2356
|
+
console.log(`Done: ${stats.ocr} OCR, ${stats.captions} caption(s), ${stats.skipped} already had text${stats.failed.length ? `, ${stats.failed.length} failed` : ""}.`);
|
|
2357
|
+
for (const f of stats.failed) console.warn(` ! ${f}`);
|
|
2358
|
+
console.log(`Output: ${output}
|
|
2359
|
+
Next: jdf chunk ${path8.basename(output)} # image text is now part of the chunks`);
|
|
2360
|
+
if (stats.failed.length && stats.ocr + stats.captions === 0) process.exitCode = 1;
|
|
2361
|
+
}
|
|
2191
2362
|
|
|
2192
2363
|
// src/commands/import-pdf.ts
|
|
2193
2364
|
async function importPdf(inputPath, outputPath, options = {}) {
|
|
2194
|
-
const input =
|
|
2195
|
-
if (!
|
|
2365
|
+
const input = path8.resolve(inputPath);
|
|
2366
|
+
if (!fs8.existsSync(input)) {
|
|
2196
2367
|
console.error(`File not found: ${input}`);
|
|
2197
2368
|
process.exit(1);
|
|
2198
2369
|
}
|
|
2199
2370
|
console.log(`Importing: ${input}`);
|
|
2200
|
-
const title =
|
|
2371
|
+
const title = path8.basename(input, path8.extname(input));
|
|
2201
2372
|
const t0 = Date.now();
|
|
2202
2373
|
const doc = await importPdfToJdf2(input, title, {
|
|
2203
2374
|
password: options.password,
|
|
2204
2375
|
invisibleText: options.dropInvisibleText ? "drop" : "keep"
|
|
2205
2376
|
});
|
|
2206
2377
|
console.log(`Parsed in ${((Date.now() - t0) / 1e3).toFixed(1)}s \u2014 ${doc.pages.length} page(s)`);
|
|
2378
|
+
const scanned = doc.pages.filter((p) => !p.elements.some((e) => e.type === "text" || e.type === "richtext" || e.type === "table") && p.elements.some((e) => e.type === "image"));
|
|
2379
|
+
if (scanned.length) {
|
|
2380
|
+
if (options.ocr && options.ocr !== "none") {
|
|
2381
|
+
console.log(`OCR: ${scanned.length} scanned page(s) \u2192 ${options.ocr}`);
|
|
2382
|
+
for (const p of scanned) for (const e of p.elements) if (e.type === "image" && !e.id) e.id = `scan-${doc.pages.indexOf(p) + 1}`;
|
|
2383
|
+
const ids = scanned.flatMap((p) => p.elements.filter((e) => e.type === "image").map((e) => e.id));
|
|
2384
|
+
for (const id of ids) {
|
|
2385
|
+
const st = await describeDocument(doc, path8.dirname(input), { element: id, ocr: options.ocr, caption: "none", quiet: true });
|
|
2386
|
+
if (st.failed.length) console.warn(` ! ${st.failed.join("; ")}`);
|
|
2387
|
+
}
|
|
2388
|
+
} else {
|
|
2389
|
+
console.warn(`! ${scanned.length} page(s) have no text layer (scanned). RAG will skip them \u2014 re-run with --ocr tesseract (local) or --ocr openai.`);
|
|
2390
|
+
}
|
|
2391
|
+
}
|
|
2207
2392
|
let output;
|
|
2208
2393
|
if (outputPath) {
|
|
2209
|
-
output =
|
|
2394
|
+
output = path8.resolve(outputPath);
|
|
2210
2395
|
} else {
|
|
2211
2396
|
const stem = input.replace(/\.pdf$/i, "");
|
|
2212
2397
|
const wantJdfx = !options.forceJson && shouldUseJdfx(doc);
|
|
@@ -2215,11 +2400,11 @@ async function importPdf(inputPath, outputPath, options = {}) {
|
|
|
2215
2400
|
console.log(`Output: ${output}`);
|
|
2216
2401
|
if (output.toLowerCase().endsWith(".jdfx")) {
|
|
2217
2402
|
const { bytes, manifest } = await packJdfx(doc);
|
|
2218
|
-
|
|
2403
|
+
fs8.writeFileSync(output, bytes);
|
|
2219
2404
|
console.log(`
|
|
2220
2405
|
Done! Created ${doc.pages.length} page(s), ${manifest.assets.length} asset(s) bundled`);
|
|
2221
2406
|
} else {
|
|
2222
|
-
|
|
2407
|
+
fs8.writeFileSync(output, JSON.stringify(doc, null, 2));
|
|
2223
2408
|
console.log(`
|
|
2224
2409
|
Done! Created ${doc.pages.length} page(s)`);
|
|
2225
2410
|
}
|
|
@@ -2232,23 +2417,23 @@ var ImportJsonError = class extends Error {
|
|
|
2232
2417
|
}
|
|
2233
2418
|
};
|
|
2234
2419
|
async function importJson(inputPath, outputPath, options = {}) {
|
|
2235
|
-
const input =
|
|
2236
|
-
if (!
|
|
2420
|
+
const input = path8.resolve(inputPath);
|
|
2421
|
+
if (!fs8.existsSync(input)) {
|
|
2237
2422
|
throw new ImportJsonError(`File not found: ${input}`);
|
|
2238
2423
|
}
|
|
2239
2424
|
console.log(`Importing: ${input}`);
|
|
2240
|
-
const raw =
|
|
2425
|
+
const raw = fs8.readFileSync(input, "utf-8");
|
|
2241
2426
|
let parsed;
|
|
2242
2427
|
try {
|
|
2243
2428
|
parsed = JSON.parse(raw);
|
|
2244
2429
|
} catch (e) {
|
|
2245
2430
|
throw new ImportJsonError(`Not valid JSON: ${e.message}`);
|
|
2246
2431
|
}
|
|
2247
|
-
const title =
|
|
2432
|
+
const title = path8.basename(input, path8.extname(input));
|
|
2248
2433
|
const doc = normaliseToJdf(parsed, title);
|
|
2249
2434
|
let output;
|
|
2250
2435
|
if (outputPath) {
|
|
2251
|
-
output =
|
|
2436
|
+
output = path8.resolve(outputPath);
|
|
2252
2437
|
} else {
|
|
2253
2438
|
const stem = input.replace(/\.json$/i, "");
|
|
2254
2439
|
const wantJdfx = !options.forceJson && shouldUseJdfx(doc);
|
|
@@ -2257,11 +2442,11 @@ async function importJson(inputPath, outputPath, options = {}) {
|
|
|
2257
2442
|
console.log(`Output: ${output}`);
|
|
2258
2443
|
if (output.toLowerCase().endsWith(".jdfx")) {
|
|
2259
2444
|
const { bytes, manifest } = await packJdfx(doc);
|
|
2260
|
-
|
|
2445
|
+
fs8.writeFileSync(output, bytes);
|
|
2261
2446
|
console.log(`
|
|
2262
2447
|
Done! Created ${doc.pages.length} page(s), ${manifest.assets.length} asset(s) bundled`);
|
|
2263
2448
|
} else {
|
|
2264
|
-
|
|
2449
|
+
fs8.writeFileSync(output, JSON.stringify(doc, null, 2));
|
|
2265
2450
|
console.log(`
|
|
2266
2451
|
Done! Created ${doc.pages.length} page(s)`);
|
|
2267
2452
|
}
|
|
@@ -2363,8 +2548,8 @@ function transcriptChunks(el, elementId2, page, crumb, windowSec, maxTokens) {
|
|
|
2363
2548
|
const body = win.map((sg) => sg.speaker ? `${sg.speaker}: ${sg.text}` : sg.text).join(" ").replace(/\s+/g, " ").trim();
|
|
2364
2549
|
const text = `[${fmtTime(t0)}\u2013${fmtTime(t1)}] ${body}`;
|
|
2365
2550
|
const chapter = chapterAt(t0);
|
|
2366
|
-
const
|
|
2367
|
-
out.push({ id: `${elementId2}@${Math.round(t0)}`, text, path:
|
|
2551
|
+
const path10 = [...crumb, ...el.title ? [String(el.title)] : [], ...chapter ? [String(chapter.title)] : []];
|
|
2552
|
+
out.push({ id: `${elementId2}@${Math.round(t0)}`, text, path: path10, page, types: ["video"], tokens: estimateTokens(text), hash: hashText(text), media: { element: elementId2, t0, t1 } });
|
|
2368
2553
|
win = [];
|
|
2369
2554
|
};
|
|
2370
2555
|
for (const sg of segs) {
|
|
@@ -2433,8 +2618,14 @@ function serializeElement(el) {
|
|
|
2433
2618
|
}
|
|
2434
2619
|
case "checkbox":
|
|
2435
2620
|
return `${e.checked ? "[x]" : "[ ]"} ${e.label ?? ""}`.trim();
|
|
2436
|
-
case "image":
|
|
2437
|
-
|
|
2621
|
+
case "image": {
|
|
2622
|
+
const parts = [];
|
|
2623
|
+
if (e.alt) parts.push(`[image: ${e.alt}]`);
|
|
2624
|
+
if (e.caption) parts.push(String(e.caption).trim());
|
|
2625
|
+
const ocr = (e.ocr?.blocks || []).map((b) => String(b.text ?? "").trim()).filter(Boolean).join("\n");
|
|
2626
|
+
if (ocr) parts.push(ocr);
|
|
2627
|
+
return parts.join("\n");
|
|
2628
|
+
}
|
|
2438
2629
|
case "video":
|
|
2439
2630
|
return e.title ? `[video: ${e.title}]` : "";
|
|
2440
2631
|
case "toc":
|
|
@@ -2554,34 +2745,62 @@ function chunkDocument(doc, options = {}) {
|
|
|
2554
2745
|
}
|
|
2555
2746
|
async function loadJdf(filePath) {
|
|
2556
2747
|
if (filePath.toLowerCase().endsWith(".jdfx")) {
|
|
2557
|
-
const zip = await JSZip.loadAsync(
|
|
2748
|
+
const zip = await JSZip.loadAsync(fs8.readFileSync(filePath));
|
|
2558
2749
|
const docFile = zip.file(JDFX_DOCUMENT_PATH);
|
|
2559
2750
|
if (!docFile) throw new Error(`Bundle missing ${JDFX_DOCUMENT_PATH}`);
|
|
2560
2751
|
return JSON.parse(await docFile.async("string"));
|
|
2561
2752
|
}
|
|
2562
|
-
return JSON.parse(
|
|
2753
|
+
return JSON.parse(fs8.readFileSync(filePath, "utf-8"));
|
|
2754
|
+
}
|
|
2755
|
+
function mediaCoverage(doc) {
|
|
2756
|
+
const cov = { images: { total: 0, covered: 0, missing: [] }, videos: { total: 0, covered: 0, missing: [] } };
|
|
2757
|
+
const walk2 = (els, page) => {
|
|
2758
|
+
for (const el of els ?? []) {
|
|
2759
|
+
if (el?.type === "image") {
|
|
2760
|
+
cov.images.total++;
|
|
2761
|
+
const has = !!(el.caption && String(el.caption).trim()) || !!el.ocr?.blocks?.some((b) => String(b.text ?? "").trim());
|
|
2762
|
+
if (has) cov.images.covered++;
|
|
2763
|
+
else cov.images.missing.push({ id: el.id, page, alt: el.alt });
|
|
2764
|
+
} else if (el?.type === "video") {
|
|
2765
|
+
cov.videos.total++;
|
|
2766
|
+
if (el.transcript?.segments?.length) cov.videos.covered++;
|
|
2767
|
+
else cov.videos.missing.push({ id: el.id, page, title: el.title });
|
|
2768
|
+
}
|
|
2769
|
+
if (el?.elements) walk2(el.elements, page);
|
|
2770
|
+
}
|
|
2771
|
+
};
|
|
2772
|
+
doc.pages.forEach((p, i) => walk2(p.elements, i + 1));
|
|
2773
|
+
return cov;
|
|
2774
|
+
}
|
|
2775
|
+
function coverageSummary(cov) {
|
|
2776
|
+
const parts = [];
|
|
2777
|
+
if (cov.images.missing.length) parts.push(`${cov.images.missing.length} of ${cov.images.total} image(s) have no caption/OCR text \u2192 jdf describe`);
|
|
2778
|
+
if (cov.videos.missing.length) parts.push(`${cov.videos.missing.length} of ${cov.videos.total} video(s) have no transcript \u2192 jdf transcribe`);
|
|
2779
|
+
return parts.length ? parts.join("; ") : null;
|
|
2563
2780
|
}
|
|
2564
2781
|
async function chunkFile(inputPath, opts = {}) {
|
|
2565
|
-
const input =
|
|
2566
|
-
if (!
|
|
2782
|
+
const input = path8.resolve(inputPath);
|
|
2783
|
+
if (!fs8.existsSync(input)) throw new Error(`File not found: ${input}`);
|
|
2567
2784
|
const doc = await loadJdf(input);
|
|
2568
2785
|
const strategy = opts.strategy ?? "section";
|
|
2569
2786
|
const chunks = chunkDocument(doc, { strategy, maxTokens: opts.maxTokens, transcriptWindowSec: opts.transcriptWindowSec });
|
|
2787
|
+
const gap = coverageSummary(mediaCoverage(doc));
|
|
2788
|
+
if (gap) console.warn(` ! media without text (skipped by retrieval): ${gap}`);
|
|
2570
2789
|
const format = opts.format ?? "jsonl";
|
|
2571
2790
|
console.log(`Chunking: ${input}`);
|
|
2572
2791
|
console.log(`Strategy: ${strategy}${opts.maxTokens ? ` (max ${opts.maxTokens} tokens)` : ""}`);
|
|
2573
2792
|
if (format === "inline") {
|
|
2574
|
-
const out = opts.output ?
|
|
2793
|
+
const out = opts.output ? path8.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".jdf");
|
|
2575
2794
|
const withIndex = { ...doc, index: { chunker: `jdf-${strategy}-v1`, chunks } };
|
|
2576
|
-
|
|
2795
|
+
fs8.writeFileSync(out, JSON.stringify(withIndex, null, 2));
|
|
2577
2796
|
console.log(`Output: ${out} (${chunks.length} chunks in "index" block)`);
|
|
2578
2797
|
} else if (format === "json") {
|
|
2579
|
-
const out = opts.output ?
|
|
2580
|
-
|
|
2798
|
+
const out = opts.output ? path8.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".chunks.json");
|
|
2799
|
+
fs8.writeFileSync(out, JSON.stringify(chunks, null, 2));
|
|
2581
2800
|
console.log(`Output: ${out} (${chunks.length} chunks)`);
|
|
2582
2801
|
} else {
|
|
2583
|
-
const out = opts.output ?
|
|
2584
|
-
|
|
2802
|
+
const out = opts.output ? path8.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".chunks.jsonl");
|
|
2803
|
+
fs8.writeFileSync(out, chunks.map((c) => JSON.stringify(c)).join("\n") + "\n");
|
|
2585
2804
|
console.log(`Output: ${out} (${chunks.length} chunks)`);
|
|
2586
2805
|
}
|
|
2587
2806
|
const totalTokens = chunks.reduce((a, c) => a + c.tokens, 0);
|
|
@@ -2604,11 +2823,11 @@ async function embedBatch(provider, model, inputs) {
|
|
|
2604
2823
|
throw new Error(`Unknown embedding provider: ${provider}`);
|
|
2605
2824
|
}
|
|
2606
2825
|
}
|
|
2607
|
-
var
|
|
2826
|
+
var OLLAMA_HOST2 = process.env.OLLAMA_HOST || "http://localhost:11434";
|
|
2608
2827
|
var OLLAMA_CONTAINER = "jdf-ollama";
|
|
2609
2828
|
async function ollamaUp() {
|
|
2610
2829
|
try {
|
|
2611
|
-
const res = await fetch(`${
|
|
2830
|
+
const res = await fetch(`${OLLAMA_HOST2}/api/tags`, { signal: AbortSignal.timeout(1500) });
|
|
2612
2831
|
return res.ok;
|
|
2613
2832
|
} catch {
|
|
2614
2833
|
return false;
|
|
@@ -2633,12 +2852,12 @@ async function ensureOllama(model, autoStart) {
|
|
|
2633
2852
|
docker run -d --name ${OLLAMA_CONTAINER} -p 11434:11434 -v jdf-ollama:/root/.ollama ollama/ollama
|
|
2634
2853
|
docker exec ${OLLAMA_CONTAINER} ollama pull ${model}`;
|
|
2635
2854
|
if (!autoStart) {
|
|
2636
|
-
throw new Error(`Ollama isn't running at ${
|
|
2855
|
+
throw new Error(`Ollama isn't running at ${OLLAMA_HOST2} and --no-auto-start was given.
|
|
2637
2856
|
${manualHint}`);
|
|
2638
2857
|
}
|
|
2639
2858
|
if (!dockerReady()) {
|
|
2640
2859
|
throw new Error(
|
|
2641
|
-
`Ollama isn't running at ${
|
|
2860
|
+
`Ollama isn't running at ${OLLAMA_HOST2}, and the Docker daemon isn't available to auto-start it.
|
|
2642
2861
|
${manualHint}`
|
|
2643
2862
|
);
|
|
2644
2863
|
}
|
|
@@ -2675,7 +2894,7 @@ ${manualHint}`);
|
|
|
2675
2894
|
}
|
|
2676
2895
|
async function ollamaPull(model) {
|
|
2677
2896
|
try {
|
|
2678
|
-
const show = await fetch(`${
|
|
2897
|
+
const show = await fetch(`${OLLAMA_HOST2}/api/show`, {
|
|
2679
2898
|
method: "POST",
|
|
2680
2899
|
headers: { "Content-Type": "application/json" },
|
|
2681
2900
|
body: JSON.stringify({ name: model })
|
|
@@ -2684,7 +2903,7 @@ async function ollamaPull(model) {
|
|
|
2684
2903
|
} catch {
|
|
2685
2904
|
}
|
|
2686
2905
|
console.log(`Pulling embedding model "${model}" into Ollama (first run only)\u2026`);
|
|
2687
|
-
const res = await fetch(`${
|
|
2906
|
+
const res = await fetch(`${OLLAMA_HOST2}/api/pull`, {
|
|
2688
2907
|
method: "POST",
|
|
2689
2908
|
headers: { "Content-Type": "application/json" },
|
|
2690
2909
|
body: JSON.stringify({ name: model, stream: false })
|
|
@@ -2695,7 +2914,7 @@ async function ollamaPull(model) {
|
|
|
2695
2914
|
async function embedOllama(model, inputs) {
|
|
2696
2915
|
const out = [];
|
|
2697
2916
|
for (const text of inputs) {
|
|
2698
|
-
const res = await fetch(`${
|
|
2917
|
+
const res = await fetch(`${OLLAMA_HOST2}/api/embeddings`, {
|
|
2699
2918
|
method: "POST",
|
|
2700
2919
|
headers: { "Content-Type": "application/json" },
|
|
2701
2920
|
body: JSON.stringify({ model, prompt: text })
|
|
@@ -2753,17 +2972,17 @@ async function embedOpenAI(model, inputs) {
|
|
|
2753
2972
|
}
|
|
2754
2973
|
async function loadJdf2(filePath) {
|
|
2755
2974
|
if (filePath.toLowerCase().endsWith(".jdfx")) {
|
|
2756
|
-
const zip = await JSZip.loadAsync(
|
|
2975
|
+
const zip = await JSZip.loadAsync(fs8.readFileSync(filePath));
|
|
2757
2976
|
const docFile = zip.file(JDFX_DOCUMENT_PATH);
|
|
2758
2977
|
if (!docFile) throw new Error(`Bundle missing ${JDFX_DOCUMENT_PATH}`);
|
|
2759
2978
|
return JSON.parse(await docFile.async("string"));
|
|
2760
2979
|
}
|
|
2761
|
-
return JSON.parse(
|
|
2980
|
+
return JSON.parse(fs8.readFileSync(filePath, "utf-8"));
|
|
2762
2981
|
}
|
|
2763
2982
|
function loadCache(cachePath) {
|
|
2764
2983
|
try {
|
|
2765
|
-
if (!
|
|
2766
|
-
return JSON.parse(
|
|
2984
|
+
if (!fs8.existsSync(cachePath)) return null;
|
|
2985
|
+
return JSON.parse(fs8.readFileSync(cachePath, "utf-8"));
|
|
2767
2986
|
} catch {
|
|
2768
2987
|
return null;
|
|
2769
2988
|
}
|
|
@@ -2774,15 +2993,15 @@ function batched(items, size) {
|
|
|
2774
2993
|
return out;
|
|
2775
2994
|
}
|
|
2776
2995
|
async function embedFile(inputPath, opts = {}) {
|
|
2777
|
-
const input =
|
|
2778
|
-
if (!
|
|
2996
|
+
const input = path8.resolve(inputPath);
|
|
2997
|
+
if (!fs8.existsSync(input)) throw new Error(`File not found: ${input}`);
|
|
2779
2998
|
const provider = opts.provider ?? "ollama";
|
|
2780
2999
|
const model = opts.model ?? DEFAULT_MODEL[provider];
|
|
2781
3000
|
const strategy = opts.strategy ?? "section";
|
|
2782
3001
|
const doc = await loadJdf2(input);
|
|
2783
3002
|
const chunks = chunkDocument(doc, { strategy, maxTokens: opts.maxTokens, transcriptWindowSec: opts.transcriptWindowSec });
|
|
2784
|
-
const output = opts.output ?
|
|
2785
|
-
const cachePath = opts.cache ?
|
|
3003
|
+
const output = opts.output ? path8.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".embeddings.json");
|
|
3004
|
+
const cachePath = opts.cache ? path8.resolve(opts.cache) : output;
|
|
2786
3005
|
console.log(`Embedding: ${input}`);
|
|
2787
3006
|
console.log(`Provider: ${provider} / ${model}${provider === "ollama" ? " (local \u2014 no data leaves this machine)" : " (remote API)"}`);
|
|
2788
3007
|
console.log(`Strategy: ${strategy} \u2192 ${chunks.length} chunks`);
|
|
@@ -2821,7 +3040,7 @@ async function embedFile(inputPath, opts = {}) {
|
|
|
2821
3040
|
chunker: `jdf-${strategy}-v1`,
|
|
2822
3041
|
vectors
|
|
2823
3042
|
};
|
|
2824
|
-
|
|
3043
|
+
fs8.writeFileSync(output, JSON.stringify(sidecar));
|
|
2825
3044
|
console.log(`
|
|
2826
3045
|
Done! ${Object.keys(vectors).length} vectors (${dims}-dim) \u2192 ${output}`);
|
|
2827
3046
|
return sidecar;
|
|
@@ -2865,9 +3084,9 @@ function parseChapters(text) {
|
|
|
2865
3084
|
return { t: toSec(m[1]), title: m[2].trim() };
|
|
2866
3085
|
});
|
|
2867
3086
|
}
|
|
2868
|
-
async function
|
|
3087
|
+
async function loadDoc2(file) {
|
|
2869
3088
|
if (file.toLowerCase().endsWith(".jdfx")) {
|
|
2870
|
-
const zip = await JSZip.loadAsync(
|
|
3089
|
+
const zip = await JSZip.loadAsync(fs8.readFileSync(file));
|
|
2871
3090
|
const f = zip.file(JDFX_DOCUMENT_PATH);
|
|
2872
3091
|
if (!f) throw new Error(`Bundle missing ${JDFX_DOCUMENT_PATH}`);
|
|
2873
3092
|
const doc = JSON.parse(await f.async("string"));
|
|
@@ -2883,7 +3102,7 @@ async function loadDoc(file) {
|
|
|
2883
3102
|
}
|
|
2884
3103
|
return { doc, bundle: true, zip };
|
|
2885
3104
|
}
|
|
2886
|
-
return { doc: JSON.parse(
|
|
3105
|
+
return { doc: JSON.parse(fs8.readFileSync(file, "utf-8")), bundle: false };
|
|
2887
3106
|
}
|
|
2888
3107
|
function findVideos(doc) {
|
|
2889
3108
|
const out = [];
|
|
@@ -2897,27 +3116,27 @@ function findVideos(doc) {
|
|
|
2897
3116
|
return out;
|
|
2898
3117
|
}
|
|
2899
3118
|
async function clipToTempFile(doc, el, docDir) {
|
|
2900
|
-
const tmp =
|
|
3119
|
+
const tmp = path8.join(fs8.mkdtempSync(path8.join(__require("os").tmpdir(), "jdf-transcribe-")), "clip.mp4");
|
|
2901
3120
|
const res = el.resource ? doc.resources?.videos?.[el.resource] ?? doc.resources?.images?.[el.resource] : void 0;
|
|
2902
3121
|
if (res?.data) {
|
|
2903
|
-
|
|
3122
|
+
fs8.writeFileSync(tmp, Buffer.from(res.data.replace(/^data:[^,]*,/, ""), "base64"));
|
|
2904
3123
|
return tmp;
|
|
2905
3124
|
}
|
|
2906
|
-
if (res?.path) return
|
|
3125
|
+
if (res?.path) return path8.resolve(docDir, res.path);
|
|
2907
3126
|
const src = el.src;
|
|
2908
3127
|
if (!src) return null;
|
|
2909
3128
|
if (src.startsWith("data:")) {
|
|
2910
|
-
|
|
3129
|
+
fs8.writeFileSync(tmp, Buffer.from(src.replace(/^data:[^,]*,/, ""), "base64"));
|
|
2911
3130
|
return tmp;
|
|
2912
3131
|
}
|
|
2913
3132
|
if (/^https?:\/\//i.test(src)) {
|
|
2914
3133
|
const r = await fetch(src);
|
|
2915
3134
|
if (!r.ok) throw new Error(`download failed ${r.status}: ${src}`);
|
|
2916
|
-
|
|
3135
|
+
fs8.writeFileSync(tmp, Buffer.from(await r.arrayBuffer()));
|
|
2917
3136
|
return tmp;
|
|
2918
3137
|
}
|
|
2919
|
-
const local =
|
|
2920
|
-
return
|
|
3138
|
+
const local = path8.resolve(docDir, src);
|
|
3139
|
+
return fs8.existsSync(local) ? local : null;
|
|
2921
3140
|
}
|
|
2922
3141
|
function whisperCli(clip, model, language, prompt2) {
|
|
2923
3142
|
const ffmpeg = spawnSync("ffmpeg", ["-version"]);
|
|
@@ -2932,7 +3151,7 @@ function whisperCli(clip, model, language, prompt2) {
|
|
|
2932
3151
|
const run = spawnSync("whisper-cli", args, { encoding: "utf-8" });
|
|
2933
3152
|
if (run.error) throw new Error("whisper-cli not found \u2014 install whisper.cpp (brew install whisper-cpp) or use --from / --provider openai");
|
|
2934
3153
|
if (run.status !== 0) throw new Error(`whisper-cli failed: ${run.stderr.slice(-400)}`);
|
|
2935
|
-
const j = JSON.parse(
|
|
3154
|
+
const j = JSON.parse(fs8.readFileSync(wav.replace(/\.wav$/, "") + ".json", "utf-8"));
|
|
2936
3155
|
const segs = j.transcription ?? j.segments ?? [];
|
|
2937
3156
|
const ms = (x) => typeof x === "number" ? x / 1e3 : toSec(String(x).replace(",", "."));
|
|
2938
3157
|
return segs.map((sg) => ({ t0: ms(sg.offsets?.from ?? sg.start), t1: ms(sg.offsets?.to ?? sg.end), text: String(sg.text ?? "").trim() })).filter((sg) => sg.text);
|
|
@@ -2942,7 +3161,7 @@ async function openaiTranscribe(clip, model, language, prompt2) {
|
|
|
2942
3161
|
if (!key) throw new Error("OPENAI_API_KEY is not set");
|
|
2943
3162
|
const base = process.env.OPENAI_BASE_URL || "https://api.openai.com/v1";
|
|
2944
3163
|
const form = new FormData();
|
|
2945
|
-
form.append("file", new Blob([
|
|
3164
|
+
form.append("file", new Blob([fs8.readFileSync(clip)]), path8.basename(clip));
|
|
2946
3165
|
form.append("model", model || "whisper-1");
|
|
2947
3166
|
form.append("response_format", "verbose_json");
|
|
2948
3167
|
form.append("timestamp_granularities[]", "segment");
|
|
@@ -2954,9 +3173,9 @@ async function openaiTranscribe(clip, model, language, prompt2) {
|
|
|
2954
3173
|
return (j.segments ?? []).map((sg) => ({ t0: Number(sg.start), t1: Number(sg.end), text: String(sg.text).trim() })).filter((sg) => sg.text);
|
|
2955
3174
|
}
|
|
2956
3175
|
async function transcribeFile(inputPath, opts = {}) {
|
|
2957
|
-
const input =
|
|
2958
|
-
if (!
|
|
2959
|
-
const { doc, bundle } = await
|
|
3176
|
+
const input = path8.resolve(inputPath);
|
|
3177
|
+
if (!fs8.existsSync(input)) throw new Error(`File not found: ${input}`);
|
|
3178
|
+
const { doc, bundle } = await loadDoc2(input);
|
|
2960
3179
|
const videos = findVideos(doc);
|
|
2961
3180
|
if (!videos.length) throw new Error("document has no video element");
|
|
2962
3181
|
let target = videos[0];
|
|
@@ -2972,40 +3191,40 @@ async function transcribeFile(inputPath, opts = {}) {
|
|
|
2972
3191
|
let segments;
|
|
2973
3192
|
let source;
|
|
2974
3193
|
if (opts.from) {
|
|
2975
|
-
segments = parseSubtitles(
|
|
2976
|
-
source = `${
|
|
3194
|
+
segments = parseSubtitles(fs8.readFileSync(path8.resolve(opts.from), "utf-8"), opts.from);
|
|
3195
|
+
source = `${path8.extname(opts.from).slice(1).toLowerCase() || "file"}-import`;
|
|
2977
3196
|
} else {
|
|
2978
3197
|
const provider = opts.provider ?? "whisper-cli";
|
|
2979
|
-
const clip = await clipToTempFile(doc, target.el,
|
|
3198
|
+
const clip = await clipToTempFile(doc, target.el, path8.dirname(input));
|
|
2980
3199
|
if (!clip) throw new Error("could not locate the clip bytes (no bundled asset, data URL, local path or http URL) \u2014 use --from to import subtitles instead");
|
|
2981
3200
|
segments = provider === "openai" ? await openaiTranscribe(clip, opts.model, opts.language, opts.prompt) : whisperCli(clip, opts.model, opts.language, opts.prompt);
|
|
2982
|
-
source = provider === "openai" ? `openai:${opts.model || "whisper-1"}` : `whisper-cli${opts.model ? ":" +
|
|
3201
|
+
source = provider === "openai" ? `openai:${opts.model || "whisper-1"}` : `whisper-cli${opts.model ? ":" + path8.basename(opts.model) : ""}`;
|
|
2983
3202
|
}
|
|
2984
3203
|
segments.sort((a, b) => a.t0 - b.t0);
|
|
2985
3204
|
const transcript = { ...opts.language ? { language: opts.language } : {}, source, created: (/* @__PURE__ */ new Date()).toISOString(), segments };
|
|
2986
3205
|
target.el.transcript = transcript;
|
|
2987
|
-
if (opts.chapters) target.el.chapters = parseChapters(
|
|
3206
|
+
if (opts.chapters) target.el.chapters = parseChapters(fs8.readFileSync(path8.resolve(opts.chapters), "utf-8"));
|
|
2988
3207
|
if (!target.el.id) target.el.id = `video-${target.index + 1}`;
|
|
2989
|
-
const output = opts.output ?
|
|
3208
|
+
const output = opts.output ? path8.resolve(opts.output) : input;
|
|
2990
3209
|
if (output.toLowerCase().endsWith(".jdfx") || bundle && !opts.output) {
|
|
2991
3210
|
const { bytes } = await packJdfx(doc);
|
|
2992
|
-
|
|
3211
|
+
fs8.writeFileSync(output, bytes);
|
|
2993
3212
|
} else {
|
|
2994
|
-
|
|
3213
|
+
fs8.writeFileSync(output, JSON.stringify(doc, null, 2));
|
|
2995
3214
|
}
|
|
2996
3215
|
const dur = segments.length ? segments[segments.length - 1].t1 : 0;
|
|
2997
|
-
console.log(`Transcribed: ${
|
|
3216
|
+
console.log(`Transcribed: ${path8.basename(input)} \u2192 element "${target.el.id}" (${segments.length} segments, ${Math.round(dur)} s, source ${source})`);
|
|
2998
3217
|
if (target.el.chapters) console.log(`Chapters: ${target.el.chapters.length}`);
|
|
2999
3218
|
console.log(`Output: ${output}
|
|
3000
|
-
Next: jdf chunk ${
|
|
3219
|
+
Next: jdf chunk ${path8.basename(output)} # transcript \u2192 time-windowed chunks with media.t0/t1`);
|
|
3001
3220
|
return transcript;
|
|
3002
3221
|
}
|
|
3003
3222
|
var CONFIG_NAME = "jdf.rag.json";
|
|
3004
3223
|
var OUT_DIR = ".jdf-rag";
|
|
3005
3224
|
function walk(dir, acc = []) {
|
|
3006
|
-
for (const ent of
|
|
3225
|
+
for (const ent of fs8.readdirSync(dir, { withFileTypes: true })) {
|
|
3007
3226
|
if (ent.name === "node_modules" || ent.name === OUT_DIR || ent.name.startsWith(".")) continue;
|
|
3008
|
-
const p =
|
|
3227
|
+
const p = path8.join(dir, ent.name);
|
|
3009
3228
|
if (ent.isDirectory()) walk(p, acc);
|
|
3010
3229
|
else if (/\.(jdf|jdfx)$/i.test(ent.name)) acc.push(p);
|
|
3011
3230
|
}
|
|
@@ -3013,10 +3232,10 @@ function walk(dir, acc = []) {
|
|
|
3013
3232
|
}
|
|
3014
3233
|
async function readDoc(file) {
|
|
3015
3234
|
if (file.toLowerCase().endsWith(".jdfx")) {
|
|
3016
|
-
const zip = await JSZip.loadAsync(
|
|
3235
|
+
const zip = await JSZip.loadAsync(fs8.readFileSync(file));
|
|
3017
3236
|
return JSON.parse(await zip.file(JDFX_DOCUMENT_PATH).async("string"));
|
|
3018
3237
|
}
|
|
3019
|
-
return JSON.parse(
|
|
3238
|
+
return JSON.parse(fs8.readFileSync(file, "utf-8"));
|
|
3020
3239
|
}
|
|
3021
3240
|
function videosIn(doc) {
|
|
3022
3241
|
const out = [];
|
|
@@ -3030,29 +3249,44 @@ function videosIn(doc) {
|
|
|
3030
3249
|
return out;
|
|
3031
3250
|
}
|
|
3032
3251
|
async function ragFolder(dirPath, cli = {}) {
|
|
3033
|
-
const dir =
|
|
3034
|
-
if (!
|
|
3035
|
-
const cfgPath =
|
|
3036
|
-
const cfg =
|
|
3252
|
+
const dir = path8.resolve(dirPath);
|
|
3253
|
+
if (!fs8.existsSync(dir) || !fs8.statSync(dir).isDirectory()) throw new Error(`Not a directory: ${dir}`);
|
|
3254
|
+
const cfgPath = path8.join(dir, CONFIG_NAME);
|
|
3255
|
+
const cfg = fs8.existsSync(cfgPath) ? JSON.parse(fs8.readFileSync(cfgPath, "utf-8")) : {};
|
|
3037
3256
|
const opts = { ...cfg, ...Object.fromEntries(Object.entries(cli).filter(([, v]) => v !== void 0)) };
|
|
3038
3257
|
const provider = opts.provider ?? "ollama";
|
|
3039
3258
|
const transcribe = opts.transcribe ?? "none";
|
|
3040
|
-
const outDir =
|
|
3259
|
+
const outDir = path8.resolve(opts.out ?? path8.join(dir, OUT_DIR));
|
|
3260
|
+
const ocr = opts.ocr ?? "none";
|
|
3261
|
+
const caption = opts.caption ?? "none";
|
|
3041
3262
|
const files = walk(dir);
|
|
3042
3263
|
console.log(`jdf rag: ${dir}
|
|
3043
|
-
files: ${files.length} (.jdf/.jdfx)${
|
|
3264
|
+
files: ${files.length} (.jdf/.jdfx)${fs8.existsSync(cfgPath) ? `
|
|
3044
3265
|
config: ${CONFIG_NAME}` : ""}
|
|
3045
3266
|
embeddings: ${opts.noEmbed ? "skipped (--no-embed)" : `${provider}${opts.model ? " / " + opts.model : ""}`}
|
|
3046
|
-
transcribe: ${transcribe}${opts.dryRun ? "\n DRY RUN \u2014 nothing written" : ""}
|
|
3267
|
+
transcribe: ${transcribe} ocr: ${ocr} caption: ${caption}${opts.dryRun ? "\n DRY RUN \u2014 nothing written" : ""}
|
|
3047
3268
|
`);
|
|
3048
3269
|
if (!files.length) {
|
|
3049
3270
|
console.log("Nothing to do.");
|
|
3050
3271
|
return;
|
|
3051
3272
|
}
|
|
3052
|
-
const manifest = {
|
|
3273
|
+
const manifest = {
|
|
3274
|
+
dir,
|
|
3275
|
+
created: (/* @__PURE__ */ new Date()).toISOString(),
|
|
3276
|
+
provider: opts.noEmbed ? null : provider,
|
|
3277
|
+
model: opts.model ?? null,
|
|
3278
|
+
strategy: opts.strategy ?? "section",
|
|
3279
|
+
transcribe,
|
|
3280
|
+
ocr,
|
|
3281
|
+
caption,
|
|
3282
|
+
files: [],
|
|
3283
|
+
totals: { files: files.length, chunks: 0, videoChunks: 0, videos: 0, transcribed: 0, untranscribed: 0, images: 0, described: 0, imagesWithoutText: 0 },
|
|
3284
|
+
/** Every media element that retrieval would still skip, by file — the thing to fix before shipping an index. */
|
|
3285
|
+
mediaWithoutText: []
|
|
3286
|
+
};
|
|
3053
3287
|
const indexLines = [];
|
|
3054
3288
|
for (const file of files) {
|
|
3055
|
-
const rel =
|
|
3289
|
+
const rel = path8.relative(dir, file);
|
|
3056
3290
|
const doc = await readDoc(file);
|
|
3057
3291
|
const vids = videosIn(doc);
|
|
3058
3292
|
let transcribedHere = 0;
|
|
@@ -3076,20 +3310,36 @@ async function ragFolder(dirPath, cli = {}) {
|
|
|
3076
3310
|
}
|
|
3077
3311
|
manifest.totals.videos += vids.length;
|
|
3078
3312
|
manifest.totals.transcribed += transcribedHere;
|
|
3313
|
+
const covBefore = mediaCoverage(doc);
|
|
3314
|
+
let describedHere = 0;
|
|
3315
|
+
if (covBefore.images.missing.length && (ocr !== "none" || caption !== "none") && !opts.dryRun) {
|
|
3316
|
+
try {
|
|
3317
|
+
await describeFile(file, { ocr, caption, captionModel: opts.captionModel, quiet: true });
|
|
3318
|
+
describedHere = covBefore.images.missing.length;
|
|
3319
|
+
} catch (e) {
|
|
3320
|
+
console.warn(` ! ${rel}: describe failed: ${e.message}`);
|
|
3321
|
+
}
|
|
3322
|
+
}
|
|
3323
|
+
const covAfter = transcribedHere || describedHere ? mediaCoverage(await readDoc(file)) : covBefore;
|
|
3324
|
+
manifest.totals.images += covAfter.images.total;
|
|
3325
|
+
manifest.totals.described += describedHere;
|
|
3326
|
+
manifest.totals.imagesWithoutText += covAfter.images.missing.length;
|
|
3327
|
+
for (const m of covAfter.images.missing) manifest.mediaWithoutText.push({ file: rel, type: "image", ...m });
|
|
3328
|
+
for (const m of covAfter.videos.missing) manifest.mediaWithoutText.push({ file: rel, type: "video", ...m });
|
|
3079
3329
|
if (opts.dryRun) {
|
|
3080
3330
|
manifest.files.push({ file: rel, videos: vids.length, wouldTranscribe: transcribedHere });
|
|
3081
3331
|
continue;
|
|
3082
3332
|
}
|
|
3083
3333
|
const chunkOpts = { strategy: opts.strategy, maxTokens: opts.maxTokens, transcriptWindowSec: opts.transcriptWindowSec };
|
|
3084
|
-
const chunkOut =
|
|
3085
|
-
|
|
3334
|
+
const chunkOut = path8.join(outDir, "chunks", rel.replace(/\.(jdf|jdfx)$/i, ".chunks.jsonl"));
|
|
3335
|
+
fs8.mkdirSync(path8.dirname(chunkOut), { recursive: true });
|
|
3086
3336
|
let chunks;
|
|
3087
3337
|
if (opts.noEmbed) {
|
|
3088
3338
|
chunks = await chunkFile(file, { ...chunkOpts, format: "jsonl", output: chunkOut });
|
|
3089
3339
|
} else {
|
|
3090
3340
|
const side = await embedFile(file, { ...chunkOpts, provider, model: opts.model, incremental: true });
|
|
3091
3341
|
chunks = await chunkFile(file, { ...chunkOpts, format: "jsonl", output: chunkOut });
|
|
3092
|
-
manifest.files.push({ file: rel, chunks: chunks.length, vectors: Object.keys(side.vectors).length, sidecar:
|
|
3342
|
+
manifest.files.push({ file: rel, chunks: chunks.length, vectors: Object.keys(side.vectors).length, sidecar: path8.relative(dir, file.replace(/\.(jdf|jdfx)$/i, ".embeddings.json")), videos: vids.length, transcribed: transcribedHere });
|
|
3093
3343
|
}
|
|
3094
3344
|
if (opts.noEmbed) manifest.files.push({ file: rel, chunks: chunks.length, videos: vids.length, transcribed: transcribedHere });
|
|
3095
3345
|
for (const c of chunks) {
|
|
@@ -3099,15 +3349,27 @@ async function ragFolder(dirPath, cli = {}) {
|
|
|
3099
3349
|
}
|
|
3100
3350
|
}
|
|
3101
3351
|
if (!opts.dryRun) {
|
|
3102
|
-
|
|
3103
|
-
|
|
3104
|
-
|
|
3352
|
+
fs8.mkdirSync(outDir, { recursive: true });
|
|
3353
|
+
fs8.writeFileSync(path8.join(outDir, "index.jsonl"), indexLines.join("\n") + (indexLines.length ? "\n" : ""));
|
|
3354
|
+
fs8.writeFileSync(path8.join(outDir, "manifest.json"), JSON.stringify(manifest, null, 2) + "\n");
|
|
3105
3355
|
}
|
|
3106
3356
|
const t = manifest.totals;
|
|
3107
3357
|
console.log(`
|
|
3108
|
-
Done. ${t.files} files \u2192 ${t.chunks} chunks (${t.videoChunks} from video transcripts)
|
|
3109
|
-
|
|
3110
|
-
|
|
3358
|
+
Done. ${t.files} files \u2192 ${t.chunks} chunks (${t.videoChunks} from video transcripts).`);
|
|
3359
|
+
console.log(`Media coverage: videos ${t.videos - t.untranscribed}/${t.videos} with transcript (transcribed now ${t.transcribed}), images ${t.images - t.imagesWithoutText}/${t.images} with caption/OCR (described now ${t.described}).`);
|
|
3360
|
+
if (manifest.mediaWithoutText.length) {
|
|
3361
|
+
console.log(`
|
|
3362
|
+
! ${manifest.mediaWithoutText.length} media element(s) still have NO text \u2014 retrieval will skip them:`);
|
|
3363
|
+
for (const m of manifest.mediaWithoutText.slice(0, 12)) console.log(` ${m.file} \xB7 ${m.type} ${m.id ?? ""} (page ${m.page})${m.title ? ` "${m.title}"` : m.alt ? ` alt="${m.alt}"` : ""}`);
|
|
3364
|
+
if (manifest.mediaWithoutText.length > 12) console.log(` \u2026 ${manifest.mediaWithoutText.length - 12} more in manifest.json`);
|
|
3365
|
+
console.log(` fix: jdf rag <dir> --transcribe whisper-cli|openai --ocr tesseract --caption ollama (or jdf transcribe / jdf describe per file)`);
|
|
3366
|
+
if (opts.strict) {
|
|
3367
|
+
console.error(`--strict: failing because media without text remains.`);
|
|
3368
|
+
process.exitCode = 1;
|
|
3369
|
+
}
|
|
3370
|
+
}
|
|
3371
|
+
if (!opts.dryRun) console.log(`Index: ${path8.join(outDir, "index.jsonl")}
|
|
3372
|
+
Report: ${path8.join(outDir, "manifest.json")}${opts.noEmbed ? "" : `
|
|
3111
3373
|
Vectors: one <file>.embeddings.json next to each document (incremental \u2014 re-run any time)`}`);
|
|
3112
3374
|
}
|
|
3113
3375
|
|
|
@@ -3123,8 +3385,11 @@ The CLI exists for these workflows:
|
|
|
3123
3385
|
\u2022 JDF \u2192 vectors embed those chunks, incrementally, for a vector store.
|
|
3124
3386
|
\u2022 video \u2192 text attach a time-stamped transcript to a video element so
|
|
3125
3387
|
RAG retrieves "video at 02:13", not just "a video".
|
|
3388
|
+
\u2022 image \u2192 text OCR + a vision caption for every image so charts and
|
|
3389
|
+
scanned pages are retrievable, not skipped.
|
|
3126
3390
|
\u2022 folder \u2192 index one command over a directory of .jdf/.jdfx: transcribe,
|
|
3127
|
-
chunk, embed incrementally, write .jdf-rag/index.jsonl.
|
|
3391
|
+
describe, chunk, embed incrementally, write .jdf-rag/index.jsonl.
|
|
3392
|
+
Reports media coverage; --strict fails when anything has no text.
|
|
3128
3393
|
|
|
3129
3394
|
Usage:
|
|
3130
3395
|
jdf validate <file.jdf>
|
|
@@ -3132,7 +3397,8 @@ Usage:
|
|
|
3132
3397
|
jdf chunk <file.{jdf,jdfx}> [--strategy section|element|fixed] [--format jsonl|json|inline] [--max-tokens N] [-o out]
|
|
3133
3398
|
jdf embed <file.{jdf,jdfx}> [--provider ollama|openai] [--model NAME] [--strategy \u2026] [--incremental] [-o out]
|
|
3134
3399
|
jdf transcribe <file.{jdf,jdfx}> [--from subs.srt|.vtt|.json] [--provider whisper-cli|openai] [--element ID] [--chapters FILE] [-o out]
|
|
3135
|
-
jdf
|
|
3400
|
+
jdf describe <file.{jdf,jdfx}> [--ocr tesseract|openai|none] [--caption ollama|openai|none] [--caption-model M] [--ocr-language eng] [--element ID] [--force] [-o out]
|
|
3401
|
+
jdf rag <dir> [--provider ollama|openai] [--model NAME] [--transcribe none|whisper-cli|openai] [--ocr none|tesseract|openai] [--caption none|ollama|openai] [--strict] [--no-embed] [--dry-run] [--out DIR]
|
|
3136
3402
|
jdf --help
|
|
3137
3403
|
|
|
3138
3404
|
Commands:
|
|
@@ -3141,7 +3407,8 @@ Commands:
|
|
|
3141
3407
|
chunk Split a JDF document into retrieval-ready chunks (offline, deterministic)
|
|
3142
3408
|
embed Compute embeddings for the chunks (local via Ollama by default)
|
|
3143
3409
|
transcribe Store time-stamped text on a video element (import SRT/VTT/JSON, or run Whisper)
|
|
3144
|
-
|
|
3410
|
+
describe Give images text: OCR blocks (tesseract.js, local) + a caption (Ollama vision model, local)
|
|
3411
|
+
rag Make a whole folder retrieval-ready (finds .jdf/.jdfx, transcribes, describes, chunks, embeds, indexes)
|
|
3145
3412
|
|
|
3146
3413
|
Flags:
|
|
3147
3414
|
-o, --output <path> Explicit output path
|
|
@@ -3168,6 +3435,12 @@ Flags:
|
|
|
3168
3435
|
--prompt <text> transcribe: Whisper vocabulary hint (names, acronyms) \u2014 not an instruction
|
|
3169
3436
|
--window <sec> chunk/embed/rag: transcript window per video chunk (default 45)
|
|
3170
3437
|
--transcribe <p> rag: none (default) | whisper-cli | openai \u2014 for videos that have no transcript yet
|
|
3438
|
+
--ocr <p> describe/rag/convert: tesseract (local WASM) | openai | none
|
|
3439
|
+
--caption <p> describe/rag: ollama (local vision model, default moondream) | openai | none
|
|
3440
|
+
--caption-model describe/rag: vision model name (ollama: qwen2.5vl:3b, llava\u2026; openai: gpt-4o-mini\u2026)
|
|
3441
|
+
--ocr-language describe: tesseract language(s), e.g. eng, tur, eng+tur (default eng)
|
|
3442
|
+
--force describe: redo images that already have text
|
|
3443
|
+
--strict rag: exit 1 if any image/video is still without text after the run
|
|
3171
3444
|
--no-embed rag: chunk + index only
|
|
3172
3445
|
--dry-run rag: list what would happen, write nothing
|
|
3173
3446
|
--out <dir> rag: index folder (default <dir>/.jdf-rag)
|
|
@@ -3187,9 +3460,11 @@ Examples:
|
|
|
3187
3460
|
jdf embed report.jdf --provider openai --incremental
|
|
3188
3461
|
jdf transcribe talk.jdfx --from talk.srt --chapters chapters.txt # then: jdf chunk talk.jdfx
|
|
3189
3462
|
jdf transcribe talk.jdfx --provider openai --language en --prompt "JDF, jdfx, Ollama"
|
|
3190
|
-
jdf
|
|
3463
|
+
jdf describe report.jdfx # OCR (tesseract) + caption (Ollama qwen2.5vl), local
|
|
3464
|
+
jdf convert scan.pdf --ocr tesseract # scanned pages get OCR text instead of silence
|
|
3465
|
+
jdf rag ./knowledge-base --transcribe openai --ocr tesseract --caption ollama --strict
|
|
3191
3466
|
`;
|
|
3192
|
-
var BOOLEAN_FLAGS = /* @__PURE__ */ new Set(["help", "h", "json", "verbose", "skip-validate", "incremental", "no-auto-start", "drop-invisible-text", "no-embed", "dry-run"]);
|
|
3467
|
+
var BOOLEAN_FLAGS = /* @__PURE__ */ new Set(["help", "h", "json", "verbose", "skip-validate", "incremental", "no-auto-start", "drop-invisible-text", "no-embed", "dry-run", "force", "strict"]);
|
|
3193
3468
|
function parseArgs(argv) {
|
|
3194
3469
|
const positional = [];
|
|
3195
3470
|
const flags = {};
|
|
@@ -3272,7 +3547,8 @@ async function main() {
|
|
|
3272
3547
|
await importPdf(input, output, {
|
|
3273
3548
|
forceJson,
|
|
3274
3549
|
password: typeof flags.password === "string" ? flags.password : void 0,
|
|
3275
|
-
dropInvisibleText: flags["drop-invisible-text"] === true
|
|
3550
|
+
dropInvisibleText: flags["drop-invisible-text"] === true,
|
|
3551
|
+
ocr: typeof flags.ocr === "string" ? flags.ocr : void 0
|
|
3276
3552
|
});
|
|
3277
3553
|
process.exit(0);
|
|
3278
3554
|
} else if (lower.endsWith(".json")) {
|
|
@@ -3316,6 +3592,23 @@ async function main() {
|
|
|
3316
3592
|
});
|
|
3317
3593
|
process.exit(0);
|
|
3318
3594
|
}
|
|
3595
|
+
case "describe": {
|
|
3596
|
+
const input = positional[0];
|
|
3597
|
+
if (!input) {
|
|
3598
|
+
console.error("Usage: jdf describe <file.{jdf,jdfx}> [--ocr tesseract|openai|none] [--caption ollama|openai|none] [--caption-model M] [--ocr-language eng] [--element id|n] [--force] [-o out]");
|
|
3599
|
+
process.exit(1);
|
|
3600
|
+
}
|
|
3601
|
+
await describeFile(input, {
|
|
3602
|
+
ocr: typeof flags.ocr === "string" ? flags.ocr : void 0,
|
|
3603
|
+
caption: typeof flags.caption === "string" ? flags.caption : void 0,
|
|
3604
|
+
captionModel: typeof flags["caption-model"] === "string" ? flags["caption-model"] : void 0,
|
|
3605
|
+
ocrLanguage: typeof flags["ocr-language"] === "string" ? flags["ocr-language"] : void 0,
|
|
3606
|
+
element: typeof flags.element === "string" ? flags.element : void 0,
|
|
3607
|
+
force: flags.force === true,
|
|
3608
|
+
output: typeof flags.output === "string" ? flags.output : void 0
|
|
3609
|
+
});
|
|
3610
|
+
process.exit(process.exitCode ?? 0);
|
|
3611
|
+
}
|
|
3319
3612
|
case "rag": {
|
|
3320
3613
|
const input = positional[0];
|
|
3321
3614
|
if (!input) {
|
|
@@ -3332,11 +3625,15 @@ async function main() {
|
|
|
3332
3625
|
transcribeModel: typeof flags["transcribe-model"] === "string" ? flags["transcribe-model"] : void 0,
|
|
3333
3626
|
language: typeof flags.language === "string" ? flags.language : void 0,
|
|
3334
3627
|
prompt: typeof flags.prompt === "string" ? flags.prompt : void 0,
|
|
3628
|
+
ocr: typeof flags.ocr === "string" ? flags.ocr : void 0,
|
|
3629
|
+
caption: typeof flags.caption === "string" ? flags.caption : void 0,
|
|
3630
|
+
captionModel: typeof flags["caption-model"] === "string" ? flags["caption-model"] : void 0,
|
|
3631
|
+
strict: flags.strict === true,
|
|
3335
3632
|
noEmbed: flags["no-embed"] === true,
|
|
3336
3633
|
dryRun: flags["dry-run"] === true,
|
|
3337
3634
|
out: typeof flags.out === "string" ? flags.out : void 0
|
|
3338
3635
|
});
|
|
3339
|
-
process.exit(0);
|
|
3636
|
+
process.exit(process.exitCode ?? 0);
|
|
3340
3637
|
}
|
|
3341
3638
|
case "embed": {
|
|
3342
3639
|
const input = positional[0];
|
package/dist/jdf-schema.json
CHANGED
|
@@ -343,7 +343,17 @@
|
|
|
343
343
|
"anyOf": [{ "required": ["resource"] }, { "required": ["src"] }],
|
|
344
344
|
"properties": {
|
|
345
345
|
"type": { "const": "image" },
|
|
346
|
+
"id": { "type": "string" },
|
|
346
347
|
"resource": { "type": "string" }, "src": { "type": "string" }, "alt": { "type": "string" },
|
|
348
|
+
"caption": { "type": "string" }, "captionSource": { "type": "string" },
|
|
349
|
+
"ocr": {
|
|
350
|
+
"type": "object",
|
|
351
|
+
"required": ["blocks"],
|
|
352
|
+
"properties": {
|
|
353
|
+
"language": { "type": "string" }, "source": { "type": "string" }, "created": { "type": "string" },
|
|
354
|
+
"blocks": { "type": "array", "items": { "type": "object", "required": ["text"], "properties": { "text": { "type": "string" }, "confidence": { "type": "number" }, "bbox": { "type": "object", "properties": { "x": { "type": "number" }, "y": { "type": "number" }, "w": { "type": "number" }, "h": { "type": "number" } } } } } }
|
|
355
|
+
}
|
|
356
|
+
},
|
|
347
357
|
"position": { "$ref": "#/definitions/Position" },
|
|
348
358
|
"width": { "type": "number" }, "height": { "type": "number" },
|
|
349
359
|
"fit": { "type": "string", "enum": ["contain","cover","fill","none"] },
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@uurtech/jdf-cli",
|
|
3
|
-
"version": "0.2.
|
|
3
|
+
"version": "0.2.2",
|
|
4
4
|
"description": "Command-line tool for the JDF (JSON Document Format) — validate and convert documents.",
|
|
5
5
|
"license": "MIT",
|
|
6
6
|
"author": "Ugur Kazdal",
|
|
@@ -52,7 +52,8 @@
|
|
|
52
52
|
"ajv": "^8.17.1",
|
|
53
53
|
"ajv-formats": "^3.0.1",
|
|
54
54
|
"jszip": "^3.10.1",
|
|
55
|
-
"pdfjs-dist": "^4.10.38"
|
|
55
|
+
"pdfjs-dist": "^4.10.38",
|
|
56
|
+
"tesseract.js": "^7.0.0"
|
|
56
57
|
},
|
|
57
58
|
"devDependencies": {
|
|
58
59
|
"@jdf/core": "workspace:*",
|