@uurtech/jdf-cli 0.2.0 → 0.2.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.js +512 -124
- package/dist/jdf-schema.json +10 -0
- package/package.json +3 -2
package/dist/index.js
CHANGED
|
@@ -1,12 +1,13 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
|
-
import
|
|
3
|
-
import
|
|
2
|
+
import fs8 from 'fs';
|
|
3
|
+
import path8 from 'path';
|
|
4
4
|
import { fileURLToPath } from 'url';
|
|
5
5
|
import Ajv from 'ajv';
|
|
6
6
|
import addFormats from 'ajv-formats';
|
|
7
7
|
import JSZip from 'jszip';
|
|
8
8
|
import crypto, { createHash } from 'crypto';
|
|
9
9
|
import { readFile } from 'fs/promises';
|
|
10
|
+
import os from 'os';
|
|
10
11
|
import { execFileSync, spawnSync } from 'child_process';
|
|
11
12
|
|
|
12
13
|
var __require = /* @__PURE__ */ ((x) => typeof require !== "undefined" ? require : typeof Proxy !== "undefined" ? new Proxy(x, {
|
|
@@ -23,17 +24,17 @@ var JDFX_MANIFEST_PATH = "manifest.json";
|
|
|
23
24
|
var JDFX_ASSET_DIR = "assets";
|
|
24
25
|
|
|
25
26
|
// src/commands/validate.ts
|
|
26
|
-
var __dirname$1 =
|
|
27
|
+
var __dirname$1 = path8.dirname(fileURLToPath(import.meta.url));
|
|
27
28
|
function resolveSchemaPath() {
|
|
28
|
-
const bundled =
|
|
29
|
-
if (
|
|
30
|
-
const dev =
|
|
29
|
+
const bundled = path8.resolve(__dirname$1, "jdf-schema.json");
|
|
30
|
+
if (fs8.existsSync(bundled)) return bundled;
|
|
31
|
+
const dev = path8.resolve(__dirname$1, "../../../../spec/jdf-schema.json");
|
|
31
32
|
return dev;
|
|
32
33
|
}
|
|
33
34
|
var SCHEMA_PATH = resolveSchemaPath();
|
|
34
35
|
async function loadDocument(filePath) {
|
|
35
36
|
if (filePath.toLowerCase().endsWith(".jdfx")) {
|
|
36
|
-
const zip = await JSZip.loadAsync(
|
|
37
|
+
const zip = await JSZip.loadAsync(fs8.readFileSync(filePath));
|
|
37
38
|
const docFile = zip.file(JDFX_DOCUMENT_PATH);
|
|
38
39
|
if (!docFile) {
|
|
39
40
|
console.error(`\u2717 Bundle missing ${JDFX_DOCUMENT_PATH}`);
|
|
@@ -58,11 +59,11 @@ async function loadDocument(filePath) {
|
|
|
58
59
|
}
|
|
59
60
|
return { doc, bundle: { manifest, assetCount } };
|
|
60
61
|
}
|
|
61
|
-
return { doc: JSON.parse(
|
|
62
|
+
return { doc: JSON.parse(fs8.readFileSync(filePath, "utf-8")) };
|
|
62
63
|
}
|
|
63
64
|
async function validate(file) {
|
|
64
|
-
const filePath =
|
|
65
|
-
if (!
|
|
65
|
+
const filePath = path8.resolve(file);
|
|
66
|
+
if (!fs8.existsSync(filePath)) {
|
|
66
67
|
console.error(`File not found: ${filePath}`);
|
|
67
68
|
return false;
|
|
68
69
|
}
|
|
@@ -75,11 +76,11 @@ async function validate(file) {
|
|
|
75
76
|
}
|
|
76
77
|
if (!loaded) return false;
|
|
77
78
|
const { doc, bundle } = loaded;
|
|
78
|
-
if (!
|
|
79
|
+
if (!fs8.existsSync(SCHEMA_PATH)) {
|
|
79
80
|
console.error(`Schema not found at ${SCHEMA_PATH}`);
|
|
80
81
|
return false;
|
|
81
82
|
}
|
|
82
|
-
const schema = JSON.parse(
|
|
83
|
+
const schema = JSON.parse(fs8.readFileSync(SCHEMA_PATH, "utf-8"));
|
|
83
84
|
const ajv = new Ajv({ allErrors: true, strict: false });
|
|
84
85
|
addFormats(ajv);
|
|
85
86
|
const validateFn = ajv.compile(schema);
|
|
@@ -88,7 +89,7 @@ async function validate(file) {
|
|
|
88
89
|
const d = doc;
|
|
89
90
|
const pageCount = Array.isArray(d.pages) ? d.pages.length : 0;
|
|
90
91
|
const elCount = Array.isArray(d.pages) ? d.pages.reduce((acc, p) => acc + (Array.isArray(p?.elements) ? p.elements.length : 0), 0) : 0;
|
|
91
|
-
console.log(`\u2713 Valid: ${
|
|
92
|
+
console.log(`\u2713 Valid: ${path8.basename(filePath)}`);
|
|
92
93
|
console.log(` Format: ${d.$jdf}${bundle ? " (jdfx bundle)" : ""}`);
|
|
93
94
|
console.log(` Title: ${d.meta?.title}`);
|
|
94
95
|
console.log(` Pages: ${pageCount}`);
|
|
@@ -101,7 +102,7 @@ async function validate(file) {
|
|
|
101
102
|
}
|
|
102
103
|
return true;
|
|
103
104
|
}
|
|
104
|
-
console.error(`\u2717 Invalid: ${
|
|
105
|
+
console.error(`\u2717 Invalid: ${path8.basename(filePath)}`);
|
|
105
106
|
for (const err of validateFn.errors || []) {
|
|
106
107
|
const loc = err.instancePath || "(root)";
|
|
107
108
|
console.error(` ${loc} \u2014 ${err.message}`);
|
|
@@ -304,13 +305,13 @@ function stripInline(text) {
|
|
|
304
305
|
return parseInline(text).map((r) => r.text).join("");
|
|
305
306
|
}
|
|
306
307
|
async function importMarkdown(inputPath, outputPath) {
|
|
307
|
-
const input =
|
|
308
|
+
const input = path8.resolve(inputPath);
|
|
308
309
|
console.log(`Importing: ${input}`);
|
|
309
|
-
const content =
|
|
310
|
-
const doc = convertMarkdownToJdf(content,
|
|
310
|
+
const content = fs8.readFileSync(input, "utf-8");
|
|
311
|
+
const doc = convertMarkdownToJdf(content, path8.basename(input, path8.extname(input)), path8.dirname(input));
|
|
311
312
|
let output;
|
|
312
313
|
if (outputPath) {
|
|
313
|
-
output =
|
|
314
|
+
output = path8.resolve(outputPath);
|
|
314
315
|
} else {
|
|
315
316
|
const stem = input.replace(/\.(md|markdown)$/i, "");
|
|
316
317
|
output = stem + (shouldUseJdfx(doc) ? ".jdfx" : ".jdf");
|
|
@@ -318,11 +319,11 @@ async function importMarkdown(inputPath, outputPath) {
|
|
|
318
319
|
console.log(`Output: ${output}`);
|
|
319
320
|
if (output.toLowerCase().endsWith(".jdfx")) {
|
|
320
321
|
const { bytes, manifest } = await packJdfx(doc);
|
|
321
|
-
|
|
322
|
+
fs8.writeFileSync(output, bytes);
|
|
322
323
|
console.log(`
|
|
323
324
|
Done! Created ${doc.pages.length} page(s), ${manifest.assets.length} asset(s) bundled`);
|
|
324
325
|
} else {
|
|
325
|
-
|
|
326
|
+
fs8.writeFileSync(output, JSON.stringify(doc, null, 2));
|
|
326
327
|
console.log(`
|
|
327
328
|
Done! Created ${doc.pages.length} page(s)`);
|
|
328
329
|
}
|
|
@@ -339,10 +340,10 @@ var MIME_BY_EXT2 = {
|
|
|
339
340
|
};
|
|
340
341
|
function resolveImageSrc(src, baseDir) {
|
|
341
342
|
if (/^(https?:|data:|file:)/i.test(src)) return src;
|
|
342
|
-
const abs =
|
|
343
|
+
const abs = path8.isAbsolute(src) ? src : path8.resolve(baseDir, src);
|
|
343
344
|
try {
|
|
344
|
-
const bytes =
|
|
345
|
-
const ext =
|
|
345
|
+
const bytes = fs8.readFileSync(abs);
|
|
346
|
+
const ext = path8.extname(abs).slice(1).toLowerCase();
|
|
346
347
|
const mime = MIME_BY_EXT2[ext] || "application/octet-stream";
|
|
347
348
|
return `data:${mime};base64,${bytes.toString("base64")}`;
|
|
348
349
|
} catch {
|
|
@@ -1699,6 +1700,10 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
|
1699
1700
|
if (arr) arr.push(op);
|
|
1700
1701
|
else opBins.set(key, [op]);
|
|
1701
1702
|
}
|
|
1703
|
+
const sizePenalty = (op, fontSize) => {
|
|
1704
|
+
if (!op.fontSize || !fontSize) return 0;
|
|
1705
|
+
return Math.abs(Math.log(op.fontSize / fontSize)) * 6;
|
|
1706
|
+
};
|
|
1702
1707
|
const findOp = (x, y, fontSize) => {
|
|
1703
1708
|
let best = null;
|
|
1704
1709
|
let bestD = Infinity;
|
|
@@ -1708,7 +1713,7 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
|
1708
1713
|
const arr = opBins.get(`${bx + dx},${by + dy}`);
|
|
1709
1714
|
if (!arr) continue;
|
|
1710
1715
|
for (const op of arr) {
|
|
1711
|
-
const d = Math.hypot(op.x - x, op.y - y);
|
|
1716
|
+
const d = Math.hypot(op.x - x, op.y - y) + sizePenalty(op, fontSize);
|
|
1712
1717
|
if (d < bestD) {
|
|
1713
1718
|
bestD = d;
|
|
1714
1719
|
best = op;
|
|
@@ -1718,8 +1723,9 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
|
1718
1723
|
}
|
|
1719
1724
|
if (best) return best;
|
|
1720
1725
|
const tol = Math.max(2, fontSize * 0.6);
|
|
1726
|
+
const sizeOk = (op) => !op.fontSize || !fontSize || op.fontSize / fontSize > 0.6 && op.fontSize / fontSize < 1.7;
|
|
1721
1727
|
for (const op of ops.textOps) {
|
|
1722
|
-
if (Math.abs(op.y - y) > tol) continue;
|
|
1728
|
+
if (!sizeOk(op) || Math.abs(op.y - y) > tol) continue;
|
|
1723
1729
|
const d = Math.abs(op.x - x) + Math.abs(op.y - y) * 4;
|
|
1724
1730
|
if (d < bestD) {
|
|
1725
1731
|
bestD = d;
|
|
@@ -1728,6 +1734,7 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
|
1728
1734
|
}
|
|
1729
1735
|
if (best) return best;
|
|
1730
1736
|
for (const op of ops.textOps) {
|
|
1737
|
+
if (!sizeOk(op)) continue;
|
|
1731
1738
|
const d = Math.hypot(op.x - x, op.y - y);
|
|
1732
1739
|
if (d < bestD) {
|
|
1733
1740
|
bestD = d;
|
|
@@ -1746,6 +1753,7 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
|
1746
1753
|
const conv = viewport.convertToViewportPoint(baseX, baseY);
|
|
1747
1754
|
const vx = safeNum(conv?.[0], 0);
|
|
1748
1755
|
const vy = safeNum(conv?.[1], 0);
|
|
1756
|
+
if (fontSize < 1.5) return;
|
|
1749
1757
|
const op = findOp(vx, vy, fontSize);
|
|
1750
1758
|
const mode = op?.mode ?? 0;
|
|
1751
1759
|
if (mode === 7) return;
|
|
@@ -1869,10 +1877,92 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
|
1869
1877
|
sizeChars.set(k, (sizeChars.get(k) ?? 0) + l.text.length);
|
|
1870
1878
|
}
|
|
1871
1879
|
const bodyFontSize = [...sizeChars.entries()].sort((a, b) => b[1] - a[1])[0]?.[0] ?? 0;
|
|
1880
|
+
const rowOf = /* @__PURE__ */ new Map();
|
|
1881
|
+
const rowStartOf = /* @__PURE__ */ new Map();
|
|
1882
|
+
const nextOnRow = /* @__PURE__ */ new Map();
|
|
1883
|
+
{
|
|
1884
|
+
const order = lines.map((_, i) => i).filter((i) => !consumedLines.has(i));
|
|
1885
|
+
for (let a = 0; a < order.length; a++) {
|
|
1886
|
+
const i = order[a], li = lines[i];
|
|
1887
|
+
const tolY = Math.max(0.6, li.fontSize * PT_TO_MM2 * 0.35);
|
|
1888
|
+
let bestNext = -1, bestX = Infinity;
|
|
1889
|
+
for (let b = 0; b < order.length; b++) {
|
|
1890
|
+
const j = order[b], lj = lines[j];
|
|
1891
|
+
if (j === i || Math.abs(lj.y - li.y) > tolY || lj.x <= li.x) continue;
|
|
1892
|
+
if (lj.x < bestX) {
|
|
1893
|
+
bestX = lj.x;
|
|
1894
|
+
bestNext = j;
|
|
1895
|
+
}
|
|
1896
|
+
}
|
|
1897
|
+
if (bestNext >= 0) nextOnRow.set(i, bestNext);
|
|
1898
|
+
}
|
|
1899
|
+
const seen = /* @__PURE__ */ new Set();
|
|
1900
|
+
for (const i of order) {
|
|
1901
|
+
if (seen.has(i)) continue;
|
|
1902
|
+
const row = [i];
|
|
1903
|
+
seen.add(i);
|
|
1904
|
+
let cur = i;
|
|
1905
|
+
while (nextOnRow.has(cur)) {
|
|
1906
|
+
const j = nextOnRow.get(cur), lc = lines[cur], lj = lines[j];
|
|
1907
|
+
const em = Math.min(lc.fontSize, lj.fontSize) * PT_TO_MM2;
|
|
1908
|
+
const gap = lj.x - (lc.x + lc.width);
|
|
1909
|
+
if (gap < -em * 0.3 || gap > em * 0.6) break;
|
|
1910
|
+
row.push(j);
|
|
1911
|
+
seen.add(j);
|
|
1912
|
+
cur = j;
|
|
1913
|
+
}
|
|
1914
|
+
rowOf.set(i, row);
|
|
1915
|
+
for (const j of row) rowStartOf.set(j, i);
|
|
1916
|
+
}
|
|
1917
|
+
}
|
|
1918
|
+
const runStyle = (l) => {
|
|
1919
|
+
const cls = fontMap.get(l.fontName) || classifyFont(l.fontName || "");
|
|
1920
|
+
return { cls, bold: cls.weight === "bold", italic: cls.style === "italic" };
|
|
1921
|
+
};
|
|
1872
1922
|
lines.forEach((l, lineIdx) => {
|
|
1873
1923
|
const tableEl = tableAtLine.get(lineIdx);
|
|
1874
1924
|
if (tableEl) elements.push(tableEl);
|
|
1875
1925
|
if (consumedLines.has(lineIdx)) return;
|
|
1926
|
+
const row = rowOf.get(lineIdx);
|
|
1927
|
+
if (!row) return;
|
|
1928
|
+
if (row.length > 1) {
|
|
1929
|
+
const first = lines[row[0]], last = lines[row[row.length - 1]];
|
|
1930
|
+
const base = runStyle(first);
|
|
1931
|
+
const rowEnd = last.x + last.width;
|
|
1932
|
+
const measuredW = Math.max((rowEnd - first.x) * 1.2 + first.fontSize * PT_TO_MM2 * 0.4, first.fontSize * PT_TO_MM2);
|
|
1933
|
+
const nextIdx2 = nextOnRow.get(row[row.length - 1]);
|
|
1934
|
+
const cap2 = nextIdx2 != null ? lines[nextIdx2].x - first.x - first.fontSize * PT_TO_MM2 * 0.3 : pageWmm - first.x;
|
|
1935
|
+
const runs2 = [];
|
|
1936
|
+
row.forEach((idx, k) => {
|
|
1937
|
+
const r = lines[idx];
|
|
1938
|
+
const st = runStyle(r);
|
|
1939
|
+
let text2 = r.text;
|
|
1940
|
+
if (k > 0) {
|
|
1941
|
+
const prev2 = lines[row[k - 1]];
|
|
1942
|
+
const gap = r.x - (prev2.x + prev2.width);
|
|
1943
|
+
if (gap > r.fontSize * PT_TO_MM2 * 0.08 && !/\s$/.test(prev2.text) && !/^\s/.test(text2)) text2 = " " + text2;
|
|
1944
|
+
}
|
|
1945
|
+
const run = { text: text2 };
|
|
1946
|
+
if (st.bold) run.bold = true;
|
|
1947
|
+
if (st.italic) run.italic = true;
|
|
1948
|
+
if (r.color !== "#000000") run.color = r.color;
|
|
1949
|
+
if (Math.abs(r.fontSize - first.fontSize) >= 0.5) run.fontSize = Math.round(r.fontSize * 10) / 10;
|
|
1950
|
+
if (st.cls.family !== base.cls.family) run.fontFamily = st.cls.family;
|
|
1951
|
+
const lk = findLinkForRun2(r);
|
|
1952
|
+
if (lk) run.link = lk.url ? lk.url : lk.destPage != null ? { type: "internal", target: `#page-${lk.destPage + 1}` } : void 0;
|
|
1953
|
+
runs2.push(run);
|
|
1954
|
+
});
|
|
1955
|
+
const style2 = { fontSize: Math.round(first.fontSize * 10) / 10, fontFamily: base.cls.family };
|
|
1956
|
+
if (first.opacity < 0.999) style2.opacity = Math.round(first.opacity * 100) / 100;
|
|
1957
|
+
elements.push({
|
|
1958
|
+
type: "richtext",
|
|
1959
|
+
runs: runs2,
|
|
1960
|
+
position: { x: Math.max(0, Math.round(first.x * 100) / 100), y: Math.max(0, Math.round(Math.min(...row.map((i) => lines[i].y)) * 100) / 100) },
|
|
1961
|
+
width: Math.max(2, Math.round(Math.max(first.fontSize * PT_TO_MM2, Math.min(measuredW, cap2)) * 100) / 100),
|
|
1962
|
+
style: style2
|
|
1963
|
+
});
|
|
1964
|
+
return;
|
|
1965
|
+
}
|
|
1876
1966
|
const cls = fontMap.get(l.fontName) || classifyFont(l.fontName || "");
|
|
1877
1967
|
const style = {
|
|
1878
1968
|
fontSize: Math.round(l.fontSize * 10) / 10,
|
|
@@ -1883,9 +1973,11 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
|
1883
1973
|
if (l.color !== "#000000") style.color = l.color;
|
|
1884
1974
|
if (l.opacity < 0.999) style.opacity = Math.round(l.opacity * 100) / 100;
|
|
1885
1975
|
const link = findLinkForRun2(l);
|
|
1886
|
-
const measured = Math.max(l.width + l.fontSize * PT_TO_MM2 * 0.4, l.fontSize * PT_TO_MM2);
|
|
1976
|
+
const measured = Math.max(l.width * 1.2 + l.fontSize * PT_TO_MM2 * 0.4, l.fontSize * PT_TO_MM2);
|
|
1887
1977
|
const remaining = Math.max(measured, pageWmm - l.x);
|
|
1888
|
-
const
|
|
1978
|
+
const nextIdx = nextOnRow.get(lineIdx);
|
|
1979
|
+
const cap = nextIdx != null ? Math.max(l.fontSize * PT_TO_MM2, lines[nextIdx].x - l.x - l.fontSize * PT_TO_MM2 * 0.3) : Infinity;
|
|
1980
|
+
const elWidth = Math.min(measured, remaining, cap);
|
|
1889
1981
|
const text = {
|
|
1890
1982
|
type: "text",
|
|
1891
1983
|
content: l.text,
|
|
@@ -2097,25 +2189,209 @@ async function importPdfToJdf2(source, title, options = {}) {
|
|
|
2097
2189
|
};
|
|
2098
2190
|
return importPdfToJdf(source, title, runtime, options);
|
|
2099
2191
|
}
|
|
2192
|
+
var DEFAULT_OLLAMA_MODEL = "qwen2.5vl:3b";
|
|
2193
|
+
var DEFAULT_OPENAI_MODEL = "gpt-4o-mini";
|
|
2194
|
+
var OLLAMA_HOST = process.env.OLLAMA_HOST || "http://localhost:11434";
|
|
2195
|
+
async function loadDoc(file) {
|
|
2196
|
+
if (file.toLowerCase().endsWith(".jdfx")) {
|
|
2197
|
+
const zip = await JSZip.loadAsync(fs8.readFileSync(file));
|
|
2198
|
+
const f = zip.file(JDFX_DOCUMENT_PATH);
|
|
2199
|
+
if (!f) throw new Error(`Bundle missing ${JDFX_DOCUMENT_PATH}`);
|
|
2200
|
+
const doc = JSON.parse(await f.async("string"));
|
|
2201
|
+
const manifest = zip.file("manifest.json") ? JSON.parse(await zip.file("manifest.json").async("string")) : { assets: [] };
|
|
2202
|
+
for (const a of manifest.assets ?? []) {
|
|
2203
|
+
const af = zip.file(a.path);
|
|
2204
|
+
if (!af) continue;
|
|
2205
|
+
const data = (await af.async("nodebuffer")).toString("base64");
|
|
2206
|
+
const res = { src: "embedded", mimeType: a.mimeType, data };
|
|
2207
|
+
doc.resources ??= {};
|
|
2208
|
+
if (/^video\//i.test(a.mimeType || "")) (doc.resources.videos ??= {})[a.id] = res;
|
|
2209
|
+
else (doc.resources.images ??= {})[a.id] = res;
|
|
2210
|
+
}
|
|
2211
|
+
return { doc, bundle: true };
|
|
2212
|
+
}
|
|
2213
|
+
return { doc: JSON.parse(fs8.readFileSync(file, "utf-8")), bundle: false };
|
|
2214
|
+
}
|
|
2215
|
+
function findImages(doc) {
|
|
2216
|
+
const out = [];
|
|
2217
|
+
const walk2 = (els, page) => {
|
|
2218
|
+
for (const el of els ?? []) {
|
|
2219
|
+
if (el?.type === "image") out.push({ el, page, index: out.length });
|
|
2220
|
+
if (el?.elements) walk2(el.elements, page);
|
|
2221
|
+
}
|
|
2222
|
+
};
|
|
2223
|
+
doc.pages.forEach((p, i) => walk2(p.elements, i + 1));
|
|
2224
|
+
return out;
|
|
2225
|
+
}
|
|
2226
|
+
async function imageBytes(doc, el, docDir) {
|
|
2227
|
+
const fromData = (d, fallback) => {
|
|
2228
|
+
const m = d.match(/^data:([^;,]+)?[^,]*,(.*)$/s);
|
|
2229
|
+
return m ? { bytes: Buffer.from(m[2], "base64"), mime: m[1] || fallback } : { bytes: Buffer.from(d, "base64"), mime: fallback };
|
|
2230
|
+
};
|
|
2231
|
+
const res = el.resource ? doc.resources?.images?.[el.resource] ?? doc.resources?.[el.resource] : void 0;
|
|
2232
|
+
if (res?.data) return fromData(String(res.data), res.mimeType || "image/png");
|
|
2233
|
+
if (res?.path) {
|
|
2234
|
+
const p = path8.resolve(docDir, res.path);
|
|
2235
|
+
return fs8.existsSync(p) ? { bytes: fs8.readFileSync(p), mime: res.mimeType || "image/png" } : null;
|
|
2236
|
+
}
|
|
2237
|
+
const src = el.src;
|
|
2238
|
+
if (!src) return null;
|
|
2239
|
+
if (src.startsWith("data:")) return fromData(src, "image/png");
|
|
2240
|
+
if (/^https?:\/\//i.test(src)) {
|
|
2241
|
+
const r = await fetch(src);
|
|
2242
|
+
if (!r.ok) throw new Error(`download failed ${r.status}: ${src}`);
|
|
2243
|
+
return { bytes: Buffer.from(await r.arrayBuffer()), mime: r.headers.get("content-type") || "image/png" };
|
|
2244
|
+
}
|
|
2245
|
+
const local = path8.resolve(docDir, src);
|
|
2246
|
+
return fs8.existsSync(local) ? { bytes: fs8.readFileSync(local), mime: "image/png" } : null;
|
|
2247
|
+
}
|
|
2248
|
+
async function ocrTesseract(bytes, lang) {
|
|
2249
|
+
const { createWorker } = await import('tesseract.js');
|
|
2250
|
+
const cachePath = path8.join(os.homedir(), ".cache", "jdf", "tesseract");
|
|
2251
|
+
fs8.mkdirSync(cachePath, { recursive: true });
|
|
2252
|
+
const worker = await createWorker(lang, 1, { cachePath, logger: () => {
|
|
2253
|
+
} });
|
|
2254
|
+
try {
|
|
2255
|
+
const { data } = await worker.recognize(bytes, {}, { text: true, blocks: true });
|
|
2256
|
+
let w = 1, h = 1;
|
|
2257
|
+
try {
|
|
2258
|
+
const { loadImage } = await import('@napi-rs/canvas');
|
|
2259
|
+
const im = await loadImage(bytes);
|
|
2260
|
+
w = im.width || 1;
|
|
2261
|
+
h = im.height || 1;
|
|
2262
|
+
} catch {
|
|
2263
|
+
}
|
|
2264
|
+
const lines = data.blocks?.flatMap((b) => b.paragraphs?.flatMap((p) => p.lines ?? []) ?? []) ?? data.lines ?? [];
|
|
2265
|
+
const blocks = lines.map((ln) => ({ text: String(ln.text ?? "").replace(/\s+/g, " ").trim(), confidence: ln.confidence != null ? Math.round(ln.confidence) / 100 : void 0, bbox: ln.bbox ? { x: +(ln.bbox.x0 / w).toFixed(4), y: +(ln.bbox.y0 / h).toFixed(4), w: +((ln.bbox.x1 - ln.bbox.x0) / w).toFixed(4), h: +((ln.bbox.y1 - ln.bbox.y0) / h).toFixed(4) } : void 0 })).filter((b) => b.text.length > 0 && (b.confidence == null || b.confidence >= 0.3));
|
|
2266
|
+
if (!blocks.length && String(data.text ?? "").trim()) blocks.push({ text: String(data.text).replace(/\s+/g, " ").trim() });
|
|
2267
|
+
return { blocks, source: `tesseract.js:${lang}` };
|
|
2268
|
+
} finally {
|
|
2269
|
+
await worker.terminate();
|
|
2270
|
+
}
|
|
2271
|
+
}
|
|
2272
|
+
async function openaiVision(bytes, mime, model, prompt2) {
|
|
2273
|
+
const key = process.env.OPENAI_API_KEY;
|
|
2274
|
+
if (!key) throw new Error("OPENAI_API_KEY is not set");
|
|
2275
|
+
const base = process.env.OPENAI_BASE_URL || "https://api.openai.com/v1";
|
|
2276
|
+
const r = await fetch(`${base}/chat/completions`, { method: "POST", headers: { Authorization: `Bearer ${key}`, "content-type": "application/json" }, body: JSON.stringify({ model, messages: [{ role: "user", content: [{ type: "text", text: prompt2 }, { type: "image_url", image_url: { url: `data:${mime};base64,${bytes.toString("base64")}` } }] }], max_tokens: 800 }) });
|
|
2277
|
+
if (!r.ok) throw new Error(`OpenAI vision failed ${r.status}: ${(await r.text()).slice(0, 300)}`);
|
|
2278
|
+
const j = await r.json();
|
|
2279
|
+
return String(j.choices?.[0]?.message?.content ?? "").trim();
|
|
2280
|
+
}
|
|
2281
|
+
async function ollamaVision(bytes, model, prompt2) {
|
|
2282
|
+
const r = await fetch(`${OLLAMA_HOST}/api/generate`, { method: "POST", body: JSON.stringify({ model, prompt: prompt2, images: [bytes.toString("base64")], stream: false, options: { temperature: 0 } }) });
|
|
2283
|
+
if (!r.ok) throw new Error(`Ollama vision failed ${r.status}: ${(await r.text()).slice(0, 300)} \u2014 is the model pulled? (ollama pull ${model})`);
|
|
2284
|
+
const j = await r.json();
|
|
2285
|
+
return String(j.response ?? "").trim();
|
|
2286
|
+
}
|
|
2287
|
+
var CAPTION_PROMPT = "Describe this image for a search index in one or two factual sentences: what it shows, any chart type, axes, trends, labels, names and numbers you can read. No preamble.";
|
|
2288
|
+
var OCR_PROMPT = "Transcribe all text visible in this image exactly, line by line, top to bottom, left to right. Output only the text.";
|
|
2289
|
+
async function describeDocument(doc, docDir, opts = {}) {
|
|
2290
|
+
const ocrP = opts.ocr ?? "tesseract";
|
|
2291
|
+
const capP = opts.caption ?? "ollama";
|
|
2292
|
+
const lang = opts.ocrLanguage ?? "eng";
|
|
2293
|
+
const images = findImages(doc);
|
|
2294
|
+
const targets = opts.element != null ? images.filter((im) => im.el.id === opts.element || String(im.index) === opts.element) : images;
|
|
2295
|
+
if (opts.element != null && !targets.length) throw new Error(`no image element "${opts.element}" (have: ${images.map((i) => i.el.id ?? `#${i.index}`).join(", ") || "none"})`);
|
|
2296
|
+
const stats = { ocr: 0, captions: 0, skipped: 0, failed: [] };
|
|
2297
|
+
for (const im of targets) {
|
|
2298
|
+
const el = im.el;
|
|
2299
|
+
if (!el.id) el.id = `image-${im.index + 1}`;
|
|
2300
|
+
const needOcr = ocrP !== "none" && (opts.force || !el.ocr?.blocks?.length);
|
|
2301
|
+
const needCap = capP !== "none" && (opts.force || !el.caption);
|
|
2302
|
+
if (!needOcr && !needCap) {
|
|
2303
|
+
stats.skipped++;
|
|
2304
|
+
continue;
|
|
2305
|
+
}
|
|
2306
|
+
const img = await imageBytes(doc, el, docDir);
|
|
2307
|
+
if (!img) {
|
|
2308
|
+
stats.failed.push(`${el.id}: image bytes not reachable`);
|
|
2309
|
+
continue;
|
|
2310
|
+
}
|
|
2311
|
+
if (needOcr) {
|
|
2312
|
+
try {
|
|
2313
|
+
if (ocrP === "tesseract") {
|
|
2314
|
+
const r = await ocrTesseract(img.bytes, lang);
|
|
2315
|
+
el.ocr = { language: lang, source: r.source, created: (/* @__PURE__ */ new Date()).toISOString(), blocks: r.blocks };
|
|
2316
|
+
} else {
|
|
2317
|
+
const text = await openaiVision(img.bytes, img.mime, opts.captionModel ?? DEFAULT_OPENAI_MODEL, OCR_PROMPT);
|
|
2318
|
+
el.ocr = { source: `openai:${opts.captionModel ?? DEFAULT_OPENAI_MODEL}`, created: (/* @__PURE__ */ new Date()).toISOString(), blocks: text.split(/\r?\n/).map((t) => t.trim()).filter(Boolean).map((t) => ({ text: t })) };
|
|
2319
|
+
}
|
|
2320
|
+
stats.ocr++;
|
|
2321
|
+
} catch (e) {
|
|
2322
|
+
stats.failed.push(`${el.id} ocr: ${e.message}`);
|
|
2323
|
+
}
|
|
2324
|
+
}
|
|
2325
|
+
if (needCap) {
|
|
2326
|
+
try {
|
|
2327
|
+
const model = opts.captionModel ?? (capP === "ollama" ? DEFAULT_OLLAMA_MODEL : DEFAULT_OPENAI_MODEL);
|
|
2328
|
+
const text = capP === "ollama" ? await ollamaVision(img.bytes, model, CAPTION_PROMPT) : await openaiVision(img.bytes, img.mime, model, CAPTION_PROMPT);
|
|
2329
|
+
if (text) {
|
|
2330
|
+
el.caption = text;
|
|
2331
|
+
el.captionSource = `${capP}:${model}`;
|
|
2332
|
+
stats.captions++;
|
|
2333
|
+
}
|
|
2334
|
+
} catch (e) {
|
|
2335
|
+
stats.failed.push(`${el.id} caption: ${e.message}`);
|
|
2336
|
+
}
|
|
2337
|
+
}
|
|
2338
|
+
if (!opts.quiet) console.log(` \xB7 ${el.id} (page ${im.page}): ${needOcr ? `ocr ${el.ocr?.blocks?.length ?? 0} block(s)` : "ocr kept"}${needCap ? ` \xB7 caption ${el.caption ? `"${String(el.caption).slice(0, 70)}${String(el.caption).length > 70 ? "\u2026" : ""}"` : "\u2014"}` : ""}`);
|
|
2339
|
+
}
|
|
2340
|
+
return stats;
|
|
2341
|
+
}
|
|
2342
|
+
async function describeFile(inputPath, opts = {}) {
|
|
2343
|
+
const input = path8.resolve(inputPath);
|
|
2344
|
+
if (!fs8.existsSync(input)) throw new Error(`File not found: ${input}`);
|
|
2345
|
+
const { doc, bundle } = await loadDoc(input);
|
|
2346
|
+
const images = findImages(doc);
|
|
2347
|
+
if (!images.length) {
|
|
2348
|
+
console.log(`No image elements in ${path8.basename(input)} \u2014 nothing to describe.`);
|
|
2349
|
+
return;
|
|
2350
|
+
}
|
|
2351
|
+
console.log(`Describing: ${path8.basename(input)} \u2014 ${images.length} image(s); ocr=${opts.ocr ?? "tesseract"} caption=${opts.caption ?? "ollama"}${(opts.caption ?? "ollama") === "ollama" ? ` (${opts.captionModel ?? DEFAULT_OLLAMA_MODEL}, local)` : ""}`);
|
|
2352
|
+
const stats = await describeDocument(doc, path8.dirname(input), opts);
|
|
2353
|
+
const output = opts.output ? path8.resolve(opts.output) : input;
|
|
2354
|
+
if (output.toLowerCase().endsWith(".jdfx") || bundle && !opts.output) fs8.writeFileSync(output, (await packJdfx(doc)).bytes);
|
|
2355
|
+
else fs8.writeFileSync(output, JSON.stringify(doc, null, 2));
|
|
2356
|
+
console.log(`Done: ${stats.ocr} OCR, ${stats.captions} caption(s), ${stats.skipped} already had text${stats.failed.length ? `, ${stats.failed.length} failed` : ""}.`);
|
|
2357
|
+
for (const f of stats.failed) console.warn(` ! ${f}`);
|
|
2358
|
+
console.log(`Output: ${output}
|
|
2359
|
+
Next: jdf chunk ${path8.basename(output)} # image text is now part of the chunks`);
|
|
2360
|
+
if (stats.failed.length && stats.ocr + stats.captions === 0) process.exitCode = 1;
|
|
2361
|
+
}
|
|
2100
2362
|
|
|
2101
2363
|
// src/commands/import-pdf.ts
|
|
2102
2364
|
async function importPdf(inputPath, outputPath, options = {}) {
|
|
2103
|
-
const input =
|
|
2104
|
-
if (!
|
|
2365
|
+
const input = path8.resolve(inputPath);
|
|
2366
|
+
if (!fs8.existsSync(input)) {
|
|
2105
2367
|
console.error(`File not found: ${input}`);
|
|
2106
2368
|
process.exit(1);
|
|
2107
2369
|
}
|
|
2108
2370
|
console.log(`Importing: ${input}`);
|
|
2109
|
-
const title =
|
|
2371
|
+
const title = path8.basename(input, path8.extname(input));
|
|
2110
2372
|
const t0 = Date.now();
|
|
2111
2373
|
const doc = await importPdfToJdf2(input, title, {
|
|
2112
2374
|
password: options.password,
|
|
2113
2375
|
invisibleText: options.dropInvisibleText ? "drop" : "keep"
|
|
2114
2376
|
});
|
|
2115
2377
|
console.log(`Parsed in ${((Date.now() - t0) / 1e3).toFixed(1)}s \u2014 ${doc.pages.length} page(s)`);
|
|
2378
|
+
const scanned = doc.pages.filter((p) => !p.elements.some((e) => e.type === "text" || e.type === "richtext" || e.type === "table") && p.elements.some((e) => e.type === "image"));
|
|
2379
|
+
if (scanned.length) {
|
|
2380
|
+
if (options.ocr && options.ocr !== "none") {
|
|
2381
|
+
console.log(`OCR: ${scanned.length} scanned page(s) \u2192 ${options.ocr}`);
|
|
2382
|
+
for (const p of scanned) for (const e of p.elements) if (e.type === "image" && !e.id) e.id = `scan-${doc.pages.indexOf(p) + 1}`;
|
|
2383
|
+
const ids = scanned.flatMap((p) => p.elements.filter((e) => e.type === "image").map((e) => e.id));
|
|
2384
|
+
for (const id of ids) {
|
|
2385
|
+
const st = await describeDocument(doc, path8.dirname(input), { element: id, ocr: options.ocr, caption: "none", quiet: true });
|
|
2386
|
+
if (st.failed.length) console.warn(` ! ${st.failed.join("; ")}`);
|
|
2387
|
+
}
|
|
2388
|
+
} else {
|
|
2389
|
+
console.warn(`! ${scanned.length} page(s) have no text layer (scanned). RAG will skip them \u2014 re-run with --ocr tesseract (local) or --ocr openai.`);
|
|
2390
|
+
}
|
|
2391
|
+
}
|
|
2116
2392
|
let output;
|
|
2117
2393
|
if (outputPath) {
|
|
2118
|
-
output =
|
|
2394
|
+
output = path8.resolve(outputPath);
|
|
2119
2395
|
} else {
|
|
2120
2396
|
const stem = input.replace(/\.pdf$/i, "");
|
|
2121
2397
|
const wantJdfx = !options.forceJson && shouldUseJdfx(doc);
|
|
@@ -2124,11 +2400,11 @@ async function importPdf(inputPath, outputPath, options = {}) {
|
|
|
2124
2400
|
console.log(`Output: ${output}`);
|
|
2125
2401
|
if (output.toLowerCase().endsWith(".jdfx")) {
|
|
2126
2402
|
const { bytes, manifest } = await packJdfx(doc);
|
|
2127
|
-
|
|
2403
|
+
fs8.writeFileSync(output, bytes);
|
|
2128
2404
|
console.log(`
|
|
2129
2405
|
Done! Created ${doc.pages.length} page(s), ${manifest.assets.length} asset(s) bundled`);
|
|
2130
2406
|
} else {
|
|
2131
|
-
|
|
2407
|
+
fs8.writeFileSync(output, JSON.stringify(doc, null, 2));
|
|
2132
2408
|
console.log(`
|
|
2133
2409
|
Done! Created ${doc.pages.length} page(s)`);
|
|
2134
2410
|
}
|
|
@@ -2141,23 +2417,23 @@ var ImportJsonError = class extends Error {
|
|
|
2141
2417
|
}
|
|
2142
2418
|
};
|
|
2143
2419
|
async function importJson(inputPath, outputPath, options = {}) {
|
|
2144
|
-
const input =
|
|
2145
|
-
if (!
|
|
2420
|
+
const input = path8.resolve(inputPath);
|
|
2421
|
+
if (!fs8.existsSync(input)) {
|
|
2146
2422
|
throw new ImportJsonError(`File not found: ${input}`);
|
|
2147
2423
|
}
|
|
2148
2424
|
console.log(`Importing: ${input}`);
|
|
2149
|
-
const raw =
|
|
2425
|
+
const raw = fs8.readFileSync(input, "utf-8");
|
|
2150
2426
|
let parsed;
|
|
2151
2427
|
try {
|
|
2152
2428
|
parsed = JSON.parse(raw);
|
|
2153
2429
|
} catch (e) {
|
|
2154
2430
|
throw new ImportJsonError(`Not valid JSON: ${e.message}`);
|
|
2155
2431
|
}
|
|
2156
|
-
const title =
|
|
2432
|
+
const title = path8.basename(input, path8.extname(input));
|
|
2157
2433
|
const doc = normaliseToJdf(parsed, title);
|
|
2158
2434
|
let output;
|
|
2159
2435
|
if (outputPath) {
|
|
2160
|
-
output =
|
|
2436
|
+
output = path8.resolve(outputPath);
|
|
2161
2437
|
} else {
|
|
2162
2438
|
const stem = input.replace(/\.json$/i, "");
|
|
2163
2439
|
const wantJdfx = !options.forceJson && shouldUseJdfx(doc);
|
|
@@ -2166,11 +2442,11 @@ async function importJson(inputPath, outputPath, options = {}) {
|
|
|
2166
2442
|
console.log(`Output: ${output}`);
|
|
2167
2443
|
if (output.toLowerCase().endsWith(".jdfx")) {
|
|
2168
2444
|
const { bytes, manifest } = await packJdfx(doc);
|
|
2169
|
-
|
|
2445
|
+
fs8.writeFileSync(output, bytes);
|
|
2170
2446
|
console.log(`
|
|
2171
2447
|
Done! Created ${doc.pages.length} page(s), ${manifest.assets.length} asset(s) bundled`);
|
|
2172
2448
|
} else {
|
|
2173
|
-
|
|
2449
|
+
fs8.writeFileSync(output, JSON.stringify(doc, null, 2));
|
|
2174
2450
|
console.log(`
|
|
2175
2451
|
Done! Created ${doc.pages.length} page(s)`);
|
|
2176
2452
|
}
|
|
@@ -2272,8 +2548,8 @@ function transcriptChunks(el, elementId2, page, crumb, windowSec, maxTokens) {
|
|
|
2272
2548
|
const body = win.map((sg) => sg.speaker ? `${sg.speaker}: ${sg.text}` : sg.text).join(" ").replace(/\s+/g, " ").trim();
|
|
2273
2549
|
const text = `[${fmtTime(t0)}\u2013${fmtTime(t1)}] ${body}`;
|
|
2274
2550
|
const chapter = chapterAt(t0);
|
|
2275
|
-
const
|
|
2276
|
-
out.push({ id: `${elementId2}@${Math.round(t0)}`, text, path:
|
|
2551
|
+
const path10 = [...crumb, ...el.title ? [String(el.title)] : [], ...chapter ? [String(chapter.title)] : []];
|
|
2552
|
+
out.push({ id: `${elementId2}@${Math.round(t0)}`, text, path: path10, page, types: ["video"], tokens: estimateTokens(text), hash: hashText(text), media: { element: elementId2, t0, t1 } });
|
|
2277
2553
|
win = [];
|
|
2278
2554
|
};
|
|
2279
2555
|
for (const sg of segs) {
|
|
@@ -2342,8 +2618,14 @@ function serializeElement(el) {
|
|
|
2342
2618
|
}
|
|
2343
2619
|
case "checkbox":
|
|
2344
2620
|
return `${e.checked ? "[x]" : "[ ]"} ${e.label ?? ""}`.trim();
|
|
2345
|
-
case "image":
|
|
2346
|
-
|
|
2621
|
+
case "image": {
|
|
2622
|
+
const parts = [];
|
|
2623
|
+
if (e.alt) parts.push(`[image: ${e.alt}]`);
|
|
2624
|
+
if (e.caption) parts.push(String(e.caption).trim());
|
|
2625
|
+
const ocr = (e.ocr?.blocks || []).map((b) => String(b.text ?? "").trim()).filter(Boolean).join("\n");
|
|
2626
|
+
if (ocr) parts.push(ocr);
|
|
2627
|
+
return parts.join("\n");
|
|
2628
|
+
}
|
|
2347
2629
|
case "video":
|
|
2348
2630
|
return e.title ? `[video: ${e.title}]` : "";
|
|
2349
2631
|
case "toc":
|
|
@@ -2463,34 +2745,62 @@ function chunkDocument(doc, options = {}) {
|
|
|
2463
2745
|
}
|
|
2464
2746
|
async function loadJdf(filePath) {
|
|
2465
2747
|
if (filePath.toLowerCase().endsWith(".jdfx")) {
|
|
2466
|
-
const zip = await JSZip.loadAsync(
|
|
2748
|
+
const zip = await JSZip.loadAsync(fs8.readFileSync(filePath));
|
|
2467
2749
|
const docFile = zip.file(JDFX_DOCUMENT_PATH);
|
|
2468
2750
|
if (!docFile) throw new Error(`Bundle missing ${JDFX_DOCUMENT_PATH}`);
|
|
2469
2751
|
return JSON.parse(await docFile.async("string"));
|
|
2470
2752
|
}
|
|
2471
|
-
return JSON.parse(
|
|
2753
|
+
return JSON.parse(fs8.readFileSync(filePath, "utf-8"));
|
|
2754
|
+
}
|
|
2755
|
+
function mediaCoverage(doc) {
|
|
2756
|
+
const cov = { images: { total: 0, covered: 0, missing: [] }, videos: { total: 0, covered: 0, missing: [] } };
|
|
2757
|
+
const walk2 = (els, page) => {
|
|
2758
|
+
for (const el of els ?? []) {
|
|
2759
|
+
if (el?.type === "image") {
|
|
2760
|
+
cov.images.total++;
|
|
2761
|
+
const has = !!(el.caption && String(el.caption).trim()) || !!el.ocr?.blocks?.some((b) => String(b.text ?? "").trim());
|
|
2762
|
+
if (has) cov.images.covered++;
|
|
2763
|
+
else cov.images.missing.push({ id: el.id, page, alt: el.alt });
|
|
2764
|
+
} else if (el?.type === "video") {
|
|
2765
|
+
cov.videos.total++;
|
|
2766
|
+
if (el.transcript?.segments?.length) cov.videos.covered++;
|
|
2767
|
+
else cov.videos.missing.push({ id: el.id, page, title: el.title });
|
|
2768
|
+
}
|
|
2769
|
+
if (el?.elements) walk2(el.elements, page);
|
|
2770
|
+
}
|
|
2771
|
+
};
|
|
2772
|
+
doc.pages.forEach((p, i) => walk2(p.elements, i + 1));
|
|
2773
|
+
return cov;
|
|
2774
|
+
}
|
|
2775
|
+
function coverageSummary(cov) {
|
|
2776
|
+
const parts = [];
|
|
2777
|
+
if (cov.images.missing.length) parts.push(`${cov.images.missing.length} of ${cov.images.total} image(s) have no caption/OCR text \u2192 jdf describe`);
|
|
2778
|
+
if (cov.videos.missing.length) parts.push(`${cov.videos.missing.length} of ${cov.videos.total} video(s) have no transcript \u2192 jdf transcribe`);
|
|
2779
|
+
return parts.length ? parts.join("; ") : null;
|
|
2472
2780
|
}
|
|
2473
2781
|
async function chunkFile(inputPath, opts = {}) {
|
|
2474
|
-
const input =
|
|
2475
|
-
if (!
|
|
2782
|
+
const input = path8.resolve(inputPath);
|
|
2783
|
+
if (!fs8.existsSync(input)) throw new Error(`File not found: ${input}`);
|
|
2476
2784
|
const doc = await loadJdf(input);
|
|
2477
2785
|
const strategy = opts.strategy ?? "section";
|
|
2478
2786
|
const chunks = chunkDocument(doc, { strategy, maxTokens: opts.maxTokens, transcriptWindowSec: opts.transcriptWindowSec });
|
|
2787
|
+
const gap = coverageSummary(mediaCoverage(doc));
|
|
2788
|
+
if (gap) console.warn(` ! media without text (skipped by retrieval): ${gap}`);
|
|
2479
2789
|
const format = opts.format ?? "jsonl";
|
|
2480
2790
|
console.log(`Chunking: ${input}`);
|
|
2481
2791
|
console.log(`Strategy: ${strategy}${opts.maxTokens ? ` (max ${opts.maxTokens} tokens)` : ""}`);
|
|
2482
2792
|
if (format === "inline") {
|
|
2483
|
-
const out = opts.output ?
|
|
2793
|
+
const out = opts.output ? path8.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".jdf");
|
|
2484
2794
|
const withIndex = { ...doc, index: { chunker: `jdf-${strategy}-v1`, chunks } };
|
|
2485
|
-
|
|
2795
|
+
fs8.writeFileSync(out, JSON.stringify(withIndex, null, 2));
|
|
2486
2796
|
console.log(`Output: ${out} (${chunks.length} chunks in "index" block)`);
|
|
2487
2797
|
} else if (format === "json") {
|
|
2488
|
-
const out = opts.output ?
|
|
2489
|
-
|
|
2798
|
+
const out = opts.output ? path8.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".chunks.json");
|
|
2799
|
+
fs8.writeFileSync(out, JSON.stringify(chunks, null, 2));
|
|
2490
2800
|
console.log(`Output: ${out} (${chunks.length} chunks)`);
|
|
2491
2801
|
} else {
|
|
2492
|
-
const out = opts.output ?
|
|
2493
|
-
|
|
2802
|
+
const out = opts.output ? path8.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".chunks.jsonl");
|
|
2803
|
+
fs8.writeFileSync(out, chunks.map((c) => JSON.stringify(c)).join("\n") + "\n");
|
|
2494
2804
|
console.log(`Output: ${out} (${chunks.length} chunks)`);
|
|
2495
2805
|
}
|
|
2496
2806
|
const totalTokens = chunks.reduce((a, c) => a + c.tokens, 0);
|
|
@@ -2513,11 +2823,11 @@ async function embedBatch(provider, model, inputs) {
|
|
|
2513
2823
|
throw new Error(`Unknown embedding provider: ${provider}`);
|
|
2514
2824
|
}
|
|
2515
2825
|
}
|
|
2516
|
-
var
|
|
2826
|
+
var OLLAMA_HOST2 = process.env.OLLAMA_HOST || "http://localhost:11434";
|
|
2517
2827
|
var OLLAMA_CONTAINER = "jdf-ollama";
|
|
2518
2828
|
async function ollamaUp() {
|
|
2519
2829
|
try {
|
|
2520
|
-
const res = await fetch(`${
|
|
2830
|
+
const res = await fetch(`${OLLAMA_HOST2}/api/tags`, { signal: AbortSignal.timeout(1500) });
|
|
2521
2831
|
return res.ok;
|
|
2522
2832
|
} catch {
|
|
2523
2833
|
return false;
|
|
@@ -2542,12 +2852,12 @@ async function ensureOllama(model, autoStart) {
|
|
|
2542
2852
|
docker run -d --name ${OLLAMA_CONTAINER} -p 11434:11434 -v jdf-ollama:/root/.ollama ollama/ollama
|
|
2543
2853
|
docker exec ${OLLAMA_CONTAINER} ollama pull ${model}`;
|
|
2544
2854
|
if (!autoStart) {
|
|
2545
|
-
throw new Error(`Ollama isn't running at ${
|
|
2855
|
+
throw new Error(`Ollama isn't running at ${OLLAMA_HOST2} and --no-auto-start was given.
|
|
2546
2856
|
${manualHint}`);
|
|
2547
2857
|
}
|
|
2548
2858
|
if (!dockerReady()) {
|
|
2549
2859
|
throw new Error(
|
|
2550
|
-
`Ollama isn't running at ${
|
|
2860
|
+
`Ollama isn't running at ${OLLAMA_HOST2}, and the Docker daemon isn't available to auto-start it.
|
|
2551
2861
|
${manualHint}`
|
|
2552
2862
|
);
|
|
2553
2863
|
}
|
|
@@ -2584,7 +2894,7 @@ ${manualHint}`);
|
|
|
2584
2894
|
}
|
|
2585
2895
|
async function ollamaPull(model) {
|
|
2586
2896
|
try {
|
|
2587
|
-
const show = await fetch(`${
|
|
2897
|
+
const show = await fetch(`${OLLAMA_HOST2}/api/show`, {
|
|
2588
2898
|
method: "POST",
|
|
2589
2899
|
headers: { "Content-Type": "application/json" },
|
|
2590
2900
|
body: JSON.stringify({ name: model })
|
|
@@ -2593,7 +2903,7 @@ async function ollamaPull(model) {
|
|
|
2593
2903
|
} catch {
|
|
2594
2904
|
}
|
|
2595
2905
|
console.log(`Pulling embedding model "${model}" into Ollama (first run only)\u2026`);
|
|
2596
|
-
const res = await fetch(`${
|
|
2906
|
+
const res = await fetch(`${OLLAMA_HOST2}/api/pull`, {
|
|
2597
2907
|
method: "POST",
|
|
2598
2908
|
headers: { "Content-Type": "application/json" },
|
|
2599
2909
|
body: JSON.stringify({ name: model, stream: false })
|
|
@@ -2604,7 +2914,7 @@ async function ollamaPull(model) {
|
|
|
2604
2914
|
async function embedOllama(model, inputs) {
|
|
2605
2915
|
const out = [];
|
|
2606
2916
|
for (const text of inputs) {
|
|
2607
|
-
const res = await fetch(`${
|
|
2917
|
+
const res = await fetch(`${OLLAMA_HOST2}/api/embeddings`, {
|
|
2608
2918
|
method: "POST",
|
|
2609
2919
|
headers: { "Content-Type": "application/json" },
|
|
2610
2920
|
body: JSON.stringify({ model, prompt: text })
|
|
@@ -2662,17 +2972,17 @@ async function embedOpenAI(model, inputs) {
|
|
|
2662
2972
|
}
|
|
2663
2973
|
async function loadJdf2(filePath) {
|
|
2664
2974
|
if (filePath.toLowerCase().endsWith(".jdfx")) {
|
|
2665
|
-
const zip = await JSZip.loadAsync(
|
|
2975
|
+
const zip = await JSZip.loadAsync(fs8.readFileSync(filePath));
|
|
2666
2976
|
const docFile = zip.file(JDFX_DOCUMENT_PATH);
|
|
2667
2977
|
if (!docFile) throw new Error(`Bundle missing ${JDFX_DOCUMENT_PATH}`);
|
|
2668
2978
|
return JSON.parse(await docFile.async("string"));
|
|
2669
2979
|
}
|
|
2670
|
-
return JSON.parse(
|
|
2980
|
+
return JSON.parse(fs8.readFileSync(filePath, "utf-8"));
|
|
2671
2981
|
}
|
|
2672
2982
|
function loadCache(cachePath) {
|
|
2673
2983
|
try {
|
|
2674
|
-
if (!
|
|
2675
|
-
return JSON.parse(
|
|
2984
|
+
if (!fs8.existsSync(cachePath)) return null;
|
|
2985
|
+
return JSON.parse(fs8.readFileSync(cachePath, "utf-8"));
|
|
2676
2986
|
} catch {
|
|
2677
2987
|
return null;
|
|
2678
2988
|
}
|
|
@@ -2683,15 +2993,15 @@ function batched(items, size) {
|
|
|
2683
2993
|
return out;
|
|
2684
2994
|
}
|
|
2685
2995
|
async function embedFile(inputPath, opts = {}) {
|
|
2686
|
-
const input =
|
|
2687
|
-
if (!
|
|
2996
|
+
const input = path8.resolve(inputPath);
|
|
2997
|
+
if (!fs8.existsSync(input)) throw new Error(`File not found: ${input}`);
|
|
2688
2998
|
const provider = opts.provider ?? "ollama";
|
|
2689
2999
|
const model = opts.model ?? DEFAULT_MODEL[provider];
|
|
2690
3000
|
const strategy = opts.strategy ?? "section";
|
|
2691
3001
|
const doc = await loadJdf2(input);
|
|
2692
3002
|
const chunks = chunkDocument(doc, { strategy, maxTokens: opts.maxTokens, transcriptWindowSec: opts.transcriptWindowSec });
|
|
2693
|
-
const output = opts.output ?
|
|
2694
|
-
const cachePath = opts.cache ?
|
|
3003
|
+
const output = opts.output ? path8.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".embeddings.json");
|
|
3004
|
+
const cachePath = opts.cache ? path8.resolve(opts.cache) : output;
|
|
2695
3005
|
console.log(`Embedding: ${input}`);
|
|
2696
3006
|
console.log(`Provider: ${provider} / ${model}${provider === "ollama" ? " (local \u2014 no data leaves this machine)" : " (remote API)"}`);
|
|
2697
3007
|
console.log(`Strategy: ${strategy} \u2192 ${chunks.length} chunks`);
|
|
@@ -2730,7 +3040,7 @@ async function embedFile(inputPath, opts = {}) {
|
|
|
2730
3040
|
chunker: `jdf-${strategy}-v1`,
|
|
2731
3041
|
vectors
|
|
2732
3042
|
};
|
|
2733
|
-
|
|
3043
|
+
fs8.writeFileSync(output, JSON.stringify(sidecar));
|
|
2734
3044
|
console.log(`
|
|
2735
3045
|
Done! ${Object.keys(vectors).length} vectors (${dims}-dim) \u2192 ${output}`);
|
|
2736
3046
|
return sidecar;
|
|
@@ -2774,9 +3084,9 @@ function parseChapters(text) {
|
|
|
2774
3084
|
return { t: toSec(m[1]), title: m[2].trim() };
|
|
2775
3085
|
});
|
|
2776
3086
|
}
|
|
2777
|
-
async function
|
|
3087
|
+
async function loadDoc2(file) {
|
|
2778
3088
|
if (file.toLowerCase().endsWith(".jdfx")) {
|
|
2779
|
-
const zip = await JSZip.loadAsync(
|
|
3089
|
+
const zip = await JSZip.loadAsync(fs8.readFileSync(file));
|
|
2780
3090
|
const f = zip.file(JDFX_DOCUMENT_PATH);
|
|
2781
3091
|
if (!f) throw new Error(`Bundle missing ${JDFX_DOCUMENT_PATH}`);
|
|
2782
3092
|
const doc = JSON.parse(await f.async("string"));
|
|
@@ -2792,7 +3102,7 @@ async function loadDoc(file) {
|
|
|
2792
3102
|
}
|
|
2793
3103
|
return { doc, bundle: true, zip };
|
|
2794
3104
|
}
|
|
2795
|
-
return { doc: JSON.parse(
|
|
3105
|
+
return { doc: JSON.parse(fs8.readFileSync(file, "utf-8")), bundle: false };
|
|
2796
3106
|
}
|
|
2797
3107
|
function findVideos(doc) {
|
|
2798
3108
|
const out = [];
|
|
@@ -2806,27 +3116,27 @@ function findVideos(doc) {
|
|
|
2806
3116
|
return out;
|
|
2807
3117
|
}
|
|
2808
3118
|
async function clipToTempFile(doc, el, docDir) {
|
|
2809
|
-
const tmp =
|
|
3119
|
+
const tmp = path8.join(fs8.mkdtempSync(path8.join(__require("os").tmpdir(), "jdf-transcribe-")), "clip.mp4");
|
|
2810
3120
|
const res = el.resource ? doc.resources?.videos?.[el.resource] ?? doc.resources?.images?.[el.resource] : void 0;
|
|
2811
3121
|
if (res?.data) {
|
|
2812
|
-
|
|
3122
|
+
fs8.writeFileSync(tmp, Buffer.from(res.data.replace(/^data:[^,]*,/, ""), "base64"));
|
|
2813
3123
|
return tmp;
|
|
2814
3124
|
}
|
|
2815
|
-
if (res?.path) return
|
|
3125
|
+
if (res?.path) return path8.resolve(docDir, res.path);
|
|
2816
3126
|
const src = el.src;
|
|
2817
3127
|
if (!src) return null;
|
|
2818
3128
|
if (src.startsWith("data:")) {
|
|
2819
|
-
|
|
3129
|
+
fs8.writeFileSync(tmp, Buffer.from(src.replace(/^data:[^,]*,/, ""), "base64"));
|
|
2820
3130
|
return tmp;
|
|
2821
3131
|
}
|
|
2822
3132
|
if (/^https?:\/\//i.test(src)) {
|
|
2823
3133
|
const r = await fetch(src);
|
|
2824
3134
|
if (!r.ok) throw new Error(`download failed ${r.status}: ${src}`);
|
|
2825
|
-
|
|
3135
|
+
fs8.writeFileSync(tmp, Buffer.from(await r.arrayBuffer()));
|
|
2826
3136
|
return tmp;
|
|
2827
3137
|
}
|
|
2828
|
-
const local =
|
|
2829
|
-
return
|
|
3138
|
+
const local = path8.resolve(docDir, src);
|
|
3139
|
+
return fs8.existsSync(local) ? local : null;
|
|
2830
3140
|
}
|
|
2831
3141
|
function whisperCli(clip, model, language, prompt2) {
|
|
2832
3142
|
const ffmpeg = spawnSync("ffmpeg", ["-version"]);
|
|
@@ -2841,7 +3151,7 @@ function whisperCli(clip, model, language, prompt2) {
|
|
|
2841
3151
|
const run = spawnSync("whisper-cli", args, { encoding: "utf-8" });
|
|
2842
3152
|
if (run.error) throw new Error("whisper-cli not found \u2014 install whisper.cpp (brew install whisper-cpp) or use --from / --provider openai");
|
|
2843
3153
|
if (run.status !== 0) throw new Error(`whisper-cli failed: ${run.stderr.slice(-400)}`);
|
|
2844
|
-
const j = JSON.parse(
|
|
3154
|
+
const j = JSON.parse(fs8.readFileSync(wav.replace(/\.wav$/, "") + ".json", "utf-8"));
|
|
2845
3155
|
const segs = j.transcription ?? j.segments ?? [];
|
|
2846
3156
|
const ms = (x) => typeof x === "number" ? x / 1e3 : toSec(String(x).replace(",", "."));
|
|
2847
3157
|
return segs.map((sg) => ({ t0: ms(sg.offsets?.from ?? sg.start), t1: ms(sg.offsets?.to ?? sg.end), text: String(sg.text ?? "").trim() })).filter((sg) => sg.text);
|
|
@@ -2851,7 +3161,7 @@ async function openaiTranscribe(clip, model, language, prompt2) {
|
|
|
2851
3161
|
if (!key) throw new Error("OPENAI_API_KEY is not set");
|
|
2852
3162
|
const base = process.env.OPENAI_BASE_URL || "https://api.openai.com/v1";
|
|
2853
3163
|
const form = new FormData();
|
|
2854
|
-
form.append("file", new Blob([
|
|
3164
|
+
form.append("file", new Blob([fs8.readFileSync(clip)]), path8.basename(clip));
|
|
2855
3165
|
form.append("model", model || "whisper-1");
|
|
2856
3166
|
form.append("response_format", "verbose_json");
|
|
2857
3167
|
form.append("timestamp_granularities[]", "segment");
|
|
@@ -2863,9 +3173,9 @@ async function openaiTranscribe(clip, model, language, prompt2) {
|
|
|
2863
3173
|
return (j.segments ?? []).map((sg) => ({ t0: Number(sg.start), t1: Number(sg.end), text: String(sg.text).trim() })).filter((sg) => sg.text);
|
|
2864
3174
|
}
|
|
2865
3175
|
async function transcribeFile(inputPath, opts = {}) {
|
|
2866
|
-
const input =
|
|
2867
|
-
if (!
|
|
2868
|
-
const { doc, bundle } = await
|
|
3176
|
+
const input = path8.resolve(inputPath);
|
|
3177
|
+
if (!fs8.existsSync(input)) throw new Error(`File not found: ${input}`);
|
|
3178
|
+
const { doc, bundle } = await loadDoc2(input);
|
|
2869
3179
|
const videos = findVideos(doc);
|
|
2870
3180
|
if (!videos.length) throw new Error("document has no video element");
|
|
2871
3181
|
let target = videos[0];
|
|
@@ -2881,40 +3191,40 @@ async function transcribeFile(inputPath, opts = {}) {
|
|
|
2881
3191
|
let segments;
|
|
2882
3192
|
let source;
|
|
2883
3193
|
if (opts.from) {
|
|
2884
|
-
segments = parseSubtitles(
|
|
2885
|
-
source = `${
|
|
3194
|
+
segments = parseSubtitles(fs8.readFileSync(path8.resolve(opts.from), "utf-8"), opts.from);
|
|
3195
|
+
source = `${path8.extname(opts.from).slice(1).toLowerCase() || "file"}-import`;
|
|
2886
3196
|
} else {
|
|
2887
3197
|
const provider = opts.provider ?? "whisper-cli";
|
|
2888
|
-
const clip = await clipToTempFile(doc, target.el,
|
|
3198
|
+
const clip = await clipToTempFile(doc, target.el, path8.dirname(input));
|
|
2889
3199
|
if (!clip) throw new Error("could not locate the clip bytes (no bundled asset, data URL, local path or http URL) \u2014 use --from to import subtitles instead");
|
|
2890
3200
|
segments = provider === "openai" ? await openaiTranscribe(clip, opts.model, opts.language, opts.prompt) : whisperCli(clip, opts.model, opts.language, opts.prompt);
|
|
2891
|
-
source = provider === "openai" ? `openai:${opts.model || "whisper-1"}` : `whisper-cli${opts.model ? ":" +
|
|
3201
|
+
source = provider === "openai" ? `openai:${opts.model || "whisper-1"}` : `whisper-cli${opts.model ? ":" + path8.basename(opts.model) : ""}`;
|
|
2892
3202
|
}
|
|
2893
3203
|
segments.sort((a, b) => a.t0 - b.t0);
|
|
2894
3204
|
const transcript = { ...opts.language ? { language: opts.language } : {}, source, created: (/* @__PURE__ */ new Date()).toISOString(), segments };
|
|
2895
3205
|
target.el.transcript = transcript;
|
|
2896
|
-
if (opts.chapters) target.el.chapters = parseChapters(
|
|
3206
|
+
if (opts.chapters) target.el.chapters = parseChapters(fs8.readFileSync(path8.resolve(opts.chapters), "utf-8"));
|
|
2897
3207
|
if (!target.el.id) target.el.id = `video-${target.index + 1}`;
|
|
2898
|
-
const output = opts.output ?
|
|
3208
|
+
const output = opts.output ? path8.resolve(opts.output) : input;
|
|
2899
3209
|
if (output.toLowerCase().endsWith(".jdfx") || bundle && !opts.output) {
|
|
2900
3210
|
const { bytes } = await packJdfx(doc);
|
|
2901
|
-
|
|
3211
|
+
fs8.writeFileSync(output, bytes);
|
|
2902
3212
|
} else {
|
|
2903
|
-
|
|
3213
|
+
fs8.writeFileSync(output, JSON.stringify(doc, null, 2));
|
|
2904
3214
|
}
|
|
2905
3215
|
const dur = segments.length ? segments[segments.length - 1].t1 : 0;
|
|
2906
|
-
console.log(`Transcribed: ${
|
|
3216
|
+
console.log(`Transcribed: ${path8.basename(input)} \u2192 element "${target.el.id}" (${segments.length} segments, ${Math.round(dur)} s, source ${source})`);
|
|
2907
3217
|
if (target.el.chapters) console.log(`Chapters: ${target.el.chapters.length}`);
|
|
2908
3218
|
console.log(`Output: ${output}
|
|
2909
|
-
Next: jdf chunk ${
|
|
3219
|
+
Next: jdf chunk ${path8.basename(output)} # transcript \u2192 time-windowed chunks with media.t0/t1`);
|
|
2910
3220
|
return transcript;
|
|
2911
3221
|
}
|
|
2912
3222
|
var CONFIG_NAME = "jdf.rag.json";
|
|
2913
3223
|
var OUT_DIR = ".jdf-rag";
|
|
2914
3224
|
function walk(dir, acc = []) {
|
|
2915
|
-
for (const ent of
|
|
3225
|
+
for (const ent of fs8.readdirSync(dir, { withFileTypes: true })) {
|
|
2916
3226
|
if (ent.name === "node_modules" || ent.name === OUT_DIR || ent.name.startsWith(".")) continue;
|
|
2917
|
-
const p =
|
|
3227
|
+
const p = path8.join(dir, ent.name);
|
|
2918
3228
|
if (ent.isDirectory()) walk(p, acc);
|
|
2919
3229
|
else if (/\.(jdf|jdfx)$/i.test(ent.name)) acc.push(p);
|
|
2920
3230
|
}
|
|
@@ -2922,10 +3232,10 @@ function walk(dir, acc = []) {
|
|
|
2922
3232
|
}
|
|
2923
3233
|
async function readDoc(file) {
|
|
2924
3234
|
if (file.toLowerCase().endsWith(".jdfx")) {
|
|
2925
|
-
const zip = await JSZip.loadAsync(
|
|
3235
|
+
const zip = await JSZip.loadAsync(fs8.readFileSync(file));
|
|
2926
3236
|
return JSON.parse(await zip.file(JDFX_DOCUMENT_PATH).async("string"));
|
|
2927
3237
|
}
|
|
2928
|
-
return JSON.parse(
|
|
3238
|
+
return JSON.parse(fs8.readFileSync(file, "utf-8"));
|
|
2929
3239
|
}
|
|
2930
3240
|
function videosIn(doc) {
|
|
2931
3241
|
const out = [];
|
|
@@ -2939,29 +3249,44 @@ function videosIn(doc) {
|
|
|
2939
3249
|
return out;
|
|
2940
3250
|
}
|
|
2941
3251
|
async function ragFolder(dirPath, cli = {}) {
|
|
2942
|
-
const dir =
|
|
2943
|
-
if (!
|
|
2944
|
-
const cfgPath =
|
|
2945
|
-
const cfg =
|
|
3252
|
+
const dir = path8.resolve(dirPath);
|
|
3253
|
+
if (!fs8.existsSync(dir) || !fs8.statSync(dir).isDirectory()) throw new Error(`Not a directory: ${dir}`);
|
|
3254
|
+
const cfgPath = path8.join(dir, CONFIG_NAME);
|
|
3255
|
+
const cfg = fs8.existsSync(cfgPath) ? JSON.parse(fs8.readFileSync(cfgPath, "utf-8")) : {};
|
|
2946
3256
|
const opts = { ...cfg, ...Object.fromEntries(Object.entries(cli).filter(([, v]) => v !== void 0)) };
|
|
2947
3257
|
const provider = opts.provider ?? "ollama";
|
|
2948
3258
|
const transcribe = opts.transcribe ?? "none";
|
|
2949
|
-
const outDir =
|
|
3259
|
+
const outDir = path8.resolve(opts.out ?? path8.join(dir, OUT_DIR));
|
|
3260
|
+
const ocr = opts.ocr ?? "none";
|
|
3261
|
+
const caption = opts.caption ?? "none";
|
|
2950
3262
|
const files = walk(dir);
|
|
2951
3263
|
console.log(`jdf rag: ${dir}
|
|
2952
|
-
files: ${files.length} (.jdf/.jdfx)${
|
|
3264
|
+
files: ${files.length} (.jdf/.jdfx)${fs8.existsSync(cfgPath) ? `
|
|
2953
3265
|
config: ${CONFIG_NAME}` : ""}
|
|
2954
3266
|
embeddings: ${opts.noEmbed ? "skipped (--no-embed)" : `${provider}${opts.model ? " / " + opts.model : ""}`}
|
|
2955
|
-
transcribe: ${transcribe}${opts.dryRun ? "\n DRY RUN \u2014 nothing written" : ""}
|
|
3267
|
+
transcribe: ${transcribe} ocr: ${ocr} caption: ${caption}${opts.dryRun ? "\n DRY RUN \u2014 nothing written" : ""}
|
|
2956
3268
|
`);
|
|
2957
3269
|
if (!files.length) {
|
|
2958
3270
|
console.log("Nothing to do.");
|
|
2959
3271
|
return;
|
|
2960
3272
|
}
|
|
2961
|
-
const manifest = {
|
|
3273
|
+
const manifest = {
|
|
3274
|
+
dir,
|
|
3275
|
+
created: (/* @__PURE__ */ new Date()).toISOString(),
|
|
3276
|
+
provider: opts.noEmbed ? null : provider,
|
|
3277
|
+
model: opts.model ?? null,
|
|
3278
|
+
strategy: opts.strategy ?? "section",
|
|
3279
|
+
transcribe,
|
|
3280
|
+
ocr,
|
|
3281
|
+
caption,
|
|
3282
|
+
files: [],
|
|
3283
|
+
totals: { files: files.length, chunks: 0, videoChunks: 0, videos: 0, transcribed: 0, untranscribed: 0, images: 0, described: 0, imagesWithoutText: 0 },
|
|
3284
|
+
/** Every media element that retrieval would still skip, by file — the thing to fix before shipping an index. */
|
|
3285
|
+
mediaWithoutText: []
|
|
3286
|
+
};
|
|
2962
3287
|
const indexLines = [];
|
|
2963
3288
|
for (const file of files) {
|
|
2964
|
-
const rel =
|
|
3289
|
+
const rel = path8.relative(dir, file);
|
|
2965
3290
|
const doc = await readDoc(file);
|
|
2966
3291
|
const vids = videosIn(doc);
|
|
2967
3292
|
let transcribedHere = 0;
|
|
@@ -2985,20 +3310,36 @@ async function ragFolder(dirPath, cli = {}) {
|
|
|
2985
3310
|
}
|
|
2986
3311
|
manifest.totals.videos += vids.length;
|
|
2987
3312
|
manifest.totals.transcribed += transcribedHere;
|
|
3313
|
+
const covBefore = mediaCoverage(doc);
|
|
3314
|
+
let describedHere = 0;
|
|
3315
|
+
if (covBefore.images.missing.length && (ocr !== "none" || caption !== "none") && !opts.dryRun) {
|
|
3316
|
+
try {
|
|
3317
|
+
await describeFile(file, { ocr, caption, captionModel: opts.captionModel, quiet: true });
|
|
3318
|
+
describedHere = covBefore.images.missing.length;
|
|
3319
|
+
} catch (e) {
|
|
3320
|
+
console.warn(` ! ${rel}: describe failed: ${e.message}`);
|
|
3321
|
+
}
|
|
3322
|
+
}
|
|
3323
|
+
const covAfter = transcribedHere || describedHere ? mediaCoverage(await readDoc(file)) : covBefore;
|
|
3324
|
+
manifest.totals.images += covAfter.images.total;
|
|
3325
|
+
manifest.totals.described += describedHere;
|
|
3326
|
+
manifest.totals.imagesWithoutText += covAfter.images.missing.length;
|
|
3327
|
+
for (const m of covAfter.images.missing) manifest.mediaWithoutText.push({ file: rel, type: "image", ...m });
|
|
3328
|
+
for (const m of covAfter.videos.missing) manifest.mediaWithoutText.push({ file: rel, type: "video", ...m });
|
|
2988
3329
|
if (opts.dryRun) {
|
|
2989
3330
|
manifest.files.push({ file: rel, videos: vids.length, wouldTranscribe: transcribedHere });
|
|
2990
3331
|
continue;
|
|
2991
3332
|
}
|
|
2992
3333
|
const chunkOpts = { strategy: opts.strategy, maxTokens: opts.maxTokens, transcriptWindowSec: opts.transcriptWindowSec };
|
|
2993
|
-
const chunkOut =
|
|
2994
|
-
|
|
3334
|
+
const chunkOut = path8.join(outDir, "chunks", rel.replace(/\.(jdf|jdfx)$/i, ".chunks.jsonl"));
|
|
3335
|
+
fs8.mkdirSync(path8.dirname(chunkOut), { recursive: true });
|
|
2995
3336
|
let chunks;
|
|
2996
3337
|
if (opts.noEmbed) {
|
|
2997
3338
|
chunks = await chunkFile(file, { ...chunkOpts, format: "jsonl", output: chunkOut });
|
|
2998
3339
|
} else {
|
|
2999
3340
|
const side = await embedFile(file, { ...chunkOpts, provider, model: opts.model, incremental: true });
|
|
3000
3341
|
chunks = await chunkFile(file, { ...chunkOpts, format: "jsonl", output: chunkOut });
|
|
3001
|
-
manifest.files.push({ file: rel, chunks: chunks.length, vectors: Object.keys(side.vectors).length, sidecar:
|
|
3342
|
+
manifest.files.push({ file: rel, chunks: chunks.length, vectors: Object.keys(side.vectors).length, sidecar: path8.relative(dir, file.replace(/\.(jdf|jdfx)$/i, ".embeddings.json")), videos: vids.length, transcribed: transcribedHere });
|
|
3002
3343
|
}
|
|
3003
3344
|
if (opts.noEmbed) manifest.files.push({ file: rel, chunks: chunks.length, videos: vids.length, transcribed: transcribedHere });
|
|
3004
3345
|
for (const c of chunks) {
|
|
@@ -3008,15 +3349,27 @@ async function ragFolder(dirPath, cli = {}) {
|
|
|
3008
3349
|
}
|
|
3009
3350
|
}
|
|
3010
3351
|
if (!opts.dryRun) {
|
|
3011
|
-
|
|
3012
|
-
|
|
3013
|
-
|
|
3352
|
+
fs8.mkdirSync(outDir, { recursive: true });
|
|
3353
|
+
fs8.writeFileSync(path8.join(outDir, "index.jsonl"), indexLines.join("\n") + (indexLines.length ? "\n" : ""));
|
|
3354
|
+
fs8.writeFileSync(path8.join(outDir, "manifest.json"), JSON.stringify(manifest, null, 2) + "\n");
|
|
3014
3355
|
}
|
|
3015
3356
|
const t = manifest.totals;
|
|
3016
3357
|
console.log(`
|
|
3017
|
-
Done. ${t.files} files \u2192 ${t.chunks} chunks (${t.videoChunks} from video transcripts)
|
|
3018
|
-
|
|
3019
|
-
|
|
3358
|
+
Done. ${t.files} files \u2192 ${t.chunks} chunks (${t.videoChunks} from video transcripts).`);
|
|
3359
|
+
console.log(`Media coverage: videos ${t.videos - t.untranscribed}/${t.videos} with transcript (transcribed now ${t.transcribed}), images ${t.images - t.imagesWithoutText}/${t.images} with caption/OCR (described now ${t.described}).`);
|
|
3360
|
+
if (manifest.mediaWithoutText.length) {
|
|
3361
|
+
console.log(`
|
|
3362
|
+
! ${manifest.mediaWithoutText.length} media element(s) still have NO text \u2014 retrieval will skip them:`);
|
|
3363
|
+
for (const m of manifest.mediaWithoutText.slice(0, 12)) console.log(` ${m.file} \xB7 ${m.type} ${m.id ?? ""} (page ${m.page})${m.title ? ` "${m.title}"` : m.alt ? ` alt="${m.alt}"` : ""}`);
|
|
3364
|
+
if (manifest.mediaWithoutText.length > 12) console.log(` \u2026 ${manifest.mediaWithoutText.length - 12} more in manifest.json`);
|
|
3365
|
+
console.log(` fix: jdf rag <dir> --transcribe whisper-cli|openai --ocr tesseract --caption ollama (or jdf transcribe / jdf describe per file)`);
|
|
3366
|
+
if (opts.strict) {
|
|
3367
|
+
console.error(`--strict: failing because media without text remains.`);
|
|
3368
|
+
process.exitCode = 1;
|
|
3369
|
+
}
|
|
3370
|
+
}
|
|
3371
|
+
if (!opts.dryRun) console.log(`Index: ${path8.join(outDir, "index.jsonl")}
|
|
3372
|
+
Report: ${path8.join(outDir, "manifest.json")}${opts.noEmbed ? "" : `
|
|
3020
3373
|
Vectors: one <file>.embeddings.json next to each document (incremental \u2014 re-run any time)`}`);
|
|
3021
3374
|
}
|
|
3022
3375
|
|
|
@@ -3032,8 +3385,11 @@ The CLI exists for these workflows:
|
|
|
3032
3385
|
\u2022 JDF \u2192 vectors embed those chunks, incrementally, for a vector store.
|
|
3033
3386
|
\u2022 video \u2192 text attach a time-stamped transcript to a video element so
|
|
3034
3387
|
RAG retrieves "video at 02:13", not just "a video".
|
|
3388
|
+
\u2022 image \u2192 text OCR + a vision caption for every image so charts and
|
|
3389
|
+
scanned pages are retrievable, not skipped.
|
|
3035
3390
|
\u2022 folder \u2192 index one command over a directory of .jdf/.jdfx: transcribe,
|
|
3036
|
-
chunk, embed incrementally, write .jdf-rag/index.jsonl.
|
|
3391
|
+
describe, chunk, embed incrementally, write .jdf-rag/index.jsonl.
|
|
3392
|
+
Reports media coverage; --strict fails when anything has no text.
|
|
3037
3393
|
|
|
3038
3394
|
Usage:
|
|
3039
3395
|
jdf validate <file.jdf>
|
|
@@ -3041,7 +3397,8 @@ Usage:
|
|
|
3041
3397
|
jdf chunk <file.{jdf,jdfx}> [--strategy section|element|fixed] [--format jsonl|json|inline] [--max-tokens N] [-o out]
|
|
3042
3398
|
jdf embed <file.{jdf,jdfx}> [--provider ollama|openai] [--model NAME] [--strategy \u2026] [--incremental] [-o out]
|
|
3043
3399
|
jdf transcribe <file.{jdf,jdfx}> [--from subs.srt|.vtt|.json] [--provider whisper-cli|openai] [--element ID] [--chapters FILE] [-o out]
|
|
3044
|
-
jdf
|
|
3400
|
+
jdf describe <file.{jdf,jdfx}> [--ocr tesseract|openai|none] [--caption ollama|openai|none] [--caption-model M] [--ocr-language eng] [--element ID] [--force] [-o out]
|
|
3401
|
+
jdf rag <dir> [--provider ollama|openai] [--model NAME] [--transcribe none|whisper-cli|openai] [--ocr none|tesseract|openai] [--caption none|ollama|openai] [--strict] [--no-embed] [--dry-run] [--out DIR]
|
|
3045
3402
|
jdf --help
|
|
3046
3403
|
|
|
3047
3404
|
Commands:
|
|
@@ -3050,7 +3407,8 @@ Commands:
|
|
|
3050
3407
|
chunk Split a JDF document into retrieval-ready chunks (offline, deterministic)
|
|
3051
3408
|
embed Compute embeddings for the chunks (local via Ollama by default)
|
|
3052
3409
|
transcribe Store time-stamped text on a video element (import SRT/VTT/JSON, or run Whisper)
|
|
3053
|
-
|
|
3410
|
+
describe Give images text: OCR blocks (tesseract.js, local) + a caption (Ollama vision model, local)
|
|
3411
|
+
rag Make a whole folder retrieval-ready (finds .jdf/.jdfx, transcribes, describes, chunks, embeds, indexes)
|
|
3054
3412
|
|
|
3055
3413
|
Flags:
|
|
3056
3414
|
-o, --output <path> Explicit output path
|
|
@@ -3077,6 +3435,12 @@ Flags:
|
|
|
3077
3435
|
--prompt <text> transcribe: Whisper vocabulary hint (names, acronyms) \u2014 not an instruction
|
|
3078
3436
|
--window <sec> chunk/embed/rag: transcript window per video chunk (default 45)
|
|
3079
3437
|
--transcribe <p> rag: none (default) | whisper-cli | openai \u2014 for videos that have no transcript yet
|
|
3438
|
+
--ocr <p> describe/rag/convert: tesseract (local WASM) | openai | none
|
|
3439
|
+
--caption <p> describe/rag: ollama (local vision model, default moondream) | openai | none
|
|
3440
|
+
--caption-model describe/rag: vision model name (ollama: qwen2.5vl:3b, llava\u2026; openai: gpt-4o-mini\u2026)
|
|
3441
|
+
--ocr-language describe: tesseract language(s), e.g. eng, tur, eng+tur (default eng)
|
|
3442
|
+
--force describe: redo images that already have text
|
|
3443
|
+
--strict rag: exit 1 if any image/video is still without text after the run
|
|
3080
3444
|
--no-embed rag: chunk + index only
|
|
3081
3445
|
--dry-run rag: list what would happen, write nothing
|
|
3082
3446
|
--out <dir> rag: index folder (default <dir>/.jdf-rag)
|
|
@@ -3096,9 +3460,11 @@ Examples:
|
|
|
3096
3460
|
jdf embed report.jdf --provider openai --incremental
|
|
3097
3461
|
jdf transcribe talk.jdfx --from talk.srt --chapters chapters.txt # then: jdf chunk talk.jdfx
|
|
3098
3462
|
jdf transcribe talk.jdfx --provider openai --language en --prompt "JDF, jdfx, Ollama"
|
|
3099
|
-
jdf
|
|
3463
|
+
jdf describe report.jdfx # OCR (tesseract) + caption (Ollama qwen2.5vl), local
|
|
3464
|
+
jdf convert scan.pdf --ocr tesseract # scanned pages get OCR text instead of silence
|
|
3465
|
+
jdf rag ./knowledge-base --transcribe openai --ocr tesseract --caption ollama --strict
|
|
3100
3466
|
`;
|
|
3101
|
-
var BOOLEAN_FLAGS = /* @__PURE__ */ new Set(["help", "h", "json", "verbose", "skip-validate", "incremental", "no-auto-start", "drop-invisible-text", "no-embed", "dry-run"]);
|
|
3467
|
+
var BOOLEAN_FLAGS = /* @__PURE__ */ new Set(["help", "h", "json", "verbose", "skip-validate", "incremental", "no-auto-start", "drop-invisible-text", "no-embed", "dry-run", "force", "strict"]);
|
|
3102
3468
|
function parseArgs(argv) {
|
|
3103
3469
|
const positional = [];
|
|
3104
3470
|
const flags = {};
|
|
@@ -3181,7 +3547,8 @@ async function main() {
|
|
|
3181
3547
|
await importPdf(input, output, {
|
|
3182
3548
|
forceJson,
|
|
3183
3549
|
password: typeof flags.password === "string" ? flags.password : void 0,
|
|
3184
|
-
dropInvisibleText: flags["drop-invisible-text"] === true
|
|
3550
|
+
dropInvisibleText: flags["drop-invisible-text"] === true,
|
|
3551
|
+
ocr: typeof flags.ocr === "string" ? flags.ocr : void 0
|
|
3185
3552
|
});
|
|
3186
3553
|
process.exit(0);
|
|
3187
3554
|
} else if (lower.endsWith(".json")) {
|
|
@@ -3225,6 +3592,23 @@ async function main() {
|
|
|
3225
3592
|
});
|
|
3226
3593
|
process.exit(0);
|
|
3227
3594
|
}
|
|
3595
|
+
case "describe": {
|
|
3596
|
+
const input = positional[0];
|
|
3597
|
+
if (!input) {
|
|
3598
|
+
console.error("Usage: jdf describe <file.{jdf,jdfx}> [--ocr tesseract|openai|none] [--caption ollama|openai|none] [--caption-model M] [--ocr-language eng] [--element id|n] [--force] [-o out]");
|
|
3599
|
+
process.exit(1);
|
|
3600
|
+
}
|
|
3601
|
+
await describeFile(input, {
|
|
3602
|
+
ocr: typeof flags.ocr === "string" ? flags.ocr : void 0,
|
|
3603
|
+
caption: typeof flags.caption === "string" ? flags.caption : void 0,
|
|
3604
|
+
captionModel: typeof flags["caption-model"] === "string" ? flags["caption-model"] : void 0,
|
|
3605
|
+
ocrLanguage: typeof flags["ocr-language"] === "string" ? flags["ocr-language"] : void 0,
|
|
3606
|
+
element: typeof flags.element === "string" ? flags.element : void 0,
|
|
3607
|
+
force: flags.force === true,
|
|
3608
|
+
output: typeof flags.output === "string" ? flags.output : void 0
|
|
3609
|
+
});
|
|
3610
|
+
process.exit(process.exitCode ?? 0);
|
|
3611
|
+
}
|
|
3228
3612
|
case "rag": {
|
|
3229
3613
|
const input = positional[0];
|
|
3230
3614
|
if (!input) {
|
|
@@ -3241,11 +3625,15 @@ async function main() {
|
|
|
3241
3625
|
transcribeModel: typeof flags["transcribe-model"] === "string" ? flags["transcribe-model"] : void 0,
|
|
3242
3626
|
language: typeof flags.language === "string" ? flags.language : void 0,
|
|
3243
3627
|
prompt: typeof flags.prompt === "string" ? flags.prompt : void 0,
|
|
3628
|
+
ocr: typeof flags.ocr === "string" ? flags.ocr : void 0,
|
|
3629
|
+
caption: typeof flags.caption === "string" ? flags.caption : void 0,
|
|
3630
|
+
captionModel: typeof flags["caption-model"] === "string" ? flags["caption-model"] : void 0,
|
|
3631
|
+
strict: flags.strict === true,
|
|
3244
3632
|
noEmbed: flags["no-embed"] === true,
|
|
3245
3633
|
dryRun: flags["dry-run"] === true,
|
|
3246
3634
|
out: typeof flags.out === "string" ? flags.out : void 0
|
|
3247
3635
|
});
|
|
3248
|
-
process.exit(0);
|
|
3636
|
+
process.exit(process.exitCode ?? 0);
|
|
3249
3637
|
}
|
|
3250
3638
|
case "embed": {
|
|
3251
3639
|
const input = positional[0];
|
package/dist/jdf-schema.json
CHANGED
|
@@ -343,7 +343,17 @@
|
|
|
343
343
|
"anyOf": [{ "required": ["resource"] }, { "required": ["src"] }],
|
|
344
344
|
"properties": {
|
|
345
345
|
"type": { "const": "image" },
|
|
346
|
+
"id": { "type": "string" },
|
|
346
347
|
"resource": { "type": "string" }, "src": { "type": "string" }, "alt": { "type": "string" },
|
|
348
|
+
"caption": { "type": "string" }, "captionSource": { "type": "string" },
|
|
349
|
+
"ocr": {
|
|
350
|
+
"type": "object",
|
|
351
|
+
"required": ["blocks"],
|
|
352
|
+
"properties": {
|
|
353
|
+
"language": { "type": "string" }, "source": { "type": "string" }, "created": { "type": "string" },
|
|
354
|
+
"blocks": { "type": "array", "items": { "type": "object", "required": ["text"], "properties": { "text": { "type": "string" }, "confidence": { "type": "number" }, "bbox": { "type": "object", "properties": { "x": { "type": "number" }, "y": { "type": "number" }, "w": { "type": "number" }, "h": { "type": "number" } } } } } }
|
|
355
|
+
}
|
|
356
|
+
},
|
|
347
357
|
"position": { "$ref": "#/definitions/Position" },
|
|
348
358
|
"width": { "type": "number" }, "height": { "type": "number" },
|
|
349
359
|
"fit": { "type": "string", "enum": ["contain","cover","fill","none"] },
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@uurtech/jdf-cli",
|
|
3
|
-
"version": "0.2.
|
|
3
|
+
"version": "0.2.2",
|
|
4
4
|
"description": "Command-line tool for the JDF (JSON Document Format) — validate and convert documents.",
|
|
5
5
|
"license": "MIT",
|
|
6
6
|
"author": "Ugur Kazdal",
|
|
@@ -52,7 +52,8 @@
|
|
|
52
52
|
"ajv": "^8.17.1",
|
|
53
53
|
"ajv-formats": "^3.0.1",
|
|
54
54
|
"jszip": "^3.10.1",
|
|
55
|
-
"pdfjs-dist": "^4.10.38"
|
|
55
|
+
"pdfjs-dist": "^4.10.38",
|
|
56
|
+
"tesseract.js": "^7.0.0"
|
|
56
57
|
},
|
|
57
58
|
"devDependencies": {
|
|
58
59
|
"@jdf/core": "workspace:*",
|