@uurtech/jdf-cli 0.2.1 → 0.2.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.js +668 -189
- package/dist/jdf-schema.json +11 -0
- package/package.json +3 -2
package/dist/index.js
CHANGED
|
@@ -1,12 +1,13 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
|
-
import
|
|
3
|
-
import
|
|
2
|
+
import fs8 from 'fs';
|
|
3
|
+
import path8 from 'path';
|
|
4
4
|
import { fileURLToPath } from 'url';
|
|
5
5
|
import Ajv from 'ajv';
|
|
6
6
|
import addFormats from 'ajv-formats';
|
|
7
7
|
import JSZip from 'jszip';
|
|
8
8
|
import crypto, { createHash } from 'crypto';
|
|
9
9
|
import { readFile } from 'fs/promises';
|
|
10
|
+
import os from 'os';
|
|
10
11
|
import { execFileSync, spawnSync } from 'child_process';
|
|
11
12
|
|
|
12
13
|
var __require = /* @__PURE__ */ ((x) => typeof require !== "undefined" ? require : typeof Proxy !== "undefined" ? new Proxy(x, {
|
|
@@ -23,17 +24,17 @@ var JDFX_MANIFEST_PATH = "manifest.json";
|
|
|
23
24
|
var JDFX_ASSET_DIR = "assets";
|
|
24
25
|
|
|
25
26
|
// src/commands/validate.ts
|
|
26
|
-
var __dirname$1 =
|
|
27
|
+
var __dirname$1 = path8.dirname(fileURLToPath(import.meta.url));
|
|
27
28
|
function resolveSchemaPath() {
|
|
28
|
-
const bundled =
|
|
29
|
-
if (
|
|
30
|
-
const dev =
|
|
29
|
+
const bundled = path8.resolve(__dirname$1, "jdf-schema.json");
|
|
30
|
+
if (fs8.existsSync(bundled)) return bundled;
|
|
31
|
+
const dev = path8.resolve(__dirname$1, "../../../../spec/jdf-schema.json");
|
|
31
32
|
return dev;
|
|
32
33
|
}
|
|
33
34
|
var SCHEMA_PATH = resolveSchemaPath();
|
|
34
35
|
async function loadDocument(filePath) {
|
|
35
36
|
if (filePath.toLowerCase().endsWith(".jdfx")) {
|
|
36
|
-
const zip = await JSZip.loadAsync(
|
|
37
|
+
const zip = await JSZip.loadAsync(fs8.readFileSync(filePath));
|
|
37
38
|
const docFile = zip.file(JDFX_DOCUMENT_PATH);
|
|
38
39
|
if (!docFile) {
|
|
39
40
|
console.error(`\u2717 Bundle missing ${JDFX_DOCUMENT_PATH}`);
|
|
@@ -58,11 +59,11 @@ async function loadDocument(filePath) {
|
|
|
58
59
|
}
|
|
59
60
|
return { doc, bundle: { manifest, assetCount } };
|
|
60
61
|
}
|
|
61
|
-
return { doc: JSON.parse(
|
|
62
|
+
return { doc: JSON.parse(fs8.readFileSync(filePath, "utf-8")) };
|
|
62
63
|
}
|
|
63
64
|
async function validate(file) {
|
|
64
|
-
const filePath =
|
|
65
|
-
if (!
|
|
65
|
+
const filePath = path8.resolve(file);
|
|
66
|
+
if (!fs8.existsSync(filePath)) {
|
|
66
67
|
console.error(`File not found: ${filePath}`);
|
|
67
68
|
return false;
|
|
68
69
|
}
|
|
@@ -75,11 +76,11 @@ async function validate(file) {
|
|
|
75
76
|
}
|
|
76
77
|
if (!loaded) return false;
|
|
77
78
|
const { doc, bundle } = loaded;
|
|
78
|
-
if (!
|
|
79
|
+
if (!fs8.existsSync(SCHEMA_PATH)) {
|
|
79
80
|
console.error(`Schema not found at ${SCHEMA_PATH}`);
|
|
80
81
|
return false;
|
|
81
82
|
}
|
|
82
|
-
const schema = JSON.parse(
|
|
83
|
+
const schema = JSON.parse(fs8.readFileSync(SCHEMA_PATH, "utf-8"));
|
|
83
84
|
const ajv = new Ajv({ allErrors: true, strict: false });
|
|
84
85
|
addFormats(ajv);
|
|
85
86
|
const validateFn = ajv.compile(schema);
|
|
@@ -88,7 +89,7 @@ async function validate(file) {
|
|
|
88
89
|
const d = doc;
|
|
89
90
|
const pageCount = Array.isArray(d.pages) ? d.pages.length : 0;
|
|
90
91
|
const elCount = Array.isArray(d.pages) ? d.pages.reduce((acc, p) => acc + (Array.isArray(p?.elements) ? p.elements.length : 0), 0) : 0;
|
|
91
|
-
console.log(`\u2713 Valid: ${
|
|
92
|
+
console.log(`\u2713 Valid: ${path8.basename(filePath)}`);
|
|
92
93
|
console.log(` Format: ${d.$jdf}${bundle ? " (jdfx bundle)" : ""}`);
|
|
93
94
|
console.log(` Title: ${d.meta?.title}`);
|
|
94
95
|
console.log(` Pages: ${pageCount}`);
|
|
@@ -101,7 +102,7 @@ async function validate(file) {
|
|
|
101
102
|
}
|
|
102
103
|
return true;
|
|
103
104
|
}
|
|
104
|
-
console.error(`\u2717 Invalid: ${
|
|
105
|
+
console.error(`\u2717 Invalid: ${path8.basename(filePath)}`);
|
|
105
106
|
for (const err of validateFn.errors || []) {
|
|
106
107
|
const loc = err.instancePath || "(root)";
|
|
107
108
|
console.error(` ${loc} \u2014 ${err.message}`);
|
|
@@ -304,13 +305,13 @@ function stripInline(text) {
|
|
|
304
305
|
return parseInline(text).map((r) => r.text).join("");
|
|
305
306
|
}
|
|
306
307
|
async function importMarkdown(inputPath, outputPath) {
|
|
307
|
-
const input =
|
|
308
|
+
const input = path8.resolve(inputPath);
|
|
308
309
|
console.log(`Importing: ${input}`);
|
|
309
|
-
const content =
|
|
310
|
-
const doc = convertMarkdownToJdf(content,
|
|
310
|
+
const content = fs8.readFileSync(input, "utf-8");
|
|
311
|
+
const doc = convertMarkdownToJdf(content, path8.basename(input, path8.extname(input)), path8.dirname(input));
|
|
311
312
|
let output;
|
|
312
313
|
if (outputPath) {
|
|
313
|
-
output =
|
|
314
|
+
output = path8.resolve(outputPath);
|
|
314
315
|
} else {
|
|
315
316
|
const stem = input.replace(/\.(md|markdown)$/i, "");
|
|
316
317
|
output = stem + (shouldUseJdfx(doc) ? ".jdfx" : ".jdf");
|
|
@@ -318,11 +319,11 @@ async function importMarkdown(inputPath, outputPath) {
|
|
|
318
319
|
console.log(`Output: ${output}`);
|
|
319
320
|
if (output.toLowerCase().endsWith(".jdfx")) {
|
|
320
321
|
const { bytes, manifest } = await packJdfx(doc);
|
|
321
|
-
|
|
322
|
+
fs8.writeFileSync(output, bytes);
|
|
322
323
|
console.log(`
|
|
323
324
|
Done! Created ${doc.pages.length} page(s), ${manifest.assets.length} asset(s) bundled`);
|
|
324
325
|
} else {
|
|
325
|
-
|
|
326
|
+
fs8.writeFileSync(output, JSON.stringify(doc, null, 2));
|
|
326
327
|
console.log(`
|
|
327
328
|
Done! Created ${doc.pages.length} page(s)`);
|
|
328
329
|
}
|
|
@@ -339,10 +340,10 @@ var MIME_BY_EXT2 = {
|
|
|
339
340
|
};
|
|
340
341
|
function resolveImageSrc(src, baseDir) {
|
|
341
342
|
if (/^(https?:|data:|file:)/i.test(src)) return src;
|
|
342
|
-
const abs =
|
|
343
|
+
const abs = path8.isAbsolute(src) ? src : path8.resolve(baseDir, src);
|
|
343
344
|
try {
|
|
344
|
-
const bytes =
|
|
345
|
-
const ext =
|
|
345
|
+
const bytes = fs8.readFileSync(abs);
|
|
346
|
+
const ext = path8.extname(abs).slice(1).toLowerCase();
|
|
346
347
|
const mime = MIME_BY_EXT2[ext] || "application/octet-stream";
|
|
347
348
|
return `data:${mime};base64,${bytes.toString("base64")}`;
|
|
348
349
|
} catch {
|
|
@@ -711,7 +712,7 @@ function columnBands(rows) {
|
|
|
711
712
|
return bands.map(({ x0, x1 }) => ({ x0, x1 }));
|
|
712
713
|
}
|
|
713
714
|
var numeric = (s) => /^[\s$€£¥+\-−–]*[\d.,]+\s*(%|ms|s|k|m|b|M|K|B|x|×)?\s*(\/\w+)?$/i.test(s.trim()) || /^[+\-−]?\d/.test(s.trim()) && /\d$/.test(s.trim().replace(/[%)]$/, ""));
|
|
714
|
-
function detectTables(runs, shapes, pageWidthMm) {
|
|
715
|
+
function detectTables(runs, shapes, pageWidthMm, gutters = []) {
|
|
715
716
|
const out = [];
|
|
716
717
|
const used = /* @__PURE__ */ new Set();
|
|
717
718
|
const rows = groupRows(runs, () => false);
|
|
@@ -760,6 +761,14 @@ function detectTables(runs, shapes, pageWidthMm) {
|
|
|
760
761
|
i++;
|
|
761
762
|
continue;
|
|
762
763
|
}
|
|
764
|
+
if (gutters.length && !hasLattice) {
|
|
765
|
+
const b = best.bands;
|
|
766
|
+
const straddles = b.some((band, k) => k < b.length - 1 && gutters.some((g) => band.x1 <= g.x1 + 1 && b[k + 1].x0 >= g.x0 - 1) && band.x1 - band.x0 > 30 && b[k + 1].x1 - b[k + 1].x0 > 30);
|
|
767
|
+
if (straddles) {
|
|
768
|
+
i++;
|
|
769
|
+
continue;
|
|
770
|
+
}
|
|
771
|
+
}
|
|
763
772
|
const bands = best.bands;
|
|
764
773
|
const cellText = (row, b) => row.cells.filter((c) => c.x0 < bands[b].x1 - 0.2 && c.x1 > bands[b].x0 + 0.2).map((c) => c.run.text.trim()).join(" ").trim();
|
|
765
774
|
const grid = [];
|
|
@@ -847,8 +856,165 @@ function detectTables(runs, shapes, pageWidthMm) {
|
|
|
847
856
|
return out;
|
|
848
857
|
}
|
|
849
858
|
|
|
850
|
-
// ../../packages/jdf-pdf-import/src/
|
|
859
|
+
// ../../packages/jdf-pdf-import/src/columns.ts
|
|
860
|
+
function detectGutters(lines, bodyFontSize, pageWmm) {
|
|
861
|
+
const body = lines.filter((l) => l.text.trim().length > 0 && Math.abs(l.fontSize - bodyFontSize) <= 1.5 && l.width < pageWmm * 0.6 && l.width > 0);
|
|
862
|
+
if (body.length < 12) return [];
|
|
863
|
+
const minX = Math.min(...body.map((l) => l.x));
|
|
864
|
+
const maxX = Math.max(...body.map((l) => l.x + l.width));
|
|
865
|
+
const span = maxX - minX;
|
|
866
|
+
if (span < pageWmm * 0.4) return [];
|
|
867
|
+
const BIN = 0.5;
|
|
868
|
+
const n = Math.ceil(span / BIN) + 1;
|
|
869
|
+
const cov = new Array(n).fill(0);
|
|
870
|
+
for (const l of body) {
|
|
871
|
+
const a = Math.max(0, Math.floor((l.x - minX) / BIN)), b = Math.min(n, Math.ceil((l.x + l.width - minX) / BIN));
|
|
872
|
+
for (let i2 = a; i2 < b; i2++) cov[i2]++;
|
|
873
|
+
}
|
|
874
|
+
const noise = Math.max(2, Math.round(body.length * 0.05));
|
|
875
|
+
const gutters = [];
|
|
876
|
+
let i = 0;
|
|
877
|
+
while (i < n) {
|
|
878
|
+
if (cov[i] > noise) {
|
|
879
|
+
i++;
|
|
880
|
+
continue;
|
|
881
|
+
}
|
|
882
|
+
let j = i;
|
|
883
|
+
while (j < n && cov[j] <= noise) j++;
|
|
884
|
+
const g0 = minX + i * BIN, g1 = minX + j * BIN;
|
|
885
|
+
if (g1 - g0 >= 3 && g0 > minX + span * 0.15 && g1 < maxX - span * 0.15) gutters.push({ x0: g0, x1: g1 });
|
|
886
|
+
i = j;
|
|
887
|
+
}
|
|
888
|
+
const solid = gutters.every((g) => body.filter((l) => l.x + l.width <= g.x0 + 0.5).length >= 4 && body.filter((l) => l.x >= g.x1 - 0.5).length >= 4);
|
|
889
|
+
return solid ? gutters : [];
|
|
890
|
+
}
|
|
891
|
+
function orderByColumns(els, gutters, pageWmm) {
|
|
892
|
+
if (!gutters.length || els.length < 2) return els;
|
|
893
|
+
const bands = [];
|
|
894
|
+
let left = 0;
|
|
895
|
+
for (const g of gutters) {
|
|
896
|
+
bands.push({ x0: left, x1: g.x0 });
|
|
897
|
+
left = g.x1;
|
|
898
|
+
}
|
|
899
|
+
bands.push({ x0: left, x1: pageWmm });
|
|
900
|
+
const colOf = (e) => {
|
|
901
|
+
const x = e.position?.x ?? 0;
|
|
902
|
+
const k = bands.findIndex((b) => x >= b.x0 && x < b.x1);
|
|
903
|
+
if (k >= 0) return k;
|
|
904
|
+
let best = 0, d = Infinity;
|
|
905
|
+
bands.forEach((b, i) => {
|
|
906
|
+
const dd = Math.min(Math.abs(x - b.x0), Math.abs(x - b.x1));
|
|
907
|
+
if (dd < d) {
|
|
908
|
+
d = dd;
|
|
909
|
+
best = i;
|
|
910
|
+
}
|
|
911
|
+
});
|
|
912
|
+
return best;
|
|
913
|
+
};
|
|
914
|
+
const spans = (e) => {
|
|
915
|
+
const c = colOf(e);
|
|
916
|
+
if (c >= bands.length - 1) return false;
|
|
917
|
+
const right = (e.position?.x ?? 0) + (e.width ?? 0);
|
|
918
|
+
const next = bands[c + 1];
|
|
919
|
+
return right > (next.x0 + next.x1) / 2;
|
|
920
|
+
};
|
|
921
|
+
const sorted = els.slice().sort((a, b) => (a.position?.y ?? 0) - (b.position?.y ?? 0) || (a.position?.x ?? 0) - (b.position?.x ?? 0));
|
|
922
|
+
const out = [];
|
|
923
|
+
let band = [];
|
|
924
|
+
const flush = () => {
|
|
925
|
+
for (let c = 0; c < bands.length; c++) for (const e of band) if (colOf(e) === c) out.push(e);
|
|
926
|
+
band = [];
|
|
927
|
+
};
|
|
928
|
+
for (const e of sorted) {
|
|
929
|
+
if (spans(e)) {
|
|
930
|
+
flush();
|
|
931
|
+
out.push(e);
|
|
932
|
+
} else band.push(e);
|
|
933
|
+
}
|
|
934
|
+
flush();
|
|
935
|
+
return out;
|
|
936
|
+
}
|
|
937
|
+
|
|
938
|
+
// ../../packages/jdf-pdf-import/src/paragraphs.ts
|
|
851
939
|
var PT_TO_MM2 = 0.352778;
|
|
940
|
+
var LIST_START = /^\s*(?:[•◦▪●■\-–—*]\s|\(?\d{1,3}[.)]\s|\(?[a-zA-Z][.)]\s|[ivx]{1,4}[.)]\s)/;
|
|
941
|
+
var TERMINAL = /[.!?:;]["'”’)\]]*$/;
|
|
942
|
+
function textOf(e) {
|
|
943
|
+
if (e.type === "text") return String(e.content ?? "");
|
|
944
|
+
return (e.runs ?? []).map((r) => String(r.text ?? "")).join("");
|
|
945
|
+
}
|
|
946
|
+
function foldParagraphs(elements, meta, pageWmm) {
|
|
947
|
+
const foldable = (e) => !!meta.get(e) && !!e.position && (e.type === "text" && !e.heading || e.type === "richtext");
|
|
948
|
+
const out = [];
|
|
949
|
+
let i = 0;
|
|
950
|
+
while (i < elements.length) {
|
|
951
|
+
const first = elements[i];
|
|
952
|
+
if (!foldable(first)) {
|
|
953
|
+
out.push(first);
|
|
954
|
+
i++;
|
|
955
|
+
continue;
|
|
956
|
+
}
|
|
957
|
+
const para = [first];
|
|
958
|
+
let j = i + 1;
|
|
959
|
+
while (j < elements.length) {
|
|
960
|
+
const prev = para[para.length - 1], next = elements[j];
|
|
961
|
+
if (!foldable(next) || !continues(para, prev, next, meta)) break;
|
|
962
|
+
para.push(next);
|
|
963
|
+
j++;
|
|
964
|
+
}
|
|
965
|
+
out.push(para.length > 1 ? merge(para, meta, pageWmm) : first);
|
|
966
|
+
i = j;
|
|
967
|
+
}
|
|
968
|
+
return out;
|
|
969
|
+
}
|
|
970
|
+
function continues(para, prev, next, meta) {
|
|
971
|
+
const mp = meta.get(prev), mn = meta.get(next); meta.get(para[0]);
|
|
972
|
+
if (mp.face !== mn.face || Math.abs(mp.size - mn.size) >= 0.5) return false;
|
|
973
|
+
const lineH = mp.size * PT_TO_MM2;
|
|
974
|
+
const pitch = next.position.y - prev.position.y;
|
|
975
|
+
if (pitch < lineH * 0.9 || pitch > lineH * 1.75) return false;
|
|
976
|
+
const dx = next.position.x - (para.length === 1 ? prev.position.x : para[0].position.x);
|
|
977
|
+
if (para.length === 1 ? Math.abs(dx) > 6 : Math.abs(dx) > 1.2) return false;
|
|
978
|
+
if (para.length >= 2 && Math.abs(next.position.x - para[1].position.x) > 1.2) return false;
|
|
979
|
+
const nText = textOf(next), pText = textOf(prev).trimEnd();
|
|
980
|
+
if (LIST_START.test(nText) || !nText.trim()) return false;
|
|
981
|
+
const maxW = Math.max(mp.w, mn.w, ...para.map((e) => meta.get(e).w));
|
|
982
|
+
const full = mp.w >= maxW * 0.85;
|
|
983
|
+
const endsHyphen = /[-‐‑]$/.test(pText);
|
|
984
|
+
const softEnd = !TERMINAL.test(pText);
|
|
985
|
+
const startsLower = /^[a-z(\[]/.test(nText.trimStart());
|
|
986
|
+
return full || endsHyphen || softEnd && (startsLower || mp.w >= maxW * 0.6);
|
|
987
|
+
}
|
|
988
|
+
function merge(para, meta, pageWmm) {
|
|
989
|
+
const first = para[0], last = para[para.length - 1];
|
|
990
|
+
const m0 = meta.get(first);
|
|
991
|
+
const lineH = m0.size * PT_TO_MM2;
|
|
992
|
+
const x = Math.min(...para.map((e) => e.position.x));
|
|
993
|
+
const indent = first.position.x - x;
|
|
994
|
+
const right = Math.max(...para.map((e) => e.position.x + meta.get(e).w));
|
|
995
|
+
const pitch = (last.position.y - first.position.y) / (para.length - 1);
|
|
996
|
+
const width = Math.min(pageWmm - x, (right - x) * 1.2 + lineH * 0.4);
|
|
997
|
+
const joiner = () => "\n";
|
|
998
|
+
const style = { ...first.style ?? {}, lineHeight: Math.round(pitch / lineH * 100) / 100, ...indent > 0.5 ? { textIndent: Math.round(indent * 100) / 100 } : {} };
|
|
999
|
+
const allText = para.every((e) => e.type === "text" && !e.link);
|
|
1000
|
+
const base = { position: { x, y: first.position.y }, width: Math.round(width * 100) / 100, style };
|
|
1001
|
+
if (allText) {
|
|
1002
|
+
const content = para.map((e) => String(e.content ?? "").trim().replace(/[ \t]+/g, " ")).join(joiner());
|
|
1003
|
+
const { height: _hh2, ...firstRest } = first;
|
|
1004
|
+
return { ...firstRest, ...base, content };
|
|
1005
|
+
}
|
|
1006
|
+
const runs = [];
|
|
1007
|
+
for (const e of para) {
|
|
1008
|
+
const lineRuns = e.type === "text" ? [{ text: String(e.content ?? "").trim(), ...e.style?.fontWeight === "bold" ? { bold: true } : {}, ...e.style?.fontStyle === "italic" ? { italic: true } : {}, ...e.style?.color && e.style.color !== "#000000" ? { color: e.style.color } : {}, ...e.link ? { link: e.link } : {} }] : (e.runs ?? []).map((r) => ({ ...r }));
|
|
1009
|
+
if (runs.length) runs[runs.length - 1].text = `${String(runs[runs.length - 1].text).trimEnd()}${joiner()}`;
|
|
1010
|
+
runs.push(...lineRuns);
|
|
1011
|
+
}
|
|
1012
|
+
const { content: _c, heading: _h, link: _l, height: _hh, ...rest } = first;
|
|
1013
|
+
return { ...rest, ...base, type: "richtext", runs, style: { fontSize: style.fontSize, fontFamily: style.fontFamily, lineHeight: style.lineHeight, ...style.textIndent ? { textIndent: style.textIndent } : {}, ...style.opacity != null ? { opacity: style.opacity } : {} } };
|
|
1014
|
+
}
|
|
1015
|
+
|
|
1016
|
+
// ../../packages/jdf-pdf-import/src/core.ts
|
|
1017
|
+
var PT_TO_MM3 = 0.352778;
|
|
852
1018
|
function classifyFont(name) {
|
|
853
1019
|
const n = (name || "").toLowerCase();
|
|
854
1020
|
const bold = /bold|black|heavy|semibold|demibold|extrabold/.test(n);
|
|
@@ -958,10 +1124,10 @@ async function walkOps(page, OPS, viewport) {
|
|
|
958
1124
|
const minY = Math.min(...ys), maxY = Math.max(...ys);
|
|
959
1125
|
imagePositions.push({
|
|
960
1126
|
name,
|
|
961
|
-
x: minX *
|
|
962
|
-
y: minY *
|
|
963
|
-
w: (maxX - minX) *
|
|
964
|
-
h: (maxY - minY) *
|
|
1127
|
+
x: minX * PT_TO_MM3,
|
|
1128
|
+
y: minY * PT_TO_MM3,
|
|
1129
|
+
w: (maxX - minX) * PT_TO_MM3,
|
|
1130
|
+
h: (maxY - minY) * PT_TO_MM3,
|
|
965
1131
|
inline,
|
|
966
1132
|
maskFill
|
|
967
1133
|
});
|
|
@@ -977,13 +1143,13 @@ async function walkOps(page, OPS, viewport) {
|
|
|
977
1143
|
const br = toViewport(r.x + r.w, r.y);
|
|
978
1144
|
shapes.push({
|
|
979
1145
|
kind: "rect",
|
|
980
|
-
x: Math.min(tl.x, br.x) *
|
|
981
|
-
y: Math.min(tl.y, br.y) *
|
|
982
|
-
width: Math.abs(br.x - tl.x) *
|
|
983
|
-
height: Math.abs(br.y - tl.y) *
|
|
1146
|
+
x: Math.min(tl.x, br.x) * PT_TO_MM3,
|
|
1147
|
+
y: Math.min(tl.y, br.y) * PT_TO_MM3,
|
|
1148
|
+
width: Math.abs(br.x - tl.x) * PT_TO_MM3,
|
|
1149
|
+
height: Math.abs(br.y - tl.y) * PT_TO_MM3,
|
|
984
1150
|
fill: isFill ? gs.fill : void 0,
|
|
985
1151
|
stroke: isStroke ? gs.stroke : void 0,
|
|
986
|
-
strokeWidth: isStroke ? gs.lineWidth *
|
|
1152
|
+
strokeWidth: isStroke ? gs.lineWidth * PT_TO_MM3 : void 0,
|
|
987
1153
|
opacity: isFill ? gs.fillAlpha : gs.strokeAlpha
|
|
988
1154
|
});
|
|
989
1155
|
}
|
|
@@ -997,13 +1163,13 @@ async function walkOps(page, OPS, viewport) {
|
|
|
997
1163
|
const h = Math.abs(br.y - tl.y);
|
|
998
1164
|
shapes.push({
|
|
999
1165
|
kind: "rect",
|
|
1000
|
-
x: x *
|
|
1001
|
-
y: y *
|
|
1002
|
-
width: w *
|
|
1003
|
-
height: h *
|
|
1166
|
+
x: x * PT_TO_MM3,
|
|
1167
|
+
y: y * PT_TO_MM3,
|
|
1168
|
+
width: w * PT_TO_MM3,
|
|
1169
|
+
height: h * PT_TO_MM3,
|
|
1004
1170
|
fill: isFill ? gs.fill : void 0,
|
|
1005
1171
|
stroke: isStroke ? gs.stroke : void 0,
|
|
1006
|
-
strokeWidth: isStroke ? gs.lineWidth *
|
|
1172
|
+
strokeWidth: isStroke ? gs.lineWidth * PT_TO_MM3 : void 0,
|
|
1007
1173
|
opacity: isFill ? gs.fillAlpha : gs.strokeAlpha
|
|
1008
1174
|
});
|
|
1009
1175
|
} else if (pathSegments.length === 2 && pathSegments[0].type === "M" && pathSegments[1].type === "L") {
|
|
@@ -1015,35 +1181,35 @@ async function walkOps(page, OPS, viewport) {
|
|
|
1015
1181
|
const minY = Math.min(va.y, vb.y);
|
|
1016
1182
|
const maxX = Math.max(va.x, vb.x);
|
|
1017
1183
|
const maxY = Math.max(va.y, vb.y);
|
|
1018
|
-
const x1Local = (va.x - minX) *
|
|
1019
|
-
const y1Local = (va.y - minY) *
|
|
1020
|
-
const x2Local = (vb.x - minX) *
|
|
1021
|
-
const y2Local = (vb.y - minY) *
|
|
1022
|
-
const wLocal = Math.max(0.05, (maxX - minX) *
|
|
1023
|
-
const hLocal = Math.max(0.05, (maxY - minY) *
|
|
1184
|
+
const x1Local = (va.x - minX) * PT_TO_MM3;
|
|
1185
|
+
const y1Local = (va.y - minY) * PT_TO_MM3;
|
|
1186
|
+
const x2Local = (vb.x - minX) * PT_TO_MM3;
|
|
1187
|
+
const y2Local = (vb.y - minY) * PT_TO_MM3;
|
|
1188
|
+
const wLocal = Math.max(0.05, (maxX - minX) * PT_TO_MM3);
|
|
1189
|
+
const hLocal = Math.max(0.05, (maxY - minY) * PT_TO_MM3);
|
|
1024
1190
|
const dx = Math.abs(va.x - vb.x);
|
|
1025
1191
|
const dy = Math.abs(va.y - vb.y);
|
|
1026
1192
|
const axisAligned = dx < 0.5 || dy < 0.5;
|
|
1027
1193
|
if (axisAligned) {
|
|
1028
1194
|
shapes.push({
|
|
1029
1195
|
kind: "line",
|
|
1030
|
-
x: minX *
|
|
1031
|
-
y: minY *
|
|
1196
|
+
x: minX * PT_TO_MM3,
|
|
1197
|
+
y: minY * PT_TO_MM3,
|
|
1032
1198
|
width: wLocal,
|
|
1033
1199
|
height: hLocal,
|
|
1034
1200
|
stroke: isStroke ? gs.stroke : void 0,
|
|
1035
|
-
strokeWidth: isStroke ? gs.lineWidth *
|
|
1201
|
+
strokeWidth: isStroke ? gs.lineWidth * PT_TO_MM3 : void 0,
|
|
1036
1202
|
opacity: gs.strokeAlpha
|
|
1037
1203
|
});
|
|
1038
1204
|
} else {
|
|
1039
1205
|
shapes.push({
|
|
1040
1206
|
kind: "path",
|
|
1041
|
-
x: minX *
|
|
1042
|
-
y: minY *
|
|
1207
|
+
x: minX * PT_TO_MM3,
|
|
1208
|
+
y: minY * PT_TO_MM3,
|
|
1043
1209
|
width: wLocal,
|
|
1044
1210
|
height: hLocal,
|
|
1045
1211
|
stroke: isStroke ? gs.stroke : void 0,
|
|
1046
|
-
strokeWidth: isStroke ? gs.lineWidth *
|
|
1212
|
+
strokeWidth: isStroke ? gs.lineWidth * PT_TO_MM3 : void 0,
|
|
1047
1213
|
opacity: gs.strokeAlpha,
|
|
1048
1214
|
path: `M ${x1Local.toFixed(2)} ${y1Local.toFixed(2)} L ${x2Local.toFixed(2)} ${y2Local.toFixed(2)}`
|
|
1049
1215
|
});
|
|
@@ -1076,20 +1242,20 @@ async function walkOps(page, OPS, viewport) {
|
|
|
1076
1242
|
if (seg.type === "Z") return "Z";
|
|
1077
1243
|
const p = [];
|
|
1078
1244
|
for (let i = 0; i < seg.pts.length; i += 2) {
|
|
1079
|
-
p.push(((seg.pts[i] - minX) *
|
|
1080
|
-
p.push(((seg.pts[i + 1] - minY) *
|
|
1245
|
+
p.push(((seg.pts[i] - minX) * PT_TO_MM3).toFixed(2));
|
|
1246
|
+
p.push(((seg.pts[i + 1] - minY) * PT_TO_MM3).toFixed(2));
|
|
1081
1247
|
}
|
|
1082
1248
|
return `${seg.type} ${p.join(" ")}`;
|
|
1083
1249
|
}).join(" ");
|
|
1084
1250
|
shapes.push({
|
|
1085
1251
|
kind: "path",
|
|
1086
|
-
x: minX *
|
|
1087
|
-
y: minY *
|
|
1088
|
-
width: bw *
|
|
1089
|
-
height: bh *
|
|
1252
|
+
x: minX * PT_TO_MM3,
|
|
1253
|
+
y: minY * PT_TO_MM3,
|
|
1254
|
+
width: bw * PT_TO_MM3,
|
|
1255
|
+
height: bh * PT_TO_MM3,
|
|
1090
1256
|
fill: isFill ? gs.fill : void 0,
|
|
1091
1257
|
stroke: isStroke ? gs.stroke : void 0,
|
|
1092
|
-
strokeWidth: isStroke ? gs.lineWidth *
|
|
1258
|
+
strokeWidth: isStroke ? gs.lineWidth * PT_TO_MM3 : void 0,
|
|
1093
1259
|
opacity: isFill ? gs.fillAlpha : gs.strokeAlpha,
|
|
1094
1260
|
path: d
|
|
1095
1261
|
});
|
|
@@ -1430,10 +1596,10 @@ async function extractLinks(doc, page, viewport) {
|
|
|
1430
1596
|
const xMax = Math.max(c1.x, c2.x);
|
|
1431
1597
|
const yMax = Math.max(c1.y, c2.y);
|
|
1432
1598
|
const rectMm = {
|
|
1433
|
-
x: xMin *
|
|
1434
|
-
y: yMin *
|
|
1435
|
-
w: (xMax - xMin) *
|
|
1436
|
-
h: (yMax - yMin) *
|
|
1599
|
+
x: xMin * PT_TO_MM3,
|
|
1600
|
+
y: yMin * PT_TO_MM3,
|
|
1601
|
+
w: (xMax - xMin) * PT_TO_MM3,
|
|
1602
|
+
h: (yMax - yMin) * PT_TO_MM3
|
|
1437
1603
|
};
|
|
1438
1604
|
const url = a.url || a.unsafeUrl;
|
|
1439
1605
|
const destPage = url ? void 0 : await resolveDestPage(doc, a.dest);
|
|
@@ -1471,10 +1637,10 @@ async function extractFormWidgets(page, viewport) {
|
|
|
1471
1637
|
})).filter((o) => o.value !== "") : [];
|
|
1472
1638
|
out.push({
|
|
1473
1639
|
rectMm: {
|
|
1474
|
-
x: xMin *
|
|
1475
|
-
y: yMin *
|
|
1476
|
-
w: (xMax - xMin) *
|
|
1477
|
-
h: (yMax - yMin) *
|
|
1640
|
+
x: xMin * PT_TO_MM3,
|
|
1641
|
+
y: yMin * PT_TO_MM3,
|
|
1642
|
+
w: (xMax - xMin) * PT_TO_MM3,
|
|
1643
|
+
h: (yMax - yMin) * PT_TO_MM3
|
|
1478
1644
|
},
|
|
1479
1645
|
fieldType: a.fieldType || "",
|
|
1480
1646
|
fieldName: a.fieldName || `field-${out.length + 1}`,
|
|
@@ -1763,12 +1929,12 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
|
1763
1929
|
const w = safeNum(it.width, 0);
|
|
1764
1930
|
runs.push({
|
|
1765
1931
|
text: it.str,
|
|
1766
|
-
x: safeNum(vx *
|
|
1767
|
-
y: safeNum(yTop *
|
|
1932
|
+
x: safeNum(vx * PT_TO_MM3, 0),
|
|
1933
|
+
y: safeNum(yTop * PT_TO_MM3, 0),
|
|
1768
1934
|
fontSize: safeNum(fontSize, 10),
|
|
1769
1935
|
fontName: it.fontName,
|
|
1770
|
-
width: safeNum(w *
|
|
1771
|
-
height: safeNum((it.height || fontSize) *
|
|
1936
|
+
width: safeNum(w * PT_TO_MM3, 0),
|
|
1937
|
+
height: safeNum((it.height || fontSize) * PT_TO_MM3, fontSize * PT_TO_MM3),
|
|
1772
1938
|
color: op?.fill || "#000000",
|
|
1773
1939
|
opacity: invisible ? 0 : safeNum(op?.alpha, 1)
|
|
1774
1940
|
});
|
|
@@ -1791,10 +1957,10 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
|
1791
1957
|
}
|
|
1792
1958
|
const sameLine = Math.abs(last.y - r.y) <= Y_TOL;
|
|
1793
1959
|
const sameStyle = Math.abs(last.fontSize - r.fontSize) < 0.4 && (last.fontName === r.fontName || fontKey(last.fontName) === fontKey(r.fontName)) && last.color === r.color && Math.abs(last.opacity - r.opacity) < 0.05;
|
|
1794
|
-
const emMm = r.fontSize *
|
|
1960
|
+
const emMm = r.fontSize * PT_TO_MM3;
|
|
1795
1961
|
const extent = (t) => {
|
|
1796
1962
|
if (!/\s$/.test(t.text)) return t.width;
|
|
1797
|
-
const em = t.fontSize *
|
|
1963
|
+
const em = t.fontSize * PT_TO_MM3;
|
|
1798
1964
|
const est = Math.max(1, t.text.trim().length) * em * kGlyph + em * 0.25;
|
|
1799
1965
|
if (stretchedSpaces) return Math.min(t.width, est);
|
|
1800
1966
|
return t.width > est * 1.4 ? est : t.width;
|
|
@@ -1813,11 +1979,19 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
|
1813
1979
|
}
|
|
1814
1980
|
}
|
|
1815
1981
|
const elements = [];
|
|
1982
|
+
const lineMeta = /* @__PURE__ */ new WeakMap();
|
|
1816
1983
|
const tRuns = lines.map((l) => {
|
|
1817
1984
|
const cls = fontMap.get(l.fontName) || classifyFont(l.fontName || "");
|
|
1818
1985
|
return { text: l.text, x: l.x, y: l.y, width: l.width, height: l.height, fontSize: l.fontSize, fontName: l.fontName, color: l.color, bold: cls.weight === "bold" };
|
|
1819
1986
|
});
|
|
1820
|
-
const
|
|
1987
|
+
const sizeChars = /* @__PURE__ */ new Map();
|
|
1988
|
+
for (const l of lines) {
|
|
1989
|
+
const k = Math.round(l.fontSize * 2) / 2;
|
|
1990
|
+
sizeChars.set(k, (sizeChars.get(k) ?? 0) + l.text.length);
|
|
1991
|
+
}
|
|
1992
|
+
const bodyFontSize = [...sizeChars.entries()].sort((a, b) => b[1] - a[1])[0]?.[0] ?? 0;
|
|
1993
|
+
const gutters = options.readingOrder === false ? [] : detectGutters(lines.map((l) => ({ text: l.text, x: l.x, y: l.y, width: l.width, fontSize: l.fontSize })), bodyFontSize, pageW * PT_TO_MM3);
|
|
1994
|
+
const detected = options.detectTables === false ? [] : detectTables(tRuns, ops.shapes, pageW * PT_TO_MM3, gutters);
|
|
1821
1995
|
const consumedLines = /* @__PURE__ */ new Set();
|
|
1822
1996
|
const consumedShapes = /* @__PURE__ */ new Set();
|
|
1823
1997
|
const tableAtLine = /* @__PURE__ */ new Map();
|
|
@@ -1826,7 +2000,7 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
|
1826
2000
|
for (const k of t.shapeIdx) consumedShapes.add(k);
|
|
1827
2001
|
tableAtLine.set(Math.min(...t.lineIdx), t.element);
|
|
1828
2002
|
}
|
|
1829
|
-
const pageWmm = pageW *
|
|
2003
|
+
const pageWmm = pageW * PT_TO_MM3, pageHmm = pageH * PT_TO_MM3;
|
|
1830
2004
|
ops.shapes.forEach((sh, shapeIdx) => {
|
|
1831
2005
|
if (consumedShapes.has(shapeIdx)) return;
|
|
1832
2006
|
if (sh.width < 0.3 && sh.height < 0.3) return;
|
|
@@ -1870,12 +2044,6 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
|
1870
2044
|
fit: "fill"
|
|
1871
2045
|
});
|
|
1872
2046
|
}
|
|
1873
|
-
const sizeChars = /* @__PURE__ */ new Map();
|
|
1874
|
-
for (const l of lines) {
|
|
1875
|
-
const k = Math.round(l.fontSize * 2) / 2;
|
|
1876
|
-
sizeChars.set(k, (sizeChars.get(k) ?? 0) + l.text.length);
|
|
1877
|
-
}
|
|
1878
|
-
const bodyFontSize = [...sizeChars.entries()].sort((a, b) => b[1] - a[1])[0]?.[0] ?? 0;
|
|
1879
2047
|
const rowOf = /* @__PURE__ */ new Map();
|
|
1880
2048
|
const rowStartOf = /* @__PURE__ */ new Map();
|
|
1881
2049
|
const nextOnRow = /* @__PURE__ */ new Map();
|
|
@@ -1883,10 +2051,10 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
|
1883
2051
|
const order = lines.map((_, i) => i).filter((i) => !consumedLines.has(i));
|
|
1884
2052
|
for (let a = 0; a < order.length; a++) {
|
|
1885
2053
|
const i = order[a], li = lines[i];
|
|
1886
|
-
const tolY = Math.max(0.6, li.fontSize * PT_TO_MM2 * 0.35);
|
|
1887
2054
|
let bestNext = -1, bestX = Infinity;
|
|
1888
2055
|
for (let b = 0; b < order.length; b++) {
|
|
1889
2056
|
const j = order[b], lj = lines[j];
|
|
2057
|
+
const tolY = Math.max(0.6, Math.max(li.fontSize, lj.fontSize) * PT_TO_MM3 * 0.5);
|
|
1890
2058
|
if (j === i || Math.abs(lj.y - li.y) > tolY || lj.x <= li.x) continue;
|
|
1891
2059
|
if (lj.x < bestX) {
|
|
1892
2060
|
bestX = lj.x;
|
|
@@ -1903,7 +2071,7 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
|
1903
2071
|
let cur = i;
|
|
1904
2072
|
while (nextOnRow.has(cur)) {
|
|
1905
2073
|
const j = nextOnRow.get(cur), lc = lines[cur], lj = lines[j];
|
|
1906
|
-
const em = Math.min(lc.fontSize, lj.fontSize) *
|
|
2074
|
+
const em = Math.min(lc.fontSize, lj.fontSize) * PT_TO_MM3;
|
|
1907
2075
|
const gap = lj.x - (lc.x + lc.width);
|
|
1908
2076
|
if (gap < -em * 0.3 || gap > em * 0.6) break;
|
|
1909
2077
|
row.push(j);
|
|
@@ -1928,9 +2096,9 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
|
1928
2096
|
const first = lines[row[0]], last = lines[row[row.length - 1]];
|
|
1929
2097
|
const base = runStyle(first);
|
|
1930
2098
|
const rowEnd = last.x + last.width;
|
|
1931
|
-
const measuredW = Math.max((rowEnd - first.x) * 1.2 + first.fontSize *
|
|
2099
|
+
const measuredW = Math.max((rowEnd - first.x) * 1.2 + first.fontSize * PT_TO_MM3 * 0.4, first.fontSize * PT_TO_MM3);
|
|
1932
2100
|
const nextIdx2 = nextOnRow.get(row[row.length - 1]);
|
|
1933
|
-
const cap2 = nextIdx2 != null ? lines[nextIdx2].x - first.x - first.fontSize *
|
|
2101
|
+
const cap2 = nextIdx2 != null ? lines[nextIdx2].x - first.x - first.fontSize * PT_TO_MM3 * 0.3 : pageWmm - first.x;
|
|
1934
2102
|
const runs2 = [];
|
|
1935
2103
|
row.forEach((idx, k) => {
|
|
1936
2104
|
const r = lines[idx];
|
|
@@ -1939,7 +2107,7 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
|
1939
2107
|
if (k > 0) {
|
|
1940
2108
|
const prev2 = lines[row[k - 1]];
|
|
1941
2109
|
const gap = r.x - (prev2.x + prev2.width);
|
|
1942
|
-
if (gap > r.fontSize *
|
|
2110
|
+
if (gap > r.fontSize * PT_TO_MM3 * 0.08 && !/\s$/.test(prev2.text) && !/^\s/.test(text2)) text2 = " " + text2;
|
|
1943
2111
|
}
|
|
1944
2112
|
const run = { text: text2 };
|
|
1945
2113
|
if (st.bold) run.bold = true;
|
|
@@ -1953,13 +2121,16 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
|
1953
2121
|
});
|
|
1954
2122
|
const style2 = { fontSize: Math.round(first.fontSize * 10) / 10, fontFamily: base.cls.family };
|
|
1955
2123
|
if (first.opacity < 0.999) style2.opacity = Math.round(first.opacity * 100) / 100;
|
|
1956
|
-
|
|
2124
|
+
const rt = {
|
|
1957
2125
|
type: "richtext",
|
|
1958
2126
|
runs: runs2,
|
|
1959
2127
|
position: { x: Math.max(0, Math.round(first.x * 100) / 100), y: Math.max(0, Math.round(Math.min(...row.map((i) => lines[i].y)) * 100) / 100) },
|
|
1960
|
-
width: Math.max(2, Math.round(Math.max(first.fontSize *
|
|
2128
|
+
width: Math.max(2, Math.round(Math.max(first.fontSize * PT_TO_MM3, Math.min(measuredW, cap2)) * 100) / 100),
|
|
1961
2129
|
style: style2
|
|
1962
|
-
}
|
|
2130
|
+
};
|
|
2131
|
+
const dominant = row.map((i) => lines[i]).sort((a, b) => b.text.trim().length - a.text.trim().length)[0];
|
|
2132
|
+
lineMeta.set(rt, { w: Math.max(first.fontSize * PT_TO_MM3, rowEnd - first.x), size: dominant.fontSize, face: fontKey(dominant.fontName) });
|
|
2133
|
+
elements.push(rt);
|
|
1963
2134
|
return;
|
|
1964
2135
|
}
|
|
1965
2136
|
const cls = fontMap.get(l.fontName) || classifyFont(l.fontName || "");
|
|
@@ -1972,10 +2143,10 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
|
1972
2143
|
if (l.color !== "#000000") style.color = l.color;
|
|
1973
2144
|
if (l.opacity < 0.999) style.opacity = Math.round(l.opacity * 100) / 100;
|
|
1974
2145
|
const link = findLinkForRun2(l);
|
|
1975
|
-
const measured = Math.max(l.width * 1.2 + l.fontSize *
|
|
2146
|
+
const measured = Math.max(l.width * 1.2 + l.fontSize * PT_TO_MM3 * 0.4, l.fontSize * PT_TO_MM3);
|
|
1976
2147
|
const remaining = Math.max(measured, pageWmm - l.x);
|
|
1977
2148
|
const nextIdx = nextOnRow.get(lineIdx);
|
|
1978
|
-
const cap = nextIdx != null ? Math.max(l.fontSize *
|
|
2149
|
+
const cap = nextIdx != null ? Math.max(l.fontSize * PT_TO_MM3, lines[nextIdx].x - l.x - l.fontSize * PT_TO_MM3 * 0.3) : Infinity;
|
|
1979
2150
|
const elWidth = Math.min(measured, remaining, cap);
|
|
1980
2151
|
const text = {
|
|
1981
2152
|
type: "text",
|
|
@@ -1996,14 +2167,26 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
|
1996
2167
|
else if (link.destPage != null) text.link = { type: "internal", target: `#page-${link.destPage + 1}` };
|
|
1997
2168
|
}
|
|
1998
2169
|
const prev = elements[elements.length - 1];
|
|
1999
|
-
if (text.heading && prev && prev.type === "text" && prev.heading === text.heading && !link && !prev.link && Math.abs(prev.style?.fontSize - style.fontSize) < 0.5 && Math.abs(prev.position.x - text.position.x) < 1 && text.position.y - prev.position.y < l.fontSize *
|
|
2170
|
+
if (text.heading && prev && prev.type === "text" && prev.heading === text.heading && !link && !prev.link && Math.abs(prev.style?.fontSize - style.fontSize) < 0.5 && Math.abs(prev.position.x - text.position.x) < 1 && text.position.y - prev.position.y < l.fontSize * PT_TO_MM3 * 2.2 && text.position.y > prev.position.y) {
|
|
2000
2171
|
prev.content = `${prev.content} ${text.content}`.replace(/\s+/g, " ");
|
|
2001
2172
|
prev.tocEntry = prev.content;
|
|
2002
2173
|
prev.width = Math.max(prev.width ?? 0, text.width ?? 0);
|
|
2003
2174
|
return;
|
|
2004
2175
|
}
|
|
2176
|
+
lineMeta.set(text, { w: Math.max(l.fontSize * PT_TO_MM3, l.width), size: l.fontSize, face: fontKey(l.fontName) });
|
|
2005
2177
|
elements.push(text);
|
|
2006
2178
|
});
|
|
2179
|
+
if (options.readingOrder !== false) {
|
|
2180
|
+
if (gutters.length) {
|
|
2181
|
+
const isFlow = (e) => e.type === "text" || e.type === "richtext" || e.type === "table";
|
|
2182
|
+
const flow = elements.filter(isFlow), rest = elements.filter((e) => !isFlow(e));
|
|
2183
|
+
elements.splice(0, elements.length, ...rest, ...orderByColumns(flow, gutters, pageWmm));
|
|
2184
|
+
}
|
|
2185
|
+
}
|
|
2186
|
+
if (options.foldParagraphs !== false) {
|
|
2187
|
+
const folded = foldParagraphs(elements, lineMeta, pageWmm);
|
|
2188
|
+
elements.splice(0, elements.length, ...folded);
|
|
2189
|
+
}
|
|
2007
2190
|
for (const w of formWidgets) {
|
|
2008
2191
|
if (w.pushButton) continue;
|
|
2009
2192
|
const baseEl = {
|
|
@@ -2038,7 +2221,7 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
|
2038
2221
|
}
|
|
2039
2222
|
pages.push({
|
|
2040
2223
|
id: `page-${pi}`,
|
|
2041
|
-
pageSize: { width: Math.round(pageW *
|
|
2224
|
+
pageSize: { width: Math.round(pageW * PT_TO_MM3 * 100) / 100, height: Math.round(pageH * PT_TO_MM3 * 100) / 100 },
|
|
2042
2225
|
margins: { top: 0, right: 0, bottom: 0, left: 0 },
|
|
2043
2226
|
elements
|
|
2044
2227
|
});
|
|
@@ -2188,25 +2371,209 @@ async function importPdfToJdf2(source, title, options = {}) {
|
|
|
2188
2371
|
};
|
|
2189
2372
|
return importPdfToJdf(source, title, runtime, options);
|
|
2190
2373
|
}
|
|
2374
|
+
var DEFAULT_OLLAMA_MODEL = "qwen2.5vl:3b";
|
|
2375
|
+
var DEFAULT_OPENAI_MODEL = "gpt-4o-mini";
|
|
2376
|
+
var OLLAMA_HOST = process.env.OLLAMA_HOST || "http://localhost:11434";
|
|
2377
|
+
async function loadDoc(file) {
|
|
2378
|
+
if (file.toLowerCase().endsWith(".jdfx")) {
|
|
2379
|
+
const zip = await JSZip.loadAsync(fs8.readFileSync(file));
|
|
2380
|
+
const f = zip.file(JDFX_DOCUMENT_PATH);
|
|
2381
|
+
if (!f) throw new Error(`Bundle missing ${JDFX_DOCUMENT_PATH}`);
|
|
2382
|
+
const doc = JSON.parse(await f.async("string"));
|
|
2383
|
+
const manifest = zip.file("manifest.json") ? JSON.parse(await zip.file("manifest.json").async("string")) : { assets: [] };
|
|
2384
|
+
for (const a of manifest.assets ?? []) {
|
|
2385
|
+
const af = zip.file(a.path);
|
|
2386
|
+
if (!af) continue;
|
|
2387
|
+
const data = (await af.async("nodebuffer")).toString("base64");
|
|
2388
|
+
const res = { src: "embedded", mimeType: a.mimeType, data };
|
|
2389
|
+
doc.resources ??= {};
|
|
2390
|
+
if (/^video\//i.test(a.mimeType || "")) (doc.resources.videos ??= {})[a.id] = res;
|
|
2391
|
+
else (doc.resources.images ??= {})[a.id] = res;
|
|
2392
|
+
}
|
|
2393
|
+
return { doc, bundle: true };
|
|
2394
|
+
}
|
|
2395
|
+
return { doc: JSON.parse(fs8.readFileSync(file, "utf-8")), bundle: false };
|
|
2396
|
+
}
|
|
2397
|
+
function findImages(doc) {
|
|
2398
|
+
const out = [];
|
|
2399
|
+
const walk2 = (els, page) => {
|
|
2400
|
+
for (const el of els ?? []) {
|
|
2401
|
+
if (el?.type === "image") out.push({ el, page, index: out.length });
|
|
2402
|
+
if (el?.elements) walk2(el.elements, page);
|
|
2403
|
+
}
|
|
2404
|
+
};
|
|
2405
|
+
doc.pages.forEach((p, i) => walk2(p.elements, i + 1));
|
|
2406
|
+
return out;
|
|
2407
|
+
}
|
|
2408
|
+
async function imageBytes(doc, el, docDir) {
|
|
2409
|
+
const fromData = (d, fallback) => {
|
|
2410
|
+
const m = d.match(/^data:([^;,]+)?[^,]*,(.*)$/s);
|
|
2411
|
+
return m ? { bytes: Buffer.from(m[2], "base64"), mime: m[1] || fallback } : { bytes: Buffer.from(d, "base64"), mime: fallback };
|
|
2412
|
+
};
|
|
2413
|
+
const res = el.resource ? doc.resources?.images?.[el.resource] ?? doc.resources?.[el.resource] : void 0;
|
|
2414
|
+
if (res?.data) return fromData(String(res.data), res.mimeType || "image/png");
|
|
2415
|
+
if (res?.path) {
|
|
2416
|
+
const p = path8.resolve(docDir, res.path);
|
|
2417
|
+
return fs8.existsSync(p) ? { bytes: fs8.readFileSync(p), mime: res.mimeType || "image/png" } : null;
|
|
2418
|
+
}
|
|
2419
|
+
const src = el.src;
|
|
2420
|
+
if (!src) return null;
|
|
2421
|
+
if (src.startsWith("data:")) return fromData(src, "image/png");
|
|
2422
|
+
if (/^https?:\/\//i.test(src)) {
|
|
2423
|
+
const r = await fetch(src);
|
|
2424
|
+
if (!r.ok) throw new Error(`download failed ${r.status}: ${src}`);
|
|
2425
|
+
return { bytes: Buffer.from(await r.arrayBuffer()), mime: r.headers.get("content-type") || "image/png" };
|
|
2426
|
+
}
|
|
2427
|
+
const local = path8.resolve(docDir, src);
|
|
2428
|
+
return fs8.existsSync(local) ? { bytes: fs8.readFileSync(local), mime: "image/png" } : null;
|
|
2429
|
+
}
|
|
2430
|
+
async function ocrTesseract(bytes, lang) {
|
|
2431
|
+
const { createWorker } = await import('tesseract.js');
|
|
2432
|
+
const cachePath = path8.join(os.homedir(), ".cache", "jdf", "tesseract");
|
|
2433
|
+
fs8.mkdirSync(cachePath, { recursive: true });
|
|
2434
|
+
const worker = await createWorker(lang, 1, { cachePath, logger: () => {
|
|
2435
|
+
} });
|
|
2436
|
+
try {
|
|
2437
|
+
const { data } = await worker.recognize(bytes, {}, { text: true, blocks: true });
|
|
2438
|
+
let w = 1, h = 1;
|
|
2439
|
+
try {
|
|
2440
|
+
const { loadImage } = await import('@napi-rs/canvas');
|
|
2441
|
+
const im = await loadImage(bytes);
|
|
2442
|
+
w = im.width || 1;
|
|
2443
|
+
h = im.height || 1;
|
|
2444
|
+
} catch {
|
|
2445
|
+
}
|
|
2446
|
+
const lines = data.blocks?.flatMap((b) => b.paragraphs?.flatMap((p) => p.lines ?? []) ?? []) ?? data.lines ?? [];
|
|
2447
|
+
const blocks = lines.map((ln) => ({ text: String(ln.text ?? "").replace(/\s+/g, " ").trim(), confidence: ln.confidence != null ? Math.round(ln.confidence) / 100 : void 0, bbox: ln.bbox ? { x: +(ln.bbox.x0 / w).toFixed(4), y: +(ln.bbox.y0 / h).toFixed(4), w: +((ln.bbox.x1 - ln.bbox.x0) / w).toFixed(4), h: +((ln.bbox.y1 - ln.bbox.y0) / h).toFixed(4) } : void 0 })).filter((b) => b.text.length > 0 && (b.confidence == null || b.confidence >= 0.3));
|
|
2448
|
+
if (!blocks.length && String(data.text ?? "").trim()) blocks.push({ text: String(data.text).replace(/\s+/g, " ").trim() });
|
|
2449
|
+
return { blocks, source: `tesseract.js:${lang}` };
|
|
2450
|
+
} finally {
|
|
2451
|
+
await worker.terminate();
|
|
2452
|
+
}
|
|
2453
|
+
}
|
|
2454
|
+
async function openaiVision(bytes, mime, model, prompt2) {
|
|
2455
|
+
const key = process.env.OPENAI_API_KEY;
|
|
2456
|
+
if (!key) throw new Error("OPENAI_API_KEY is not set");
|
|
2457
|
+
const base = process.env.OPENAI_BASE_URL || "https://api.openai.com/v1";
|
|
2458
|
+
const r = await fetch(`${base}/chat/completions`, { method: "POST", headers: { Authorization: `Bearer ${key}`, "content-type": "application/json" }, body: JSON.stringify({ model, messages: [{ role: "user", content: [{ type: "text", text: prompt2 }, { type: "image_url", image_url: { url: `data:${mime};base64,${bytes.toString("base64")}` } }] }], max_tokens: 800 }) });
|
|
2459
|
+
if (!r.ok) throw new Error(`OpenAI vision failed ${r.status}: ${(await r.text()).slice(0, 300)}`);
|
|
2460
|
+
const j = await r.json();
|
|
2461
|
+
return String(j.choices?.[0]?.message?.content ?? "").trim();
|
|
2462
|
+
}
|
|
2463
|
+
async function ollamaVision(bytes, model, prompt2) {
|
|
2464
|
+
const r = await fetch(`${OLLAMA_HOST}/api/generate`, { method: "POST", body: JSON.stringify({ model, prompt: prompt2, images: [bytes.toString("base64")], stream: false, options: { temperature: 0 } }) });
|
|
2465
|
+
if (!r.ok) throw new Error(`Ollama vision failed ${r.status}: ${(await r.text()).slice(0, 300)} \u2014 is the model pulled? (ollama pull ${model})`);
|
|
2466
|
+
const j = await r.json();
|
|
2467
|
+
return String(j.response ?? "").trim();
|
|
2468
|
+
}
|
|
2469
|
+
var CAPTION_PROMPT = "Describe this image for a search index in one or two factual sentences: what it shows, any chart type, axes, trends, labels, names and numbers you can read. No preamble.";
|
|
2470
|
+
var OCR_PROMPT = "Transcribe all text visible in this image exactly, line by line, top to bottom, left to right. Output only the text.";
|
|
2471
|
+
async function describeDocument(doc, docDir, opts = {}) {
|
|
2472
|
+
const ocrP = opts.ocr ?? "tesseract";
|
|
2473
|
+
const capP = opts.caption ?? "ollama";
|
|
2474
|
+
const lang = opts.ocrLanguage ?? "eng";
|
|
2475
|
+
const images = findImages(doc);
|
|
2476
|
+
const targets = opts.element != null ? images.filter((im) => im.el.id === opts.element || String(im.index) === opts.element) : images;
|
|
2477
|
+
if (opts.element != null && !targets.length) throw new Error(`no image element "${opts.element}" (have: ${images.map((i) => i.el.id ?? `#${i.index}`).join(", ") || "none"})`);
|
|
2478
|
+
const stats = { ocr: 0, captions: 0, skipped: 0, failed: [] };
|
|
2479
|
+
for (const im of targets) {
|
|
2480
|
+
const el = im.el;
|
|
2481
|
+
if (!el.id) el.id = `image-${im.index + 1}`;
|
|
2482
|
+
const needOcr = ocrP !== "none" && (opts.force || !el.ocr?.blocks?.length);
|
|
2483
|
+
const needCap = capP !== "none" && (opts.force || !el.caption);
|
|
2484
|
+
if (!needOcr && !needCap) {
|
|
2485
|
+
stats.skipped++;
|
|
2486
|
+
continue;
|
|
2487
|
+
}
|
|
2488
|
+
const img = await imageBytes(doc, el, docDir);
|
|
2489
|
+
if (!img) {
|
|
2490
|
+
stats.failed.push(`${el.id}: image bytes not reachable`);
|
|
2491
|
+
continue;
|
|
2492
|
+
}
|
|
2493
|
+
if (needOcr) {
|
|
2494
|
+
try {
|
|
2495
|
+
if (ocrP === "tesseract") {
|
|
2496
|
+
const r = await ocrTesseract(img.bytes, lang);
|
|
2497
|
+
el.ocr = { language: lang, source: r.source, created: (/* @__PURE__ */ new Date()).toISOString(), blocks: r.blocks };
|
|
2498
|
+
} else {
|
|
2499
|
+
const text = await openaiVision(img.bytes, img.mime, opts.captionModel ?? DEFAULT_OPENAI_MODEL, OCR_PROMPT);
|
|
2500
|
+
el.ocr = { source: `openai:${opts.captionModel ?? DEFAULT_OPENAI_MODEL}`, created: (/* @__PURE__ */ new Date()).toISOString(), blocks: text.split(/\r?\n/).map((t) => t.trim()).filter(Boolean).map((t) => ({ text: t })) };
|
|
2501
|
+
}
|
|
2502
|
+
stats.ocr++;
|
|
2503
|
+
} catch (e) {
|
|
2504
|
+
stats.failed.push(`${el.id} ocr: ${e.message}`);
|
|
2505
|
+
}
|
|
2506
|
+
}
|
|
2507
|
+
if (needCap) {
|
|
2508
|
+
try {
|
|
2509
|
+
const model = opts.captionModel ?? (capP === "ollama" ? DEFAULT_OLLAMA_MODEL : DEFAULT_OPENAI_MODEL);
|
|
2510
|
+
const text = capP === "ollama" ? await ollamaVision(img.bytes, model, CAPTION_PROMPT) : await openaiVision(img.bytes, img.mime, model, CAPTION_PROMPT);
|
|
2511
|
+
if (text) {
|
|
2512
|
+
el.caption = text;
|
|
2513
|
+
el.captionSource = `${capP}:${model}`;
|
|
2514
|
+
stats.captions++;
|
|
2515
|
+
}
|
|
2516
|
+
} catch (e) {
|
|
2517
|
+
stats.failed.push(`${el.id} caption: ${e.message}`);
|
|
2518
|
+
}
|
|
2519
|
+
}
|
|
2520
|
+
if (!opts.quiet) console.log(` \xB7 ${el.id} (page ${im.page}): ${needOcr ? `ocr ${el.ocr?.blocks?.length ?? 0} block(s)` : "ocr kept"}${needCap ? ` \xB7 caption ${el.caption ? `"${String(el.caption).slice(0, 70)}${String(el.caption).length > 70 ? "\u2026" : ""}"` : "\u2014"}` : ""}`);
|
|
2521
|
+
}
|
|
2522
|
+
return stats;
|
|
2523
|
+
}
|
|
2524
|
+
async function describeFile(inputPath, opts = {}) {
|
|
2525
|
+
const input = path8.resolve(inputPath);
|
|
2526
|
+
if (!fs8.existsSync(input)) throw new Error(`File not found: ${input}`);
|
|
2527
|
+
const { doc, bundle } = await loadDoc(input);
|
|
2528
|
+
const images = findImages(doc);
|
|
2529
|
+
if (!images.length) {
|
|
2530
|
+
console.log(`No image elements in ${path8.basename(input)} \u2014 nothing to describe.`);
|
|
2531
|
+
return;
|
|
2532
|
+
}
|
|
2533
|
+
console.log(`Describing: ${path8.basename(input)} \u2014 ${images.length} image(s); ocr=${opts.ocr ?? "tesseract"} caption=${opts.caption ?? "ollama"}${(opts.caption ?? "ollama") === "ollama" ? ` (${opts.captionModel ?? DEFAULT_OLLAMA_MODEL}, local)` : ""}`);
|
|
2534
|
+
const stats = await describeDocument(doc, path8.dirname(input), opts);
|
|
2535
|
+
const output = opts.output ? path8.resolve(opts.output) : input;
|
|
2536
|
+
if (output.toLowerCase().endsWith(".jdfx") || bundle && !opts.output) fs8.writeFileSync(output, (await packJdfx(doc)).bytes);
|
|
2537
|
+
else fs8.writeFileSync(output, JSON.stringify(doc, null, 2));
|
|
2538
|
+
console.log(`Done: ${stats.ocr} OCR, ${stats.captions} caption(s), ${stats.skipped} already had text${stats.failed.length ? `, ${stats.failed.length} failed` : ""}.`);
|
|
2539
|
+
for (const f of stats.failed) console.warn(` ! ${f}`);
|
|
2540
|
+
console.log(`Output: ${output}
|
|
2541
|
+
Next: jdf chunk ${path8.basename(output)} # image text is now part of the chunks`);
|
|
2542
|
+
if (stats.failed.length && stats.ocr + stats.captions === 0) process.exitCode = 1;
|
|
2543
|
+
}
|
|
2191
2544
|
|
|
2192
2545
|
// src/commands/import-pdf.ts
|
|
2193
2546
|
async function importPdf(inputPath, outputPath, options = {}) {
|
|
2194
|
-
const input =
|
|
2195
|
-
if (!
|
|
2547
|
+
const input = path8.resolve(inputPath);
|
|
2548
|
+
if (!fs8.existsSync(input)) {
|
|
2196
2549
|
console.error(`File not found: ${input}`);
|
|
2197
2550
|
process.exit(1);
|
|
2198
2551
|
}
|
|
2199
2552
|
console.log(`Importing: ${input}`);
|
|
2200
|
-
const title =
|
|
2553
|
+
const title = path8.basename(input, path8.extname(input));
|
|
2201
2554
|
const t0 = Date.now();
|
|
2202
2555
|
const doc = await importPdfToJdf2(input, title, {
|
|
2203
2556
|
password: options.password,
|
|
2204
2557
|
invisibleText: options.dropInvisibleText ? "drop" : "keep"
|
|
2205
2558
|
});
|
|
2206
2559
|
console.log(`Parsed in ${((Date.now() - t0) / 1e3).toFixed(1)}s \u2014 ${doc.pages.length} page(s)`);
|
|
2560
|
+
const scanned = doc.pages.filter((p) => !p.elements.some((e) => e.type === "text" || e.type === "richtext" || e.type === "table") && p.elements.some((e) => e.type === "image"));
|
|
2561
|
+
if (scanned.length) {
|
|
2562
|
+
if (options.ocr && options.ocr !== "none") {
|
|
2563
|
+
console.log(`OCR: ${scanned.length} scanned page(s) \u2192 ${options.ocr}`);
|
|
2564
|
+
for (const p of scanned) for (const e of p.elements) if (e.type === "image" && !e.id) e.id = `scan-${doc.pages.indexOf(p) + 1}`;
|
|
2565
|
+
const ids = scanned.flatMap((p) => p.elements.filter((e) => e.type === "image").map((e) => e.id));
|
|
2566
|
+
for (const id of ids) {
|
|
2567
|
+
const st = await describeDocument(doc, path8.dirname(input), { element: id, ocr: options.ocr, caption: "none", quiet: true });
|
|
2568
|
+
if (st.failed.length) console.warn(` ! ${st.failed.join("; ")}`);
|
|
2569
|
+
}
|
|
2570
|
+
} else {
|
|
2571
|
+
console.warn(`! ${scanned.length} page(s) have no text layer (scanned). RAG will skip them \u2014 re-run with --ocr tesseract (local) or --ocr openai.`);
|
|
2572
|
+
}
|
|
2573
|
+
}
|
|
2207
2574
|
let output;
|
|
2208
2575
|
if (outputPath) {
|
|
2209
|
-
output =
|
|
2576
|
+
output = path8.resolve(outputPath);
|
|
2210
2577
|
} else {
|
|
2211
2578
|
const stem = input.replace(/\.pdf$/i, "");
|
|
2212
2579
|
const wantJdfx = !options.forceJson && shouldUseJdfx(doc);
|
|
@@ -2215,11 +2582,11 @@ async function importPdf(inputPath, outputPath, options = {}) {
|
|
|
2215
2582
|
console.log(`Output: ${output}`);
|
|
2216
2583
|
if (output.toLowerCase().endsWith(".jdfx")) {
|
|
2217
2584
|
const { bytes, manifest } = await packJdfx(doc);
|
|
2218
|
-
|
|
2585
|
+
fs8.writeFileSync(output, bytes);
|
|
2219
2586
|
console.log(`
|
|
2220
2587
|
Done! Created ${doc.pages.length} page(s), ${manifest.assets.length} asset(s) bundled`);
|
|
2221
2588
|
} else {
|
|
2222
|
-
|
|
2589
|
+
fs8.writeFileSync(output, JSON.stringify(doc, null, 2));
|
|
2223
2590
|
console.log(`
|
|
2224
2591
|
Done! Created ${doc.pages.length} page(s)`);
|
|
2225
2592
|
}
|
|
@@ -2232,23 +2599,23 @@ var ImportJsonError = class extends Error {
|
|
|
2232
2599
|
}
|
|
2233
2600
|
};
|
|
2234
2601
|
async function importJson(inputPath, outputPath, options = {}) {
|
|
2235
|
-
const input =
|
|
2236
|
-
if (!
|
|
2602
|
+
const input = path8.resolve(inputPath);
|
|
2603
|
+
if (!fs8.existsSync(input)) {
|
|
2237
2604
|
throw new ImportJsonError(`File not found: ${input}`);
|
|
2238
2605
|
}
|
|
2239
2606
|
console.log(`Importing: ${input}`);
|
|
2240
|
-
const raw =
|
|
2607
|
+
const raw = fs8.readFileSync(input, "utf-8");
|
|
2241
2608
|
let parsed;
|
|
2242
2609
|
try {
|
|
2243
2610
|
parsed = JSON.parse(raw);
|
|
2244
2611
|
} catch (e) {
|
|
2245
2612
|
throw new ImportJsonError(`Not valid JSON: ${e.message}`);
|
|
2246
2613
|
}
|
|
2247
|
-
const title =
|
|
2614
|
+
const title = path8.basename(input, path8.extname(input));
|
|
2248
2615
|
const doc = normaliseToJdf(parsed, title);
|
|
2249
2616
|
let output;
|
|
2250
2617
|
if (outputPath) {
|
|
2251
|
-
output =
|
|
2618
|
+
output = path8.resolve(outputPath);
|
|
2252
2619
|
} else {
|
|
2253
2620
|
const stem = input.replace(/\.json$/i, "");
|
|
2254
2621
|
const wantJdfx = !options.forceJson && shouldUseJdfx(doc);
|
|
@@ -2257,11 +2624,11 @@ async function importJson(inputPath, outputPath, options = {}) {
|
|
|
2257
2624
|
console.log(`Output: ${output}`);
|
|
2258
2625
|
if (output.toLowerCase().endsWith(".jdfx")) {
|
|
2259
2626
|
const { bytes, manifest } = await packJdfx(doc);
|
|
2260
|
-
|
|
2627
|
+
fs8.writeFileSync(output, bytes);
|
|
2261
2628
|
console.log(`
|
|
2262
2629
|
Done! Created ${doc.pages.length} page(s), ${manifest.assets.length} asset(s) bundled`);
|
|
2263
2630
|
} else {
|
|
2264
|
-
|
|
2631
|
+
fs8.writeFileSync(output, JSON.stringify(doc, null, 2));
|
|
2265
2632
|
console.log(`
|
|
2266
2633
|
Done! Created ${doc.pages.length} page(s)`);
|
|
2267
2634
|
}
|
|
@@ -2363,8 +2730,8 @@ function transcriptChunks(el, elementId2, page, crumb, windowSec, maxTokens) {
|
|
|
2363
2730
|
const body = win.map((sg) => sg.speaker ? `${sg.speaker}: ${sg.text}` : sg.text).join(" ").replace(/\s+/g, " ").trim();
|
|
2364
2731
|
const text = `[${fmtTime(t0)}\u2013${fmtTime(t1)}] ${body}`;
|
|
2365
2732
|
const chapter = chapterAt(t0);
|
|
2366
|
-
const
|
|
2367
|
-
out.push({ id: `${elementId2}@${Math.round(t0)}`, text, path:
|
|
2733
|
+
const path10 = [...crumb, ...el.title ? [String(el.title)] : [], ...chapter ? [String(chapter.title)] : []];
|
|
2734
|
+
out.push({ id: `${elementId2}@${Math.round(t0)}`, text, path: path10, page, types: ["video"], tokens: estimateTokens(text), hash: hashText(text), media: { element: elementId2, t0, t1 } });
|
|
2368
2735
|
win = [];
|
|
2369
2736
|
};
|
|
2370
2737
|
for (const sg of segs) {
|
|
@@ -2433,8 +2800,14 @@ function serializeElement(el) {
|
|
|
2433
2800
|
}
|
|
2434
2801
|
case "checkbox":
|
|
2435
2802
|
return `${e.checked ? "[x]" : "[ ]"} ${e.label ?? ""}`.trim();
|
|
2436
|
-
case "image":
|
|
2437
|
-
|
|
2803
|
+
case "image": {
|
|
2804
|
+
const parts = [];
|
|
2805
|
+
if (e.alt) parts.push(`[image: ${e.alt}]`);
|
|
2806
|
+
if (e.caption) parts.push(String(e.caption).trim());
|
|
2807
|
+
const ocr = (e.ocr?.blocks || []).map((b) => String(b.text ?? "").trim()).filter(Boolean).join("\n");
|
|
2808
|
+
if (ocr) parts.push(ocr);
|
|
2809
|
+
return parts.join("\n");
|
|
2810
|
+
}
|
|
2438
2811
|
case "video":
|
|
2439
2812
|
return e.title ? `[video: ${e.title}]` : "";
|
|
2440
2813
|
case "toc":
|
|
@@ -2554,34 +2927,62 @@ function chunkDocument(doc, options = {}) {
|
|
|
2554
2927
|
}
|
|
2555
2928
|
async function loadJdf(filePath) {
|
|
2556
2929
|
if (filePath.toLowerCase().endsWith(".jdfx")) {
|
|
2557
|
-
const zip = await JSZip.loadAsync(
|
|
2930
|
+
const zip = await JSZip.loadAsync(fs8.readFileSync(filePath));
|
|
2558
2931
|
const docFile = zip.file(JDFX_DOCUMENT_PATH);
|
|
2559
2932
|
if (!docFile) throw new Error(`Bundle missing ${JDFX_DOCUMENT_PATH}`);
|
|
2560
2933
|
return JSON.parse(await docFile.async("string"));
|
|
2561
2934
|
}
|
|
2562
|
-
return JSON.parse(
|
|
2935
|
+
return JSON.parse(fs8.readFileSync(filePath, "utf-8"));
|
|
2936
|
+
}
|
|
2937
|
+
function mediaCoverage(doc) {
|
|
2938
|
+
const cov = { images: { total: 0, covered: 0, missing: [] }, videos: { total: 0, covered: 0, missing: [] } };
|
|
2939
|
+
const walk2 = (els, page) => {
|
|
2940
|
+
for (const el of els ?? []) {
|
|
2941
|
+
if (el?.type === "image") {
|
|
2942
|
+
cov.images.total++;
|
|
2943
|
+
const has = !!(el.caption && String(el.caption).trim()) || !!el.ocr?.blocks?.some((b) => String(b.text ?? "").trim());
|
|
2944
|
+
if (has) cov.images.covered++;
|
|
2945
|
+
else cov.images.missing.push({ id: el.id, page, alt: el.alt });
|
|
2946
|
+
} else if (el?.type === "video") {
|
|
2947
|
+
cov.videos.total++;
|
|
2948
|
+
if (el.transcript?.segments?.length) cov.videos.covered++;
|
|
2949
|
+
else cov.videos.missing.push({ id: el.id, page, title: el.title });
|
|
2950
|
+
}
|
|
2951
|
+
if (el?.elements) walk2(el.elements, page);
|
|
2952
|
+
}
|
|
2953
|
+
};
|
|
2954
|
+
doc.pages.forEach((p, i) => walk2(p.elements, i + 1));
|
|
2955
|
+
return cov;
|
|
2956
|
+
}
|
|
2957
|
+
function coverageSummary(cov) {
|
|
2958
|
+
const parts = [];
|
|
2959
|
+
if (cov.images.missing.length) parts.push(`${cov.images.missing.length} of ${cov.images.total} image(s) have no caption/OCR text \u2192 jdf describe`);
|
|
2960
|
+
if (cov.videos.missing.length) parts.push(`${cov.videos.missing.length} of ${cov.videos.total} video(s) have no transcript \u2192 jdf transcribe`);
|
|
2961
|
+
return parts.length ? parts.join("; ") : null;
|
|
2563
2962
|
}
|
|
2564
2963
|
async function chunkFile(inputPath, opts = {}) {
|
|
2565
|
-
const input =
|
|
2566
|
-
if (!
|
|
2964
|
+
const input = path8.resolve(inputPath);
|
|
2965
|
+
if (!fs8.existsSync(input)) throw new Error(`File not found: ${input}`);
|
|
2567
2966
|
const doc = await loadJdf(input);
|
|
2568
2967
|
const strategy = opts.strategy ?? "section";
|
|
2569
2968
|
const chunks = chunkDocument(doc, { strategy, maxTokens: opts.maxTokens, transcriptWindowSec: opts.transcriptWindowSec });
|
|
2969
|
+
const gap = coverageSummary(mediaCoverage(doc));
|
|
2970
|
+
if (gap) console.warn(` ! media without text (skipped by retrieval): ${gap}`);
|
|
2570
2971
|
const format = opts.format ?? "jsonl";
|
|
2571
2972
|
console.log(`Chunking: ${input}`);
|
|
2572
2973
|
console.log(`Strategy: ${strategy}${opts.maxTokens ? ` (max ${opts.maxTokens} tokens)` : ""}`);
|
|
2573
2974
|
if (format === "inline") {
|
|
2574
|
-
const out = opts.output ?
|
|
2975
|
+
const out = opts.output ? path8.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".jdf");
|
|
2575
2976
|
const withIndex = { ...doc, index: { chunker: `jdf-${strategy}-v1`, chunks } };
|
|
2576
|
-
|
|
2977
|
+
fs8.writeFileSync(out, JSON.stringify(withIndex, null, 2));
|
|
2577
2978
|
console.log(`Output: ${out} (${chunks.length} chunks in "index" block)`);
|
|
2578
2979
|
} else if (format === "json") {
|
|
2579
|
-
const out = opts.output ?
|
|
2580
|
-
|
|
2980
|
+
const out = opts.output ? path8.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".chunks.json");
|
|
2981
|
+
fs8.writeFileSync(out, JSON.stringify(chunks, null, 2));
|
|
2581
2982
|
console.log(`Output: ${out} (${chunks.length} chunks)`);
|
|
2582
2983
|
} else {
|
|
2583
|
-
const out = opts.output ?
|
|
2584
|
-
|
|
2984
|
+
const out = opts.output ? path8.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".chunks.jsonl");
|
|
2985
|
+
fs8.writeFileSync(out, chunks.map((c) => JSON.stringify(c)).join("\n") + "\n");
|
|
2585
2986
|
console.log(`Output: ${out} (${chunks.length} chunks)`);
|
|
2586
2987
|
}
|
|
2587
2988
|
const totalTokens = chunks.reduce((a, c) => a + c.tokens, 0);
|
|
@@ -2604,11 +3005,11 @@ async function embedBatch(provider, model, inputs) {
|
|
|
2604
3005
|
throw new Error(`Unknown embedding provider: ${provider}`);
|
|
2605
3006
|
}
|
|
2606
3007
|
}
|
|
2607
|
-
var
|
|
3008
|
+
var OLLAMA_HOST2 = process.env.OLLAMA_HOST || "http://localhost:11434";
|
|
2608
3009
|
var OLLAMA_CONTAINER = "jdf-ollama";
|
|
2609
3010
|
async function ollamaUp() {
|
|
2610
3011
|
try {
|
|
2611
|
-
const res = await fetch(`${
|
|
3012
|
+
const res = await fetch(`${OLLAMA_HOST2}/api/tags`, { signal: AbortSignal.timeout(1500) });
|
|
2612
3013
|
return res.ok;
|
|
2613
3014
|
} catch {
|
|
2614
3015
|
return false;
|
|
@@ -2633,12 +3034,12 @@ async function ensureOllama(model, autoStart) {
|
|
|
2633
3034
|
docker run -d --name ${OLLAMA_CONTAINER} -p 11434:11434 -v jdf-ollama:/root/.ollama ollama/ollama
|
|
2634
3035
|
docker exec ${OLLAMA_CONTAINER} ollama pull ${model}`;
|
|
2635
3036
|
if (!autoStart) {
|
|
2636
|
-
throw new Error(`Ollama isn't running at ${
|
|
3037
|
+
throw new Error(`Ollama isn't running at ${OLLAMA_HOST2} and --no-auto-start was given.
|
|
2637
3038
|
${manualHint}`);
|
|
2638
3039
|
}
|
|
2639
3040
|
if (!dockerReady()) {
|
|
2640
3041
|
throw new Error(
|
|
2641
|
-
`Ollama isn't running at ${
|
|
3042
|
+
`Ollama isn't running at ${OLLAMA_HOST2}, and the Docker daemon isn't available to auto-start it.
|
|
2642
3043
|
${manualHint}`
|
|
2643
3044
|
);
|
|
2644
3045
|
}
|
|
@@ -2675,7 +3076,7 @@ ${manualHint}`);
|
|
|
2675
3076
|
}
|
|
2676
3077
|
async function ollamaPull(model) {
|
|
2677
3078
|
try {
|
|
2678
|
-
const show = await fetch(`${
|
|
3079
|
+
const show = await fetch(`${OLLAMA_HOST2}/api/show`, {
|
|
2679
3080
|
method: "POST",
|
|
2680
3081
|
headers: { "Content-Type": "application/json" },
|
|
2681
3082
|
body: JSON.stringify({ name: model })
|
|
@@ -2684,7 +3085,7 @@ async function ollamaPull(model) {
|
|
|
2684
3085
|
} catch {
|
|
2685
3086
|
}
|
|
2686
3087
|
console.log(`Pulling embedding model "${model}" into Ollama (first run only)\u2026`);
|
|
2687
|
-
const res = await fetch(`${
|
|
3088
|
+
const res = await fetch(`${OLLAMA_HOST2}/api/pull`, {
|
|
2688
3089
|
method: "POST",
|
|
2689
3090
|
headers: { "Content-Type": "application/json" },
|
|
2690
3091
|
body: JSON.stringify({ name: model, stream: false })
|
|
@@ -2695,7 +3096,7 @@ async function ollamaPull(model) {
|
|
|
2695
3096
|
async function embedOllama(model, inputs) {
|
|
2696
3097
|
const out = [];
|
|
2697
3098
|
for (const text of inputs) {
|
|
2698
|
-
const res = await fetch(`${
|
|
3099
|
+
const res = await fetch(`${OLLAMA_HOST2}/api/embeddings`, {
|
|
2699
3100
|
method: "POST",
|
|
2700
3101
|
headers: { "Content-Type": "application/json" },
|
|
2701
3102
|
body: JSON.stringify({ model, prompt: text })
|
|
@@ -2753,17 +3154,17 @@ async function embedOpenAI(model, inputs) {
|
|
|
2753
3154
|
}
|
|
2754
3155
|
async function loadJdf2(filePath) {
|
|
2755
3156
|
if (filePath.toLowerCase().endsWith(".jdfx")) {
|
|
2756
|
-
const zip = await JSZip.loadAsync(
|
|
3157
|
+
const zip = await JSZip.loadAsync(fs8.readFileSync(filePath));
|
|
2757
3158
|
const docFile = zip.file(JDFX_DOCUMENT_PATH);
|
|
2758
3159
|
if (!docFile) throw new Error(`Bundle missing ${JDFX_DOCUMENT_PATH}`);
|
|
2759
3160
|
return JSON.parse(await docFile.async("string"));
|
|
2760
3161
|
}
|
|
2761
|
-
return JSON.parse(
|
|
3162
|
+
return JSON.parse(fs8.readFileSync(filePath, "utf-8"));
|
|
2762
3163
|
}
|
|
2763
3164
|
function loadCache(cachePath) {
|
|
2764
3165
|
try {
|
|
2765
|
-
if (!
|
|
2766
|
-
return JSON.parse(
|
|
3166
|
+
if (!fs8.existsSync(cachePath)) return null;
|
|
3167
|
+
return JSON.parse(fs8.readFileSync(cachePath, "utf-8"));
|
|
2767
3168
|
} catch {
|
|
2768
3169
|
return null;
|
|
2769
3170
|
}
|
|
@@ -2774,15 +3175,15 @@ function batched(items, size) {
|
|
|
2774
3175
|
return out;
|
|
2775
3176
|
}
|
|
2776
3177
|
async function embedFile(inputPath, opts = {}) {
|
|
2777
|
-
const input =
|
|
2778
|
-
if (!
|
|
3178
|
+
const input = path8.resolve(inputPath);
|
|
3179
|
+
if (!fs8.existsSync(input)) throw new Error(`File not found: ${input}`);
|
|
2779
3180
|
const provider = opts.provider ?? "ollama";
|
|
2780
3181
|
const model = opts.model ?? DEFAULT_MODEL[provider];
|
|
2781
3182
|
const strategy = opts.strategy ?? "section";
|
|
2782
3183
|
const doc = await loadJdf2(input);
|
|
2783
3184
|
const chunks = chunkDocument(doc, { strategy, maxTokens: opts.maxTokens, transcriptWindowSec: opts.transcriptWindowSec });
|
|
2784
|
-
const output = opts.output ?
|
|
2785
|
-
const cachePath = opts.cache ?
|
|
3185
|
+
const output = opts.output ? path8.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".embeddings.json");
|
|
3186
|
+
const cachePath = opts.cache ? path8.resolve(opts.cache) : output;
|
|
2786
3187
|
console.log(`Embedding: ${input}`);
|
|
2787
3188
|
console.log(`Provider: ${provider} / ${model}${provider === "ollama" ? " (local \u2014 no data leaves this machine)" : " (remote API)"}`);
|
|
2788
3189
|
console.log(`Strategy: ${strategy} \u2192 ${chunks.length} chunks`);
|
|
@@ -2821,7 +3222,7 @@ async function embedFile(inputPath, opts = {}) {
|
|
|
2821
3222
|
chunker: `jdf-${strategy}-v1`,
|
|
2822
3223
|
vectors
|
|
2823
3224
|
};
|
|
2824
|
-
|
|
3225
|
+
fs8.writeFileSync(output, JSON.stringify(sidecar));
|
|
2825
3226
|
console.log(`
|
|
2826
3227
|
Done! ${Object.keys(vectors).length} vectors (${dims}-dim) \u2192 ${output}`);
|
|
2827
3228
|
return sidecar;
|
|
@@ -2865,9 +3266,9 @@ function parseChapters(text) {
|
|
|
2865
3266
|
return { t: toSec(m[1]), title: m[2].trim() };
|
|
2866
3267
|
});
|
|
2867
3268
|
}
|
|
2868
|
-
async function
|
|
3269
|
+
async function loadDoc2(file) {
|
|
2869
3270
|
if (file.toLowerCase().endsWith(".jdfx")) {
|
|
2870
|
-
const zip = await JSZip.loadAsync(
|
|
3271
|
+
const zip = await JSZip.loadAsync(fs8.readFileSync(file));
|
|
2871
3272
|
const f = zip.file(JDFX_DOCUMENT_PATH);
|
|
2872
3273
|
if (!f) throw new Error(`Bundle missing ${JDFX_DOCUMENT_PATH}`);
|
|
2873
3274
|
const doc = JSON.parse(await f.async("string"));
|
|
@@ -2883,7 +3284,7 @@ async function loadDoc(file) {
|
|
|
2883
3284
|
}
|
|
2884
3285
|
return { doc, bundle: true, zip };
|
|
2885
3286
|
}
|
|
2886
|
-
return { doc: JSON.parse(
|
|
3287
|
+
return { doc: JSON.parse(fs8.readFileSync(file, "utf-8")), bundle: false };
|
|
2887
3288
|
}
|
|
2888
3289
|
function findVideos(doc) {
|
|
2889
3290
|
const out = [];
|
|
@@ -2897,27 +3298,27 @@ function findVideos(doc) {
|
|
|
2897
3298
|
return out;
|
|
2898
3299
|
}
|
|
2899
3300
|
async function clipToTempFile(doc, el, docDir) {
|
|
2900
|
-
const tmp =
|
|
3301
|
+
const tmp = path8.join(fs8.mkdtempSync(path8.join(__require("os").tmpdir(), "jdf-transcribe-")), "clip.mp4");
|
|
2901
3302
|
const res = el.resource ? doc.resources?.videos?.[el.resource] ?? doc.resources?.images?.[el.resource] : void 0;
|
|
2902
3303
|
if (res?.data) {
|
|
2903
|
-
|
|
3304
|
+
fs8.writeFileSync(tmp, Buffer.from(res.data.replace(/^data:[^,]*,/, ""), "base64"));
|
|
2904
3305
|
return tmp;
|
|
2905
3306
|
}
|
|
2906
|
-
if (res?.path) return
|
|
3307
|
+
if (res?.path) return path8.resolve(docDir, res.path);
|
|
2907
3308
|
const src = el.src;
|
|
2908
3309
|
if (!src) return null;
|
|
2909
3310
|
if (src.startsWith("data:")) {
|
|
2910
|
-
|
|
3311
|
+
fs8.writeFileSync(tmp, Buffer.from(src.replace(/^data:[^,]*,/, ""), "base64"));
|
|
2911
3312
|
return tmp;
|
|
2912
3313
|
}
|
|
2913
3314
|
if (/^https?:\/\//i.test(src)) {
|
|
2914
3315
|
const r = await fetch(src);
|
|
2915
3316
|
if (!r.ok) throw new Error(`download failed ${r.status}: ${src}`);
|
|
2916
|
-
|
|
3317
|
+
fs8.writeFileSync(tmp, Buffer.from(await r.arrayBuffer()));
|
|
2917
3318
|
return tmp;
|
|
2918
3319
|
}
|
|
2919
|
-
const local =
|
|
2920
|
-
return
|
|
3320
|
+
const local = path8.resolve(docDir, src);
|
|
3321
|
+
return fs8.existsSync(local) ? local : null;
|
|
2921
3322
|
}
|
|
2922
3323
|
function whisperCli(clip, model, language, prompt2) {
|
|
2923
3324
|
const ffmpeg = spawnSync("ffmpeg", ["-version"]);
|
|
@@ -2932,7 +3333,7 @@ function whisperCli(clip, model, language, prompt2) {
|
|
|
2932
3333
|
const run = spawnSync("whisper-cli", args, { encoding: "utf-8" });
|
|
2933
3334
|
if (run.error) throw new Error("whisper-cli not found \u2014 install whisper.cpp (brew install whisper-cpp) or use --from / --provider openai");
|
|
2934
3335
|
if (run.status !== 0) throw new Error(`whisper-cli failed: ${run.stderr.slice(-400)}`);
|
|
2935
|
-
const j = JSON.parse(
|
|
3336
|
+
const j = JSON.parse(fs8.readFileSync(wav.replace(/\.wav$/, "") + ".json", "utf-8"));
|
|
2936
3337
|
const segs = j.transcription ?? j.segments ?? [];
|
|
2937
3338
|
const ms = (x) => typeof x === "number" ? x / 1e3 : toSec(String(x).replace(",", "."));
|
|
2938
3339
|
return segs.map((sg) => ({ t0: ms(sg.offsets?.from ?? sg.start), t1: ms(sg.offsets?.to ?? sg.end), text: String(sg.text ?? "").trim() })).filter((sg) => sg.text);
|
|
@@ -2942,7 +3343,7 @@ async function openaiTranscribe(clip, model, language, prompt2) {
|
|
|
2942
3343
|
if (!key) throw new Error("OPENAI_API_KEY is not set");
|
|
2943
3344
|
const base = process.env.OPENAI_BASE_URL || "https://api.openai.com/v1";
|
|
2944
3345
|
const form = new FormData();
|
|
2945
|
-
form.append("file", new Blob([
|
|
3346
|
+
form.append("file", new Blob([fs8.readFileSync(clip)]), path8.basename(clip));
|
|
2946
3347
|
form.append("model", model || "whisper-1");
|
|
2947
3348
|
form.append("response_format", "verbose_json");
|
|
2948
3349
|
form.append("timestamp_granularities[]", "segment");
|
|
@@ -2954,9 +3355,9 @@ async function openaiTranscribe(clip, model, language, prompt2) {
|
|
|
2954
3355
|
return (j.segments ?? []).map((sg) => ({ t0: Number(sg.start), t1: Number(sg.end), text: String(sg.text).trim() })).filter((sg) => sg.text);
|
|
2955
3356
|
}
|
|
2956
3357
|
async function transcribeFile(inputPath, opts = {}) {
|
|
2957
|
-
const input =
|
|
2958
|
-
if (!
|
|
2959
|
-
const { doc, bundle } = await
|
|
3358
|
+
const input = path8.resolve(inputPath);
|
|
3359
|
+
if (!fs8.existsSync(input)) throw new Error(`File not found: ${input}`);
|
|
3360
|
+
const { doc, bundle } = await loadDoc2(input);
|
|
2960
3361
|
const videos = findVideos(doc);
|
|
2961
3362
|
if (!videos.length) throw new Error("document has no video element");
|
|
2962
3363
|
let target = videos[0];
|
|
@@ -2972,40 +3373,40 @@ async function transcribeFile(inputPath, opts = {}) {
|
|
|
2972
3373
|
let segments;
|
|
2973
3374
|
let source;
|
|
2974
3375
|
if (opts.from) {
|
|
2975
|
-
segments = parseSubtitles(
|
|
2976
|
-
source = `${
|
|
3376
|
+
segments = parseSubtitles(fs8.readFileSync(path8.resolve(opts.from), "utf-8"), opts.from);
|
|
3377
|
+
source = `${path8.extname(opts.from).slice(1).toLowerCase() || "file"}-import`;
|
|
2977
3378
|
} else {
|
|
2978
3379
|
const provider = opts.provider ?? "whisper-cli";
|
|
2979
|
-
const clip = await clipToTempFile(doc, target.el,
|
|
3380
|
+
const clip = await clipToTempFile(doc, target.el, path8.dirname(input));
|
|
2980
3381
|
if (!clip) throw new Error("could not locate the clip bytes (no bundled asset, data URL, local path or http URL) \u2014 use --from to import subtitles instead");
|
|
2981
3382
|
segments = provider === "openai" ? await openaiTranscribe(clip, opts.model, opts.language, opts.prompt) : whisperCli(clip, opts.model, opts.language, opts.prompt);
|
|
2982
|
-
source = provider === "openai" ? `openai:${opts.model || "whisper-1"}` : `whisper-cli${opts.model ? ":" +
|
|
3383
|
+
source = provider === "openai" ? `openai:${opts.model || "whisper-1"}` : `whisper-cli${opts.model ? ":" + path8.basename(opts.model) : ""}`;
|
|
2983
3384
|
}
|
|
2984
3385
|
segments.sort((a, b) => a.t0 - b.t0);
|
|
2985
3386
|
const transcript = { ...opts.language ? { language: opts.language } : {}, source, created: (/* @__PURE__ */ new Date()).toISOString(), segments };
|
|
2986
3387
|
target.el.transcript = transcript;
|
|
2987
|
-
if (opts.chapters) target.el.chapters = parseChapters(
|
|
3388
|
+
if (opts.chapters) target.el.chapters = parseChapters(fs8.readFileSync(path8.resolve(opts.chapters), "utf-8"));
|
|
2988
3389
|
if (!target.el.id) target.el.id = `video-${target.index + 1}`;
|
|
2989
|
-
const output = opts.output ?
|
|
3390
|
+
const output = opts.output ? path8.resolve(opts.output) : input;
|
|
2990
3391
|
if (output.toLowerCase().endsWith(".jdfx") || bundle && !opts.output) {
|
|
2991
3392
|
const { bytes } = await packJdfx(doc);
|
|
2992
|
-
|
|
3393
|
+
fs8.writeFileSync(output, bytes);
|
|
2993
3394
|
} else {
|
|
2994
|
-
|
|
3395
|
+
fs8.writeFileSync(output, JSON.stringify(doc, null, 2));
|
|
2995
3396
|
}
|
|
2996
3397
|
const dur = segments.length ? segments[segments.length - 1].t1 : 0;
|
|
2997
|
-
console.log(`Transcribed: ${
|
|
3398
|
+
console.log(`Transcribed: ${path8.basename(input)} \u2192 element "${target.el.id}" (${segments.length} segments, ${Math.round(dur)} s, source ${source})`);
|
|
2998
3399
|
if (target.el.chapters) console.log(`Chapters: ${target.el.chapters.length}`);
|
|
2999
3400
|
console.log(`Output: ${output}
|
|
3000
|
-
Next: jdf chunk ${
|
|
3401
|
+
Next: jdf chunk ${path8.basename(output)} # transcript \u2192 time-windowed chunks with media.t0/t1`);
|
|
3001
3402
|
return transcript;
|
|
3002
3403
|
}
|
|
3003
3404
|
var CONFIG_NAME = "jdf.rag.json";
|
|
3004
3405
|
var OUT_DIR = ".jdf-rag";
|
|
3005
3406
|
function walk(dir, acc = []) {
|
|
3006
|
-
for (const ent of
|
|
3407
|
+
for (const ent of fs8.readdirSync(dir, { withFileTypes: true })) {
|
|
3007
3408
|
if (ent.name === "node_modules" || ent.name === OUT_DIR || ent.name.startsWith(".")) continue;
|
|
3008
|
-
const p =
|
|
3409
|
+
const p = path8.join(dir, ent.name);
|
|
3009
3410
|
if (ent.isDirectory()) walk(p, acc);
|
|
3010
3411
|
else if (/\.(jdf|jdfx)$/i.test(ent.name)) acc.push(p);
|
|
3011
3412
|
}
|
|
@@ -3013,10 +3414,10 @@ function walk(dir, acc = []) {
|
|
|
3013
3414
|
}
|
|
3014
3415
|
async function readDoc(file) {
|
|
3015
3416
|
if (file.toLowerCase().endsWith(".jdfx")) {
|
|
3016
|
-
const zip = await JSZip.loadAsync(
|
|
3417
|
+
const zip = await JSZip.loadAsync(fs8.readFileSync(file));
|
|
3017
3418
|
return JSON.parse(await zip.file(JDFX_DOCUMENT_PATH).async("string"));
|
|
3018
3419
|
}
|
|
3019
|
-
return JSON.parse(
|
|
3420
|
+
return JSON.parse(fs8.readFileSync(file, "utf-8"));
|
|
3020
3421
|
}
|
|
3021
3422
|
function videosIn(doc) {
|
|
3022
3423
|
const out = [];
|
|
@@ -3030,29 +3431,44 @@ function videosIn(doc) {
|
|
|
3030
3431
|
return out;
|
|
3031
3432
|
}
|
|
3032
3433
|
async function ragFolder(dirPath, cli = {}) {
|
|
3033
|
-
const dir =
|
|
3034
|
-
if (!
|
|
3035
|
-
const cfgPath =
|
|
3036
|
-
const cfg =
|
|
3434
|
+
const dir = path8.resolve(dirPath);
|
|
3435
|
+
if (!fs8.existsSync(dir) || !fs8.statSync(dir).isDirectory()) throw new Error(`Not a directory: ${dir}`);
|
|
3436
|
+
const cfgPath = path8.join(dir, CONFIG_NAME);
|
|
3437
|
+
const cfg = fs8.existsSync(cfgPath) ? JSON.parse(fs8.readFileSync(cfgPath, "utf-8")) : {};
|
|
3037
3438
|
const opts = { ...cfg, ...Object.fromEntries(Object.entries(cli).filter(([, v]) => v !== void 0)) };
|
|
3038
3439
|
const provider = opts.provider ?? "ollama";
|
|
3039
3440
|
const transcribe = opts.transcribe ?? "none";
|
|
3040
|
-
const outDir =
|
|
3441
|
+
const outDir = path8.resolve(opts.out ?? path8.join(dir, OUT_DIR));
|
|
3442
|
+
const ocr = opts.ocr ?? "none";
|
|
3443
|
+
const caption = opts.caption ?? "none";
|
|
3041
3444
|
const files = walk(dir);
|
|
3042
3445
|
console.log(`jdf rag: ${dir}
|
|
3043
|
-
files: ${files.length} (.jdf/.jdfx)${
|
|
3446
|
+
files: ${files.length} (.jdf/.jdfx)${fs8.existsSync(cfgPath) ? `
|
|
3044
3447
|
config: ${CONFIG_NAME}` : ""}
|
|
3045
3448
|
embeddings: ${opts.noEmbed ? "skipped (--no-embed)" : `${provider}${opts.model ? " / " + opts.model : ""}`}
|
|
3046
|
-
transcribe: ${transcribe}${opts.dryRun ? "\n DRY RUN \u2014 nothing written" : ""}
|
|
3449
|
+
transcribe: ${transcribe} ocr: ${ocr} caption: ${caption}${opts.dryRun ? "\n DRY RUN \u2014 nothing written" : ""}
|
|
3047
3450
|
`);
|
|
3048
3451
|
if (!files.length) {
|
|
3049
3452
|
console.log("Nothing to do.");
|
|
3050
3453
|
return;
|
|
3051
3454
|
}
|
|
3052
|
-
const manifest = {
|
|
3455
|
+
const manifest = {
|
|
3456
|
+
dir,
|
|
3457
|
+
created: (/* @__PURE__ */ new Date()).toISOString(),
|
|
3458
|
+
provider: opts.noEmbed ? null : provider,
|
|
3459
|
+
model: opts.model ?? null,
|
|
3460
|
+
strategy: opts.strategy ?? "section",
|
|
3461
|
+
transcribe,
|
|
3462
|
+
ocr,
|
|
3463
|
+
caption,
|
|
3464
|
+
files: [],
|
|
3465
|
+
totals: { files: files.length, chunks: 0, videoChunks: 0, videos: 0, transcribed: 0, untranscribed: 0, images: 0, described: 0, imagesWithoutText: 0 },
|
|
3466
|
+
/** Every media element that retrieval would still skip, by file — the thing to fix before shipping an index. */
|
|
3467
|
+
mediaWithoutText: []
|
|
3468
|
+
};
|
|
3053
3469
|
const indexLines = [];
|
|
3054
3470
|
for (const file of files) {
|
|
3055
|
-
const rel =
|
|
3471
|
+
const rel = path8.relative(dir, file);
|
|
3056
3472
|
const doc = await readDoc(file);
|
|
3057
3473
|
const vids = videosIn(doc);
|
|
3058
3474
|
let transcribedHere = 0;
|
|
@@ -3076,20 +3492,36 @@ async function ragFolder(dirPath, cli = {}) {
|
|
|
3076
3492
|
}
|
|
3077
3493
|
manifest.totals.videos += vids.length;
|
|
3078
3494
|
manifest.totals.transcribed += transcribedHere;
|
|
3495
|
+
const covBefore = mediaCoverage(doc);
|
|
3496
|
+
let describedHere = 0;
|
|
3497
|
+
if (covBefore.images.missing.length && (ocr !== "none" || caption !== "none") && !opts.dryRun) {
|
|
3498
|
+
try {
|
|
3499
|
+
await describeFile(file, { ocr, caption, captionModel: opts.captionModel, quiet: true });
|
|
3500
|
+
describedHere = covBefore.images.missing.length;
|
|
3501
|
+
} catch (e) {
|
|
3502
|
+
console.warn(` ! ${rel}: describe failed: ${e.message}`);
|
|
3503
|
+
}
|
|
3504
|
+
}
|
|
3505
|
+
const covAfter = transcribedHere || describedHere ? mediaCoverage(await readDoc(file)) : covBefore;
|
|
3506
|
+
manifest.totals.images += covAfter.images.total;
|
|
3507
|
+
manifest.totals.described += describedHere;
|
|
3508
|
+
manifest.totals.imagesWithoutText += covAfter.images.missing.length;
|
|
3509
|
+
for (const m of covAfter.images.missing) manifest.mediaWithoutText.push({ file: rel, type: "image", ...m });
|
|
3510
|
+
for (const m of covAfter.videos.missing) manifest.mediaWithoutText.push({ file: rel, type: "video", ...m });
|
|
3079
3511
|
if (opts.dryRun) {
|
|
3080
3512
|
manifest.files.push({ file: rel, videos: vids.length, wouldTranscribe: transcribedHere });
|
|
3081
3513
|
continue;
|
|
3082
3514
|
}
|
|
3083
3515
|
const chunkOpts = { strategy: opts.strategy, maxTokens: opts.maxTokens, transcriptWindowSec: opts.transcriptWindowSec };
|
|
3084
|
-
const chunkOut =
|
|
3085
|
-
|
|
3516
|
+
const chunkOut = path8.join(outDir, "chunks", rel.replace(/\.(jdf|jdfx)$/i, ".chunks.jsonl"));
|
|
3517
|
+
fs8.mkdirSync(path8.dirname(chunkOut), { recursive: true });
|
|
3086
3518
|
let chunks;
|
|
3087
3519
|
if (opts.noEmbed) {
|
|
3088
3520
|
chunks = await chunkFile(file, { ...chunkOpts, format: "jsonl", output: chunkOut });
|
|
3089
3521
|
} else {
|
|
3090
3522
|
const side = await embedFile(file, { ...chunkOpts, provider, model: opts.model, incremental: true });
|
|
3091
3523
|
chunks = await chunkFile(file, { ...chunkOpts, format: "jsonl", output: chunkOut });
|
|
3092
|
-
manifest.files.push({ file: rel, chunks: chunks.length, vectors: Object.keys(side.vectors).length, sidecar:
|
|
3524
|
+
manifest.files.push({ file: rel, chunks: chunks.length, vectors: Object.keys(side.vectors).length, sidecar: path8.relative(dir, file.replace(/\.(jdf|jdfx)$/i, ".embeddings.json")), videos: vids.length, transcribed: transcribedHere });
|
|
3093
3525
|
}
|
|
3094
3526
|
if (opts.noEmbed) manifest.files.push({ file: rel, chunks: chunks.length, videos: vids.length, transcribed: transcribedHere });
|
|
3095
3527
|
for (const c of chunks) {
|
|
@@ -3099,15 +3531,27 @@ async function ragFolder(dirPath, cli = {}) {
|
|
|
3099
3531
|
}
|
|
3100
3532
|
}
|
|
3101
3533
|
if (!opts.dryRun) {
|
|
3102
|
-
|
|
3103
|
-
|
|
3104
|
-
|
|
3534
|
+
fs8.mkdirSync(outDir, { recursive: true });
|
|
3535
|
+
fs8.writeFileSync(path8.join(outDir, "index.jsonl"), indexLines.join("\n") + (indexLines.length ? "\n" : ""));
|
|
3536
|
+
fs8.writeFileSync(path8.join(outDir, "manifest.json"), JSON.stringify(manifest, null, 2) + "\n");
|
|
3105
3537
|
}
|
|
3106
3538
|
const t = manifest.totals;
|
|
3107
3539
|
console.log(`
|
|
3108
|
-
Done. ${t.files} files \u2192 ${t.chunks} chunks (${t.videoChunks} from video transcripts)
|
|
3109
|
-
|
|
3110
|
-
|
|
3540
|
+
Done. ${t.files} files \u2192 ${t.chunks} chunks (${t.videoChunks} from video transcripts).`);
|
|
3541
|
+
console.log(`Media coverage: videos ${t.videos - t.untranscribed}/${t.videos} with transcript (transcribed now ${t.transcribed}), images ${t.images - t.imagesWithoutText}/${t.images} with caption/OCR (described now ${t.described}).`);
|
|
3542
|
+
if (manifest.mediaWithoutText.length) {
|
|
3543
|
+
console.log(`
|
|
3544
|
+
! ${manifest.mediaWithoutText.length} media element(s) still have NO text \u2014 retrieval will skip them:`);
|
|
3545
|
+
for (const m of manifest.mediaWithoutText.slice(0, 12)) console.log(` ${m.file} \xB7 ${m.type} ${m.id ?? ""} (page ${m.page})${m.title ? ` "${m.title}"` : m.alt ? ` alt="${m.alt}"` : ""}`);
|
|
3546
|
+
if (manifest.mediaWithoutText.length > 12) console.log(` \u2026 ${manifest.mediaWithoutText.length - 12} more in manifest.json`);
|
|
3547
|
+
console.log(` fix: jdf rag <dir> --transcribe whisper-cli|openai --ocr tesseract --caption ollama (or jdf transcribe / jdf describe per file)`);
|
|
3548
|
+
if (opts.strict) {
|
|
3549
|
+
console.error(`--strict: failing because media without text remains.`);
|
|
3550
|
+
process.exitCode = 1;
|
|
3551
|
+
}
|
|
3552
|
+
}
|
|
3553
|
+
if (!opts.dryRun) console.log(`Index: ${path8.join(outDir, "index.jsonl")}
|
|
3554
|
+
Report: ${path8.join(outDir, "manifest.json")}${opts.noEmbed ? "" : `
|
|
3111
3555
|
Vectors: one <file>.embeddings.json next to each document (incremental \u2014 re-run any time)`}`);
|
|
3112
3556
|
}
|
|
3113
3557
|
|
|
@@ -3123,8 +3567,11 @@ The CLI exists for these workflows:
|
|
|
3123
3567
|
\u2022 JDF \u2192 vectors embed those chunks, incrementally, for a vector store.
|
|
3124
3568
|
\u2022 video \u2192 text attach a time-stamped transcript to a video element so
|
|
3125
3569
|
RAG retrieves "video at 02:13", not just "a video".
|
|
3570
|
+
\u2022 image \u2192 text OCR + a vision caption for every image so charts and
|
|
3571
|
+
scanned pages are retrievable, not skipped.
|
|
3126
3572
|
\u2022 folder \u2192 index one command over a directory of .jdf/.jdfx: transcribe,
|
|
3127
|
-
chunk, embed incrementally, write .jdf-rag/index.jsonl.
|
|
3573
|
+
describe, chunk, embed incrementally, write .jdf-rag/index.jsonl.
|
|
3574
|
+
Reports media coverage; --strict fails when anything has no text.
|
|
3128
3575
|
|
|
3129
3576
|
Usage:
|
|
3130
3577
|
jdf validate <file.jdf>
|
|
@@ -3132,7 +3579,8 @@ Usage:
|
|
|
3132
3579
|
jdf chunk <file.{jdf,jdfx}> [--strategy section|element|fixed] [--format jsonl|json|inline] [--max-tokens N] [-o out]
|
|
3133
3580
|
jdf embed <file.{jdf,jdfx}> [--provider ollama|openai] [--model NAME] [--strategy \u2026] [--incremental] [-o out]
|
|
3134
3581
|
jdf transcribe <file.{jdf,jdfx}> [--from subs.srt|.vtt|.json] [--provider whisper-cli|openai] [--element ID] [--chapters FILE] [-o out]
|
|
3135
|
-
jdf
|
|
3582
|
+
jdf describe <file.{jdf,jdfx}> [--ocr tesseract|openai|none] [--caption ollama|openai|none] [--caption-model M] [--ocr-language eng] [--element ID] [--force] [-o out]
|
|
3583
|
+
jdf rag <dir> [--provider ollama|openai] [--model NAME] [--transcribe none|whisper-cli|openai] [--ocr none|tesseract|openai] [--caption none|ollama|openai] [--strict] [--no-embed] [--dry-run] [--out DIR]
|
|
3136
3584
|
jdf --help
|
|
3137
3585
|
|
|
3138
3586
|
Commands:
|
|
@@ -3141,7 +3589,8 @@ Commands:
|
|
|
3141
3589
|
chunk Split a JDF document into retrieval-ready chunks (offline, deterministic)
|
|
3142
3590
|
embed Compute embeddings for the chunks (local via Ollama by default)
|
|
3143
3591
|
transcribe Store time-stamped text on a video element (import SRT/VTT/JSON, or run Whisper)
|
|
3144
|
-
|
|
3592
|
+
describe Give images text: OCR blocks (tesseract.js, local) + a caption (Ollama vision model, local)
|
|
3593
|
+
rag Make a whole folder retrieval-ready (finds .jdf/.jdfx, transcribes, describes, chunks, embeds, indexes)
|
|
3145
3594
|
|
|
3146
3595
|
Flags:
|
|
3147
3596
|
-o, --output <path> Explicit output path
|
|
@@ -3168,6 +3617,12 @@ Flags:
|
|
|
3168
3617
|
--prompt <text> transcribe: Whisper vocabulary hint (names, acronyms) \u2014 not an instruction
|
|
3169
3618
|
--window <sec> chunk/embed/rag: transcript window per video chunk (default 45)
|
|
3170
3619
|
--transcribe <p> rag: none (default) | whisper-cli | openai \u2014 for videos that have no transcript yet
|
|
3620
|
+
--ocr <p> describe/rag/convert: tesseract (local WASM) | openai | none
|
|
3621
|
+
--caption <p> describe/rag: ollama (local vision model, default moondream) | openai | none
|
|
3622
|
+
--caption-model describe/rag: vision model name (ollama: qwen2.5vl:3b, llava\u2026; openai: gpt-4o-mini\u2026)
|
|
3623
|
+
--ocr-language describe: tesseract language(s), e.g. eng, tur, eng+tur (default eng)
|
|
3624
|
+
--force describe: redo images that already have text
|
|
3625
|
+
--strict rag: exit 1 if any image/video is still without text after the run
|
|
3171
3626
|
--no-embed rag: chunk + index only
|
|
3172
3627
|
--dry-run rag: list what would happen, write nothing
|
|
3173
3628
|
--out <dir> rag: index folder (default <dir>/.jdf-rag)
|
|
@@ -3187,9 +3642,11 @@ Examples:
|
|
|
3187
3642
|
jdf embed report.jdf --provider openai --incremental
|
|
3188
3643
|
jdf transcribe talk.jdfx --from talk.srt --chapters chapters.txt # then: jdf chunk talk.jdfx
|
|
3189
3644
|
jdf transcribe talk.jdfx --provider openai --language en --prompt "JDF, jdfx, Ollama"
|
|
3190
|
-
jdf
|
|
3645
|
+
jdf describe report.jdfx # OCR (tesseract) + caption (Ollama qwen2.5vl), local
|
|
3646
|
+
jdf convert scan.pdf --ocr tesseract # scanned pages get OCR text instead of silence
|
|
3647
|
+
jdf rag ./knowledge-base --transcribe openai --ocr tesseract --caption ollama --strict
|
|
3191
3648
|
`;
|
|
3192
|
-
var BOOLEAN_FLAGS = /* @__PURE__ */ new Set(["help", "h", "json", "verbose", "skip-validate", "incremental", "no-auto-start", "drop-invisible-text", "no-embed", "dry-run"]);
|
|
3649
|
+
var BOOLEAN_FLAGS = /* @__PURE__ */ new Set(["help", "h", "json", "verbose", "skip-validate", "incremental", "no-auto-start", "drop-invisible-text", "no-embed", "dry-run", "force", "strict"]);
|
|
3193
3650
|
function parseArgs(argv) {
|
|
3194
3651
|
const positional = [];
|
|
3195
3652
|
const flags = {};
|
|
@@ -3272,7 +3729,8 @@ async function main() {
|
|
|
3272
3729
|
await importPdf(input, output, {
|
|
3273
3730
|
forceJson,
|
|
3274
3731
|
password: typeof flags.password === "string" ? flags.password : void 0,
|
|
3275
|
-
dropInvisibleText: flags["drop-invisible-text"] === true
|
|
3732
|
+
dropInvisibleText: flags["drop-invisible-text"] === true,
|
|
3733
|
+
ocr: typeof flags.ocr === "string" ? flags.ocr : void 0
|
|
3276
3734
|
});
|
|
3277
3735
|
process.exit(0);
|
|
3278
3736
|
} else if (lower.endsWith(".json")) {
|
|
@@ -3316,6 +3774,23 @@ async function main() {
|
|
|
3316
3774
|
});
|
|
3317
3775
|
process.exit(0);
|
|
3318
3776
|
}
|
|
3777
|
+
case "describe": {
|
|
3778
|
+
const input = positional[0];
|
|
3779
|
+
if (!input) {
|
|
3780
|
+
console.error("Usage: jdf describe <file.{jdf,jdfx}> [--ocr tesseract|openai|none] [--caption ollama|openai|none] [--caption-model M] [--ocr-language eng] [--element id|n] [--force] [-o out]");
|
|
3781
|
+
process.exit(1);
|
|
3782
|
+
}
|
|
3783
|
+
await describeFile(input, {
|
|
3784
|
+
ocr: typeof flags.ocr === "string" ? flags.ocr : void 0,
|
|
3785
|
+
caption: typeof flags.caption === "string" ? flags.caption : void 0,
|
|
3786
|
+
captionModel: typeof flags["caption-model"] === "string" ? flags["caption-model"] : void 0,
|
|
3787
|
+
ocrLanguage: typeof flags["ocr-language"] === "string" ? flags["ocr-language"] : void 0,
|
|
3788
|
+
element: typeof flags.element === "string" ? flags.element : void 0,
|
|
3789
|
+
force: flags.force === true,
|
|
3790
|
+
output: typeof flags.output === "string" ? flags.output : void 0
|
|
3791
|
+
});
|
|
3792
|
+
process.exit(process.exitCode ?? 0);
|
|
3793
|
+
}
|
|
3319
3794
|
case "rag": {
|
|
3320
3795
|
const input = positional[0];
|
|
3321
3796
|
if (!input) {
|
|
@@ -3332,11 +3807,15 @@ async function main() {
|
|
|
3332
3807
|
transcribeModel: typeof flags["transcribe-model"] === "string" ? flags["transcribe-model"] : void 0,
|
|
3333
3808
|
language: typeof flags.language === "string" ? flags.language : void 0,
|
|
3334
3809
|
prompt: typeof flags.prompt === "string" ? flags.prompt : void 0,
|
|
3810
|
+
ocr: typeof flags.ocr === "string" ? flags.ocr : void 0,
|
|
3811
|
+
caption: typeof flags.caption === "string" ? flags.caption : void 0,
|
|
3812
|
+
captionModel: typeof flags["caption-model"] === "string" ? flags["caption-model"] : void 0,
|
|
3813
|
+
strict: flags.strict === true,
|
|
3335
3814
|
noEmbed: flags["no-embed"] === true,
|
|
3336
3815
|
dryRun: flags["dry-run"] === true,
|
|
3337
3816
|
out: typeof flags.out === "string" ? flags.out : void 0
|
|
3338
3817
|
});
|
|
3339
|
-
process.exit(0);
|
|
3818
|
+
process.exit(process.exitCode ?? 0);
|
|
3340
3819
|
}
|
|
3341
3820
|
case "embed": {
|
|
3342
3821
|
const input = positional[0];
|