@uurtech/jdf-cli 0.1.26 → 0.2.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.js +952 -159
- package/dist/jdf-schema.json +33 -2
- package/package.json +1 -1
package/dist/index.js
CHANGED
|
@@ -1,13 +1,20 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
|
-
import
|
|
3
|
-
import
|
|
2
|
+
import fs7 from 'fs';
|
|
3
|
+
import path7 from 'path';
|
|
4
4
|
import { fileURLToPath } from 'url';
|
|
5
5
|
import Ajv from 'ajv';
|
|
6
6
|
import addFormats from 'ajv-formats';
|
|
7
7
|
import JSZip from 'jszip';
|
|
8
8
|
import crypto, { createHash } from 'crypto';
|
|
9
9
|
import { readFile } from 'fs/promises';
|
|
10
|
-
import { execFileSync } from 'child_process';
|
|
10
|
+
import { execFileSync, spawnSync } from 'child_process';
|
|
11
|
+
|
|
12
|
+
var __require = /* @__PURE__ */ ((x) => typeof require !== "undefined" ? require : typeof Proxy !== "undefined" ? new Proxy(x, {
|
|
13
|
+
get: (a, b) => (typeof require !== "undefined" ? require : a)[b]
|
|
14
|
+
}) : x)(function(x) {
|
|
15
|
+
if (typeof require !== "undefined") return require.apply(this, arguments);
|
|
16
|
+
throw Error('Dynamic require of "' + x + '" is not supported');
|
|
17
|
+
});
|
|
11
18
|
|
|
12
19
|
// ../../packages/jdf-core/src/manifest.ts
|
|
13
20
|
var JDFX_MANIFEST_VERSION = "1.0.0";
|
|
@@ -16,17 +23,17 @@ var JDFX_MANIFEST_PATH = "manifest.json";
|
|
|
16
23
|
var JDFX_ASSET_DIR = "assets";
|
|
17
24
|
|
|
18
25
|
// src/commands/validate.ts
|
|
19
|
-
var __dirname$1 =
|
|
26
|
+
var __dirname$1 = path7.dirname(fileURLToPath(import.meta.url));
|
|
20
27
|
function resolveSchemaPath() {
|
|
21
|
-
const bundled =
|
|
22
|
-
if (
|
|
23
|
-
const dev =
|
|
28
|
+
const bundled = path7.resolve(__dirname$1, "jdf-schema.json");
|
|
29
|
+
if (fs7.existsSync(bundled)) return bundled;
|
|
30
|
+
const dev = path7.resolve(__dirname$1, "../../../../spec/jdf-schema.json");
|
|
24
31
|
return dev;
|
|
25
32
|
}
|
|
26
33
|
var SCHEMA_PATH = resolveSchemaPath();
|
|
27
34
|
async function loadDocument(filePath) {
|
|
28
35
|
if (filePath.toLowerCase().endsWith(".jdfx")) {
|
|
29
|
-
const zip = await JSZip.loadAsync(
|
|
36
|
+
const zip = await JSZip.loadAsync(fs7.readFileSync(filePath));
|
|
30
37
|
const docFile = zip.file(JDFX_DOCUMENT_PATH);
|
|
31
38
|
if (!docFile) {
|
|
32
39
|
console.error(`\u2717 Bundle missing ${JDFX_DOCUMENT_PATH}`);
|
|
@@ -51,11 +58,11 @@ async function loadDocument(filePath) {
|
|
|
51
58
|
}
|
|
52
59
|
return { doc, bundle: { manifest, assetCount } };
|
|
53
60
|
}
|
|
54
|
-
return { doc: JSON.parse(
|
|
61
|
+
return { doc: JSON.parse(fs7.readFileSync(filePath, "utf-8")) };
|
|
55
62
|
}
|
|
56
63
|
async function validate(file) {
|
|
57
|
-
const filePath =
|
|
58
|
-
if (!
|
|
64
|
+
const filePath = path7.resolve(file);
|
|
65
|
+
if (!fs7.existsSync(filePath)) {
|
|
59
66
|
console.error(`File not found: ${filePath}`);
|
|
60
67
|
return false;
|
|
61
68
|
}
|
|
@@ -68,11 +75,11 @@ async function validate(file) {
|
|
|
68
75
|
}
|
|
69
76
|
if (!loaded) return false;
|
|
70
77
|
const { doc, bundle } = loaded;
|
|
71
|
-
if (!
|
|
78
|
+
if (!fs7.existsSync(SCHEMA_PATH)) {
|
|
72
79
|
console.error(`Schema not found at ${SCHEMA_PATH}`);
|
|
73
80
|
return false;
|
|
74
81
|
}
|
|
75
|
-
const schema = JSON.parse(
|
|
82
|
+
const schema = JSON.parse(fs7.readFileSync(SCHEMA_PATH, "utf-8"));
|
|
76
83
|
const ajv = new Ajv({ allErrors: true, strict: false });
|
|
77
84
|
addFormats(ajv);
|
|
78
85
|
const validateFn = ajv.compile(schema);
|
|
@@ -81,7 +88,7 @@ async function validate(file) {
|
|
|
81
88
|
const d = doc;
|
|
82
89
|
const pageCount = Array.isArray(d.pages) ? d.pages.length : 0;
|
|
83
90
|
const elCount = Array.isArray(d.pages) ? d.pages.reduce((acc, p) => acc + (Array.isArray(p?.elements) ? p.elements.length : 0), 0) : 0;
|
|
84
|
-
console.log(`\u2713 Valid: ${
|
|
91
|
+
console.log(`\u2713 Valid: ${path7.basename(filePath)}`);
|
|
85
92
|
console.log(` Format: ${d.$jdf}${bundle ? " (jdfx bundle)" : ""}`);
|
|
86
93
|
console.log(` Title: ${d.meta?.title}`);
|
|
87
94
|
console.log(` Pages: ${pageCount}`);
|
|
@@ -94,7 +101,7 @@ async function validate(file) {
|
|
|
94
101
|
}
|
|
95
102
|
return true;
|
|
96
103
|
}
|
|
97
|
-
console.error(`\u2717 Invalid: ${
|
|
104
|
+
console.error(`\u2717 Invalid: ${path7.basename(filePath)}`);
|
|
98
105
|
for (const err of validateFn.errors || []) {
|
|
99
106
|
const loc = err.instancePath || "(root)";
|
|
100
107
|
console.error(` ${loc} \u2014 ${err.message}`);
|
|
@@ -109,15 +116,15 @@ function hashBytes(bytes) {
|
|
|
109
116
|
return createHash("sha1").update(bytes).digest("hex").slice(0, 16);
|
|
110
117
|
}
|
|
111
118
|
function rewriteResourceRefs(doc, oldKey, newKey) {
|
|
112
|
-
function
|
|
119
|
+
function walk2(els) {
|
|
113
120
|
if (!els) return;
|
|
114
121
|
for (const el of els) {
|
|
115
122
|
if (el?.resource === oldKey) el.resource = newKey;
|
|
116
|
-
if (el?.elements)
|
|
117
|
-
if (el?.children)
|
|
123
|
+
if (el?.elements) walk2(el.elements);
|
|
124
|
+
if (el?.children) walk2(el.children);
|
|
118
125
|
}
|
|
119
126
|
}
|
|
120
|
-
for (const page of doc.pages || [])
|
|
127
|
+
for (const page of doc.pages || []) walk2(page.elements);
|
|
121
128
|
}
|
|
122
129
|
function extractAssets(doc) {
|
|
123
130
|
const assets = [];
|
|
@@ -149,10 +156,10 @@ function extractAssets(doc) {
|
|
|
149
156
|
assets.push({ id, bytes, mimeType, ext });
|
|
150
157
|
return id;
|
|
151
158
|
}
|
|
152
|
-
function
|
|
159
|
+
function walk2(els) {
|
|
153
160
|
if (!els) return;
|
|
154
161
|
for (const el of els) {
|
|
155
|
-
if (el?.type === "image" && typeof el.src === "string" && el.src.startsWith("data:")) {
|
|
162
|
+
if ((el?.type === "image" || el?.type === "video") && typeof el.src === "string" && el.src.startsWith("data:")) {
|
|
156
163
|
const m = el.src.match(/^data:([^;]+);base64,(.*)$/);
|
|
157
164
|
if (m) {
|
|
158
165
|
const mimeType = m[1];
|
|
@@ -162,19 +169,21 @@ function extractAssets(doc) {
|
|
|
162
169
|
el.resource = id;
|
|
163
170
|
}
|
|
164
171
|
}
|
|
165
|
-
if (el?.elements)
|
|
166
|
-
if (el?.children)
|
|
172
|
+
if (el?.elements) walk2(el.elements);
|
|
173
|
+
if (el?.children) walk2(el.children);
|
|
167
174
|
}
|
|
168
175
|
}
|
|
169
|
-
for (const page of cloned.pages || [])
|
|
170
|
-
|
|
171
|
-
|
|
176
|
+
for (const page of cloned.pages || []) walk2(page.elements);
|
|
177
|
+
for (const bucket of ["images", "videos"]) {
|
|
178
|
+
const store = cloned.resources?.[bucket];
|
|
179
|
+
if (!store) continue;
|
|
180
|
+
for (const [key, res] of Object.entries(store)) {
|
|
172
181
|
if (!res || typeof res !== "object" || !("data" in res) || !res.data) continue;
|
|
173
182
|
const data = String(res.data);
|
|
174
183
|
const m = data.match(/^data:([^;]+);base64,(.*)$/);
|
|
175
184
|
const b64 = m ? m[2] : data;
|
|
176
|
-
const mimeType = m ? m[1] : res.mimeType || "image/png";
|
|
177
|
-
const ext = mimeType.split("/")[1]?.replace("jpeg", "jpg") || "bin";
|
|
185
|
+
const mimeType = m ? m[1] : res.mimeType || (bucket === "videos" ? "video/mp4" : "image/png");
|
|
186
|
+
const ext = mimeType.split("/")[1]?.replace("jpeg", "jpg").replace("quicktime", "mov") || "bin";
|
|
178
187
|
const bytes = decodeBase64(b64);
|
|
179
188
|
const h = hashBytes(bytes);
|
|
180
189
|
let canonicalId = hashToId.get(h);
|
|
@@ -188,10 +197,10 @@ function extractAssets(doc) {
|
|
|
188
197
|
delete updated.data;
|
|
189
198
|
updated.src = "embedded";
|
|
190
199
|
if (canonicalId !== key) {
|
|
191
|
-
delete
|
|
200
|
+
delete store[key];
|
|
192
201
|
rewriteResourceRefs(cloned, key, canonicalId);
|
|
193
202
|
} else {
|
|
194
|
-
|
|
203
|
+
store[key] = updated;
|
|
195
204
|
}
|
|
196
205
|
}
|
|
197
206
|
}
|
|
@@ -221,21 +230,22 @@ async function packJdfx(doc) {
|
|
|
221
230
|
return { bytes, manifest };
|
|
222
231
|
}
|
|
223
232
|
function shouldUseJdfx(doc) {
|
|
224
|
-
const
|
|
225
|
-
|
|
226
|
-
|
|
233
|
+
for (const store of [doc.resources?.images ?? {}, doc.resources?.videos ?? {}]) {
|
|
234
|
+
for (const v of Object.values(store)) {
|
|
235
|
+
if (v && typeof v === "object" && "data" in v && v.data) return true;
|
|
236
|
+
}
|
|
227
237
|
}
|
|
228
|
-
function
|
|
238
|
+
function walk2(els) {
|
|
229
239
|
if (!els) return false;
|
|
230
240
|
for (const el of els) {
|
|
231
|
-
if (el?.type === "image" && typeof el.src === "string" && el.src.startsWith("data:")) return true;
|
|
232
|
-
if (el?.elements &&
|
|
233
|
-
if (el?.children &&
|
|
241
|
+
if ((el?.type === "image" || el?.type === "video") && typeof el.src === "string" && el.src.startsWith("data:")) return true;
|
|
242
|
+
if (el?.elements && walk2(el.elements)) return true;
|
|
243
|
+
if (el?.children && walk2(el.children)) return true;
|
|
234
244
|
}
|
|
235
245
|
return false;
|
|
236
246
|
}
|
|
237
247
|
for (const page of doc.pages || []) {
|
|
238
|
-
if (
|
|
248
|
+
if (walk2(page.elements)) return true;
|
|
239
249
|
}
|
|
240
250
|
return false;
|
|
241
251
|
}
|
|
@@ -294,13 +304,13 @@ function stripInline(text) {
|
|
|
294
304
|
return parseInline(text).map((r) => r.text).join("");
|
|
295
305
|
}
|
|
296
306
|
async function importMarkdown(inputPath, outputPath) {
|
|
297
|
-
const input =
|
|
307
|
+
const input = path7.resolve(inputPath);
|
|
298
308
|
console.log(`Importing: ${input}`);
|
|
299
|
-
const content =
|
|
300
|
-
const doc = convertMarkdownToJdf(content,
|
|
309
|
+
const content = fs7.readFileSync(input, "utf-8");
|
|
310
|
+
const doc = convertMarkdownToJdf(content, path7.basename(input, path7.extname(input)), path7.dirname(input));
|
|
301
311
|
let output;
|
|
302
312
|
if (outputPath) {
|
|
303
|
-
output =
|
|
313
|
+
output = path7.resolve(outputPath);
|
|
304
314
|
} else {
|
|
305
315
|
const stem = input.replace(/\.(md|markdown)$/i, "");
|
|
306
316
|
output = stem + (shouldUseJdfx(doc) ? ".jdfx" : ".jdf");
|
|
@@ -308,11 +318,11 @@ async function importMarkdown(inputPath, outputPath) {
|
|
|
308
318
|
console.log(`Output: ${output}`);
|
|
309
319
|
if (output.toLowerCase().endsWith(".jdfx")) {
|
|
310
320
|
const { bytes, manifest } = await packJdfx(doc);
|
|
311
|
-
|
|
321
|
+
fs7.writeFileSync(output, bytes);
|
|
312
322
|
console.log(`
|
|
313
323
|
Done! Created ${doc.pages.length} page(s), ${manifest.assets.length} asset(s) bundled`);
|
|
314
324
|
} else {
|
|
315
|
-
|
|
325
|
+
fs7.writeFileSync(output, JSON.stringify(doc, null, 2));
|
|
316
326
|
console.log(`
|
|
317
327
|
Done! Created ${doc.pages.length} page(s)`);
|
|
318
328
|
}
|
|
@@ -329,10 +339,10 @@ var MIME_BY_EXT2 = {
|
|
|
329
339
|
};
|
|
330
340
|
function resolveImageSrc(src, baseDir) {
|
|
331
341
|
if (/^(https?:|data:|file:)/i.test(src)) return src;
|
|
332
|
-
const abs =
|
|
342
|
+
const abs = path7.isAbsolute(src) ? src : path7.resolve(baseDir, src);
|
|
333
343
|
try {
|
|
334
|
-
const bytes =
|
|
335
|
-
const ext =
|
|
344
|
+
const bytes = fs7.readFileSync(abs);
|
|
345
|
+
const ext = path7.extname(abs).slice(1).toLowerCase();
|
|
336
346
|
const mime = MIME_BY_EXT2[ext] || "application/octet-stream";
|
|
337
347
|
return `data:${mime};base64,${bytes.toString("base64")}`;
|
|
338
348
|
} catch {
|
|
@@ -613,8 +623,232 @@ function convertMarkdownToJdf(md, title, baseDir = process.cwd()) {
|
|
|
613
623
|
};
|
|
614
624
|
}
|
|
615
625
|
|
|
616
|
-
// ../../packages/jdf-pdf-import/src/
|
|
626
|
+
// ../../packages/jdf-pdf-import/src/tables.ts
|
|
617
627
|
var PT_TO_MM = 0.352778;
|
|
628
|
+
function calibrateGlyphWidth(runs) {
|
|
629
|
+
const ks = [];
|
|
630
|
+
for (const r of runs) {
|
|
631
|
+
const t = r.text;
|
|
632
|
+
if (/\s$/.test(t) || t.trim().length < 3 || r.width <= 0) continue;
|
|
633
|
+
ks.push(r.width / (t.length * r.fontSize * PT_TO_MM));
|
|
634
|
+
}
|
|
635
|
+
if (ks.length < 3) return 0.55;
|
|
636
|
+
ks.sort((a, b) => a - b);
|
|
637
|
+
return Math.min(0.7, Math.max(0.4, ks[Math.floor(ks.length / 2)]));
|
|
638
|
+
}
|
|
639
|
+
function hasStretchedSpaces(runs, k) {
|
|
640
|
+
let n = 0, wide = 0;
|
|
641
|
+
for (const r of runs) {
|
|
642
|
+
if (!/\s$/.test(r.text) || r.text.trim().length === 0) continue;
|
|
643
|
+
const em = r.fontSize * PT_TO_MM;
|
|
644
|
+
const est = r.text.trim().length * em * k + em * 0.25;
|
|
645
|
+
n++;
|
|
646
|
+
if (r.width > est * 1.4) wide++;
|
|
647
|
+
}
|
|
648
|
+
return n >= 4 && wide / n >= 0.3;
|
|
649
|
+
}
|
|
650
|
+
function textExtent(r, k, stretched) {
|
|
651
|
+
const em = r.fontSize * PT_TO_MM;
|
|
652
|
+
if (!/\s$/.test(r.text)) return Math.max(em * 0.5, r.width);
|
|
653
|
+
const chars = Math.max(1, r.text.trim().length);
|
|
654
|
+
const est = chars * em * k + em * 0.25;
|
|
655
|
+
if (stretched) return Math.max(em * 0.5, Math.min(r.width, est));
|
|
656
|
+
return Math.max(em * 0.5, r.width > est * 1.4 ? est : r.width);
|
|
657
|
+
}
|
|
658
|
+
function groupRows(runs, skip) {
|
|
659
|
+
const k = calibrateGlyphWidth(runs);
|
|
660
|
+
const stretched = hasStretchedSpaces(runs, k);
|
|
661
|
+
const idx = runs.map((_, i) => i).filter((i) => !skip(runs[i]) && runs[i].text.trim().length > 0);
|
|
662
|
+
idx.sort((a, b) => runs[a].y - runs[b].y || runs[a].x - runs[b].x);
|
|
663
|
+
const rows = [];
|
|
664
|
+
for (const i of idx) {
|
|
665
|
+
const r = runs[i];
|
|
666
|
+
const tol = Math.max(0.8, r.fontSize * PT_TO_MM * 0.35);
|
|
667
|
+
const last = rows[rows.length - 1];
|
|
668
|
+
const cell = { run: r, idx: i, x0: r.x, x1: r.x + textExtent(r, k, stretched) };
|
|
669
|
+
if (last && Math.abs(last.y - r.y) <= tol) {
|
|
670
|
+
last.cells.push(cell);
|
|
671
|
+
last.h = Math.max(last.h, r.height);
|
|
672
|
+
} else {
|
|
673
|
+
rows.push({ y: r.y, h: r.height, cells: [cell] });
|
|
674
|
+
}
|
|
675
|
+
}
|
|
676
|
+
for (const row of rows) {
|
|
677
|
+
row.cells.sort((a, b) => a.x0 - b.x0);
|
|
678
|
+
const merged = [];
|
|
679
|
+
for (const c of row.cells) {
|
|
680
|
+
const last = merged[merged.length - 1];
|
|
681
|
+
const em = c.run.fontSize * PT_TO_MM;
|
|
682
|
+
if (last && c.x0 - last.x1 <= em * 1) {
|
|
683
|
+
last.x1 = Math.max(last.x1, c.x1);
|
|
684
|
+
last.run = { ...last.run, text: `${last.run.text.replace(/\s+$/, "")} ${c.run.text.replace(/^\s+/, "")}`, width: last.x1 - last.x0 };
|
|
685
|
+
last.extra = [...last.extra ?? [], c.idx];
|
|
686
|
+
} else merged.push({ ...c });
|
|
687
|
+
}
|
|
688
|
+
row.cells = merged;
|
|
689
|
+
}
|
|
690
|
+
return rows;
|
|
691
|
+
}
|
|
692
|
+
function columnBands(rows) {
|
|
693
|
+
const cells = rows.flatMap((r) => r.cells);
|
|
694
|
+
const sorted = cells.slice().sort((a, b) => a.x0 - b.x0);
|
|
695
|
+
const bands = [];
|
|
696
|
+
for (const c of sorted) {
|
|
697
|
+
const last = bands[bands.length - 1];
|
|
698
|
+
if (last && c.x0 <= last.x1 - 0.2) {
|
|
699
|
+
last.x1 = Math.max(last.x1, c.x1);
|
|
700
|
+
last.members.push(c);
|
|
701
|
+
} else bands.push({ x0: c.x0, x1: c.x1, members: [c] });
|
|
702
|
+
}
|
|
703
|
+
for (const b of bands) {
|
|
704
|
+
const rowsSeen = /* @__PURE__ */ new Set();
|
|
705
|
+
for (const m of b.members) {
|
|
706
|
+
const row = rows.find((r) => r.cells.includes(m));
|
|
707
|
+
if (rowsSeen.has(row)) return null;
|
|
708
|
+
rowsSeen.add(row);
|
|
709
|
+
}
|
|
710
|
+
}
|
|
711
|
+
return bands.map(({ x0, x1 }) => ({ x0, x1 }));
|
|
712
|
+
}
|
|
713
|
+
var numeric = (s) => /^[\s$€£¥+\-−–]*[\d.,]+\s*(%|ms|s|k|m|b|M|K|B|x|×)?\s*(\/\w+)?$/i.test(s.trim()) || /^[+\-−]?\d/.test(s.trim()) && /\d$/.test(s.trim().replace(/[%)]$/, ""));
|
|
714
|
+
function detectTables(runs, shapes, pageWidthMm) {
|
|
715
|
+
const out = [];
|
|
716
|
+
const used = /* @__PURE__ */ new Set();
|
|
717
|
+
const rows = groupRows(runs, () => false);
|
|
718
|
+
const pageW = pageWidthMm;
|
|
719
|
+
let i = 0;
|
|
720
|
+
while (i < rows.length) {
|
|
721
|
+
if (rows[i].cells.length < 2) {
|
|
722
|
+
i++;
|
|
723
|
+
continue;
|
|
724
|
+
}
|
|
725
|
+
let j = i;
|
|
726
|
+
let cur = columnBands([rows[i]]);
|
|
727
|
+
let best = null;
|
|
728
|
+
while (j + 1 < rows.length && cur) {
|
|
729
|
+
const next = rows[j + 1];
|
|
730
|
+
const gap = next.y - (rows[j].y + rows[j].h);
|
|
731
|
+
const rowH = Math.max(rows[j].h, next.h);
|
|
732
|
+
if (gap > rowH * 2.2) break;
|
|
733
|
+
if (next.cells.length === 1) {
|
|
734
|
+
const c = next.cells[0];
|
|
735
|
+
const inBand = cur.findIndex((b) => c.x0 < b.x1 - 0.2 && c.x1 > b.x0 + 0.2);
|
|
736
|
+
const spansSeveral = cur.filter((b) => c.x0 < b.x1 - 0.2 && c.x1 > b.x0 + 0.2).length > 1;
|
|
737
|
+
if (inBand <= 0 || spansSeveral) break;
|
|
738
|
+
j++;
|
|
739
|
+
continue;
|
|
740
|
+
}
|
|
741
|
+
const nb = columnBands(rows.slice(i, j + 2));
|
|
742
|
+
if (!nb || nb.length < 2) break;
|
|
743
|
+
if (nb.length > cur.length && j - i >= 2) break;
|
|
744
|
+
cur = nb;
|
|
745
|
+
j++;
|
|
746
|
+
if (cur.length >= 2) best = { j, bands: cur };
|
|
747
|
+
}
|
|
748
|
+
const multiRows = best ? rows.slice(i, best.j + 1).filter((r) => r.cells.length >= 2).length : 0;
|
|
749
|
+
const blockRows = best ? rows.slice(i, best.j + 1) : [];
|
|
750
|
+
const bbox = blockRows.length ? {
|
|
751
|
+
x0: Math.min(...best.bands.map((b) => b.x0)) - 5,
|
|
752
|
+
x1: Math.max(...best.bands.map((b) => b.x1)) + 5,
|
|
753
|
+
// Cell padding puts backgrounds/borders well above the first baseline and below the last.
|
|
754
|
+
y0: blockRows[0].y - blockRows[0].h * 2.5,
|
|
755
|
+
y1: blockRows[blockRows.length - 1].y + blockRows[blockRows.length - 1].h * 3
|
|
756
|
+
} : null;
|
|
757
|
+
const gridShapes = bbox ? shapes.map((s, k) => ({ s, k })).filter(({ s }) => s.x >= bbox.x0 - 1 && s.x + s.width <= bbox.x1 + 1 && s.y >= bbox.y0 - 1 && s.y + s.height <= bbox.y1 + 1 && (s.kind === "line" || s.kind === "rect")) : [];
|
|
758
|
+
const hasLattice = gridShapes.length >= 3;
|
|
759
|
+
if (!best || multiRows < (hasLattice ? 2 : 3) || best.bands.length < 2) {
|
|
760
|
+
i++;
|
|
761
|
+
continue;
|
|
762
|
+
}
|
|
763
|
+
const bands = best.bands;
|
|
764
|
+
const cellText = (row, b) => row.cells.filter((c) => c.x0 < bands[b].x1 - 0.2 && c.x1 > bands[b].x0 + 0.2).map((c) => c.run.text.trim()).join(" ").trim();
|
|
765
|
+
const grid = [];
|
|
766
|
+
const lineIdx = [];
|
|
767
|
+
for (const row of blockRows) {
|
|
768
|
+
if (row.cells.length === 1 && grid.length) {
|
|
769
|
+
const c = row.cells[0];
|
|
770
|
+
const b = bands.findIndex((bb) => c.x0 < bb.x1 - 0.2 && c.x1 > bb.x0 + 0.2);
|
|
771
|
+
if (b > 0) {
|
|
772
|
+
grid[grid.length - 1][b] = (grid[grid.length - 1][b] + " " + c.run.text.trim()).trim();
|
|
773
|
+
lineIdx.push(c.idx, ...c.extra ?? []);
|
|
774
|
+
continue;
|
|
775
|
+
}
|
|
776
|
+
}
|
|
777
|
+
grid.push(bands.map((_, b) => cellText(row, b)));
|
|
778
|
+
for (const c of row.cells) {
|
|
779
|
+
lineIdx.push(c.idx);
|
|
780
|
+
for (const k of c.extra ?? []) lineIdx.push(k);
|
|
781
|
+
}
|
|
782
|
+
}
|
|
783
|
+
if (lineIdx.some((k) => used.has(k))) {
|
|
784
|
+
i = best.j + 1;
|
|
785
|
+
continue;
|
|
786
|
+
}
|
|
787
|
+
const first = blockRows[0];
|
|
788
|
+
const tableW = bbox.x1 - bbox.x0;
|
|
789
|
+
const rowFill = (row) => {
|
|
790
|
+
const cy = row.y + row.h * 0.5;
|
|
791
|
+
const rects = gridShapes.filter(({ s }) => s.kind === "rect" && s.fill && s.fill.toLowerCase() !== "#ffffff" && s.height >= row.h * 0.6 && s.height < row.h * 4.5 && s.y <= cy && s.y + s.height >= cy);
|
|
792
|
+
const covered = rects.reduce((a, { s }) => a + s.width, 0);
|
|
793
|
+
if (!rects.length || covered < tableW * 0.5) return null;
|
|
794
|
+
const counts = /* @__PURE__ */ new Map();
|
|
795
|
+
for (const { s } of rects) counts.set(s.fill, (counts.get(s.fill) ?? 0) + s.width);
|
|
796
|
+
const fill = [...counts.entries()].sort((a, b) => b[1] - a[1])[0][0];
|
|
797
|
+
return { fill, rects };
|
|
798
|
+
};
|
|
799
|
+
const headerBg = rowFill(first);
|
|
800
|
+
const firstBold = first.cells.every((c) => c.run.bold);
|
|
801
|
+
const bodyFills = blockRows.slice(1).map(rowFill);
|
|
802
|
+
const headerDistinct = !!headerBg && !bodyFills.every((f) => f?.fill === headerBg.fill);
|
|
803
|
+
const isHeader = headerDistinct || firstBold && !blockRows.slice(1).every((r) => r.cells.every((c) => c.run.bold));
|
|
804
|
+
const columns = bands.map((b, k) => {
|
|
805
|
+
const vals = grid.slice(isHeader ? 1 : 0).map((r) => r[k]).filter(Boolean);
|
|
806
|
+
const numericShare = vals.length ? vals.filter(numeric).length / vals.length : 0;
|
|
807
|
+
const col = { width: Math.round((b.x1 - b.x0) * 10) / 10 };
|
|
808
|
+
if (numericShare >= 0.7) col.align = "right";
|
|
809
|
+
return col;
|
|
810
|
+
});
|
|
811
|
+
const x0 = Math.max(0, bands[0].x0 - 2.5);
|
|
812
|
+
const x1 = Math.min(pageW, bands[bands.length - 1].x1 + 2.5);
|
|
813
|
+
for (let k = 0; k < bands.length; k++) {
|
|
814
|
+
const left = k === 0 ? x0 : (bands[k - 1].x1 + bands[k].x0) / 2;
|
|
815
|
+
const right = k === bands.length - 1 ? x1 : (bands[k].x1 + bands[k + 1].x0) / 2;
|
|
816
|
+
columns[k].width = Math.round((right - left) * 10) / 10;
|
|
817
|
+
}
|
|
818
|
+
const rowFills = (isHeader ? bodyFills : [headerBg, ...bodyFills]).map((f) => f?.fill ?? null);
|
|
819
|
+
const odd = rowFills.filter((_, k) => k % 2 === 1), even = rowFills.filter((_, k) => k % 2 === 0);
|
|
820
|
+
const altColor = odd.length && odd[0] && odd.every((f) => f === odd[0]) && even.every((f) => f !== odd[0]) ? odd[0] : void 0;
|
|
821
|
+
const borderShape = gridShapes.find(({ s }) => s.kind === "line" || s.kind === "rect" && (s.height < 0.6 || s.width < 0.6) && (s.fill || s.stroke));
|
|
822
|
+
const borderColor = borderShape ? borderShape.s.stroke || borderShape.s.fill : void 0;
|
|
823
|
+
const fontSize = Math.round(first.cells[0].run.fontSize * 10) / 10;
|
|
824
|
+
const y0 = headerBg ? Math.min(...headerBg.rects.map(({ s }) => s.y)) : first.y - first.h * 0.5;
|
|
825
|
+
const element = {
|
|
826
|
+
type: "table",
|
|
827
|
+
position: { x: Math.round(x0 * 100) / 100, y: Math.round(Math.max(0, y0) * 100) / 100 },
|
|
828
|
+
width: Math.round((x1 - x0) * 100) / 100,
|
|
829
|
+
columns,
|
|
830
|
+
rows: isHeader ? grid.slice(1) : grid,
|
|
831
|
+
style: { fontSize }
|
|
832
|
+
};
|
|
833
|
+
if (isHeader) {
|
|
834
|
+
element.headers = grid[0];
|
|
835
|
+
const hs = { fontWeight: "bold" };
|
|
836
|
+
if (headerBg) hs.backgroundColor = headerBg.fill;
|
|
837
|
+
const hc = first.cells[0].run.color;
|
|
838
|
+
if (hc && hc !== "#000000") hs.color = hc;
|
|
839
|
+
element.headerStyle = hs;
|
|
840
|
+
}
|
|
841
|
+
if (altColor) element.alternatingRowColor = altColor;
|
|
842
|
+
element.borders = borderColor ? { outer: true, inner: true, color: borderColor, width: 1 } : false;
|
|
843
|
+
for (const k of lineIdx) used.add(k);
|
|
844
|
+
out.push({ element, lineIdx, shapeIdx: gridShapes.map(({ k }) => k) });
|
|
845
|
+
i = best.j + 1;
|
|
846
|
+
}
|
|
847
|
+
return out;
|
|
848
|
+
}
|
|
849
|
+
|
|
850
|
+
// ../../packages/jdf-pdf-import/src/core.ts
|
|
851
|
+
var PT_TO_MM2 = 0.352778;
|
|
618
852
|
function classifyFont(name) {
|
|
619
853
|
const n = (name || "").toLowerCase();
|
|
620
854
|
const bold = /bold|black|heavy|semibold|demibold|extrabold/.test(n);
|
|
@@ -724,19 +958,36 @@ async function walkOps(page, OPS, viewport) {
|
|
|
724
958
|
const minY = Math.min(...ys), maxY = Math.max(...ys);
|
|
725
959
|
imagePositions.push({
|
|
726
960
|
name,
|
|
727
|
-
x: minX *
|
|
728
|
-
y: minY *
|
|
729
|
-
w: (maxX - minX) *
|
|
730
|
-
h: (maxY - minY) *
|
|
961
|
+
x: minX * PT_TO_MM2,
|
|
962
|
+
y: minY * PT_TO_MM2,
|
|
963
|
+
w: (maxX - minX) * PT_TO_MM2,
|
|
964
|
+
h: (maxY - minY) * PT_TO_MM2,
|
|
731
965
|
inline,
|
|
732
966
|
maskFill
|
|
733
967
|
});
|
|
734
968
|
}
|
|
735
969
|
let pathSegments = [];
|
|
736
970
|
let pathRect = null;
|
|
971
|
+
let pathRects = [];
|
|
737
972
|
let pathStart = null;
|
|
738
973
|
let pathLast = null;
|
|
739
974
|
function flushPath(isFill, isStroke) {
|
|
975
|
+
for (const r of pathRects) {
|
|
976
|
+
const tl = toViewport(r.x, r.y + r.h);
|
|
977
|
+
const br = toViewport(r.x + r.w, r.y);
|
|
978
|
+
shapes.push({
|
|
979
|
+
kind: "rect",
|
|
980
|
+
x: Math.min(tl.x, br.x) * PT_TO_MM2,
|
|
981
|
+
y: Math.min(tl.y, br.y) * PT_TO_MM2,
|
|
982
|
+
width: Math.abs(br.x - tl.x) * PT_TO_MM2,
|
|
983
|
+
height: Math.abs(br.y - tl.y) * PT_TO_MM2,
|
|
984
|
+
fill: isFill ? gs.fill : void 0,
|
|
985
|
+
stroke: isStroke ? gs.stroke : void 0,
|
|
986
|
+
strokeWidth: isStroke ? gs.lineWidth * PT_TO_MM2 : void 0,
|
|
987
|
+
opacity: isFill ? gs.fillAlpha : gs.strokeAlpha
|
|
988
|
+
});
|
|
989
|
+
}
|
|
990
|
+
pathRects = [];
|
|
740
991
|
if (pathRect) {
|
|
741
992
|
const tl = toViewport(pathRect.x, pathRect.y + pathRect.h);
|
|
742
993
|
const br = toViewport(pathRect.x + pathRect.w, pathRect.y);
|
|
@@ -746,13 +997,13 @@ async function walkOps(page, OPS, viewport) {
|
|
|
746
997
|
const h = Math.abs(br.y - tl.y);
|
|
747
998
|
shapes.push({
|
|
748
999
|
kind: "rect",
|
|
749
|
-
x: x *
|
|
750
|
-
y: y *
|
|
751
|
-
width: w *
|
|
752
|
-
height: h *
|
|
1000
|
+
x: x * PT_TO_MM2,
|
|
1001
|
+
y: y * PT_TO_MM2,
|
|
1002
|
+
width: w * PT_TO_MM2,
|
|
1003
|
+
height: h * PT_TO_MM2,
|
|
753
1004
|
fill: isFill ? gs.fill : void 0,
|
|
754
1005
|
stroke: isStroke ? gs.stroke : void 0,
|
|
755
|
-
strokeWidth: isStroke ? gs.lineWidth *
|
|
1006
|
+
strokeWidth: isStroke ? gs.lineWidth * PT_TO_MM2 : void 0,
|
|
756
1007
|
opacity: isFill ? gs.fillAlpha : gs.strokeAlpha
|
|
757
1008
|
});
|
|
758
1009
|
} else if (pathSegments.length === 2 && pathSegments[0].type === "M" && pathSegments[1].type === "L") {
|
|
@@ -764,35 +1015,35 @@ async function walkOps(page, OPS, viewport) {
|
|
|
764
1015
|
const minY = Math.min(va.y, vb.y);
|
|
765
1016
|
const maxX = Math.max(va.x, vb.x);
|
|
766
1017
|
const maxY = Math.max(va.y, vb.y);
|
|
767
|
-
const x1Local = (va.x - minX) *
|
|
768
|
-
const y1Local = (va.y - minY) *
|
|
769
|
-
const x2Local = (vb.x - minX) *
|
|
770
|
-
const y2Local = (vb.y - minY) *
|
|
771
|
-
const wLocal = Math.max(0.05, (maxX - minX) *
|
|
772
|
-
const hLocal = Math.max(0.05, (maxY - minY) *
|
|
1018
|
+
const x1Local = (va.x - minX) * PT_TO_MM2;
|
|
1019
|
+
const y1Local = (va.y - minY) * PT_TO_MM2;
|
|
1020
|
+
const x2Local = (vb.x - minX) * PT_TO_MM2;
|
|
1021
|
+
const y2Local = (vb.y - minY) * PT_TO_MM2;
|
|
1022
|
+
const wLocal = Math.max(0.05, (maxX - minX) * PT_TO_MM2);
|
|
1023
|
+
const hLocal = Math.max(0.05, (maxY - minY) * PT_TO_MM2);
|
|
773
1024
|
const dx = Math.abs(va.x - vb.x);
|
|
774
1025
|
const dy = Math.abs(va.y - vb.y);
|
|
775
1026
|
const axisAligned = dx < 0.5 || dy < 0.5;
|
|
776
1027
|
if (axisAligned) {
|
|
777
1028
|
shapes.push({
|
|
778
1029
|
kind: "line",
|
|
779
|
-
x: minX *
|
|
780
|
-
y: minY *
|
|
1030
|
+
x: minX * PT_TO_MM2,
|
|
1031
|
+
y: minY * PT_TO_MM2,
|
|
781
1032
|
width: wLocal,
|
|
782
1033
|
height: hLocal,
|
|
783
1034
|
stroke: isStroke ? gs.stroke : void 0,
|
|
784
|
-
strokeWidth: isStroke ? gs.lineWidth *
|
|
1035
|
+
strokeWidth: isStroke ? gs.lineWidth * PT_TO_MM2 : void 0,
|
|
785
1036
|
opacity: gs.strokeAlpha
|
|
786
1037
|
});
|
|
787
1038
|
} else {
|
|
788
1039
|
shapes.push({
|
|
789
1040
|
kind: "path",
|
|
790
|
-
x: minX *
|
|
791
|
-
y: minY *
|
|
1041
|
+
x: minX * PT_TO_MM2,
|
|
1042
|
+
y: minY * PT_TO_MM2,
|
|
792
1043
|
width: wLocal,
|
|
793
1044
|
height: hLocal,
|
|
794
1045
|
stroke: isStroke ? gs.stroke : void 0,
|
|
795
|
-
strokeWidth: isStroke ? gs.lineWidth *
|
|
1046
|
+
strokeWidth: isStroke ? gs.lineWidth * PT_TO_MM2 : void 0,
|
|
796
1047
|
opacity: gs.strokeAlpha,
|
|
797
1048
|
path: `M ${x1Local.toFixed(2)} ${y1Local.toFixed(2)} L ${x2Local.toFixed(2)} ${y2Local.toFixed(2)}`
|
|
798
1049
|
});
|
|
@@ -825,20 +1076,20 @@ async function walkOps(page, OPS, viewport) {
|
|
|
825
1076
|
if (seg.type === "Z") return "Z";
|
|
826
1077
|
const p = [];
|
|
827
1078
|
for (let i = 0; i < seg.pts.length; i += 2) {
|
|
828
|
-
p.push(((seg.pts[i] - minX) *
|
|
829
|
-
p.push(((seg.pts[i + 1] - minY) *
|
|
1079
|
+
p.push(((seg.pts[i] - minX) * PT_TO_MM2).toFixed(2));
|
|
1080
|
+
p.push(((seg.pts[i + 1] - minY) * PT_TO_MM2).toFixed(2));
|
|
830
1081
|
}
|
|
831
1082
|
return `${seg.type} ${p.join(" ")}`;
|
|
832
1083
|
}).join(" ");
|
|
833
1084
|
shapes.push({
|
|
834
1085
|
kind: "path",
|
|
835
|
-
x: minX *
|
|
836
|
-
y: minY *
|
|
837
|
-
width: bw *
|
|
838
|
-
height: bh *
|
|
1086
|
+
x: minX * PT_TO_MM2,
|
|
1087
|
+
y: minY * PT_TO_MM2,
|
|
1088
|
+
width: bw * PT_TO_MM2,
|
|
1089
|
+
height: bh * PT_TO_MM2,
|
|
839
1090
|
fill: isFill ? gs.fill : void 0,
|
|
840
1091
|
stroke: isStroke ? gs.stroke : void 0,
|
|
841
|
-
strokeWidth: isStroke ? gs.lineWidth *
|
|
1092
|
+
strokeWidth: isStroke ? gs.lineWidth * PT_TO_MM2 : void 0,
|
|
842
1093
|
opacity: isFill ? gs.fillAlpha : gs.strokeAlpha,
|
|
843
1094
|
path: d
|
|
844
1095
|
});
|
|
@@ -1000,6 +1251,12 @@ async function walkOps(page, OPS, viewport) {
|
|
|
1000
1251
|
} else if (op === OPS.closePath) {
|
|
1001
1252
|
pathSegments.push({ type: "Z", pts: [] });
|
|
1002
1253
|
if (pathStart) pathLast = { ...pathStart };
|
|
1254
|
+
} else if (op === OPS.rectangle) {
|
|
1255
|
+
const x = pathArgs[ai], y = pathArgs[ai + 1], w = pathArgs[ai + 2], h = pathArgs[ai + 3];
|
|
1256
|
+
ai += 4;
|
|
1257
|
+
const p1 = tx(gs.ctm, x, y);
|
|
1258
|
+
const p3 = tx(gs.ctm, x + w, y + h);
|
|
1259
|
+
pathRects.push({ x: Math.min(p1.x, p3.x), y: Math.min(p1.y, p3.y), w: Math.abs(p3.x - p1.x), h: Math.abs(p3.y - p1.y) });
|
|
1003
1260
|
}
|
|
1004
1261
|
}
|
|
1005
1262
|
} else if (fn === OPS.fill || fn === OPS.stroke || fn === OPS.fillStroke || fn === OPS.eoFill || fn === OPS.eoFillStroke || fn === OPS.closeFillStroke || fn === OPS.closeStroke || fn === OPS.closeEOFillStroke) {
|
|
@@ -1008,6 +1265,7 @@ async function walkOps(page, OPS, viewport) {
|
|
|
1008
1265
|
flushPath(isFill, isStroke);
|
|
1009
1266
|
} else if (fn === OPS.endPath || fn === OPS.clip || fn === OPS.eoClip) {
|
|
1010
1267
|
pathSegments = [];
|
|
1268
|
+
pathRects = [];
|
|
1011
1269
|
pathRect = null;
|
|
1012
1270
|
pathStart = null;
|
|
1013
1271
|
pathLast = null;
|
|
@@ -1172,10 +1430,10 @@ async function extractLinks(doc, page, viewport) {
|
|
|
1172
1430
|
const xMax = Math.max(c1.x, c2.x);
|
|
1173
1431
|
const yMax = Math.max(c1.y, c2.y);
|
|
1174
1432
|
const rectMm = {
|
|
1175
|
-
x: xMin *
|
|
1176
|
-
y: yMin *
|
|
1177
|
-
w: (xMax - xMin) *
|
|
1178
|
-
h: (yMax - yMin) *
|
|
1433
|
+
x: xMin * PT_TO_MM2,
|
|
1434
|
+
y: yMin * PT_TO_MM2,
|
|
1435
|
+
w: (xMax - xMin) * PT_TO_MM2,
|
|
1436
|
+
h: (yMax - yMin) * PT_TO_MM2
|
|
1179
1437
|
};
|
|
1180
1438
|
const url = a.url || a.unsafeUrl;
|
|
1181
1439
|
const destPage = url ? void 0 : await resolveDestPage(doc, a.dest);
|
|
@@ -1213,10 +1471,10 @@ async function extractFormWidgets(page, viewport) {
|
|
|
1213
1471
|
})).filter((o) => o.value !== "") : [];
|
|
1214
1472
|
out.push({
|
|
1215
1473
|
rectMm: {
|
|
1216
|
-
x: xMin *
|
|
1217
|
-
y: yMin *
|
|
1218
|
-
w: (xMax - xMin) *
|
|
1219
|
-
h: (yMax - yMin) *
|
|
1474
|
+
x: xMin * PT_TO_MM2,
|
|
1475
|
+
y: yMin * PT_TO_MM2,
|
|
1476
|
+
w: (xMax - xMin) * PT_TO_MM2,
|
|
1477
|
+
h: (yMax - yMin) * PT_TO_MM2
|
|
1220
1478
|
},
|
|
1221
1479
|
fieldType: a.fieldType || "",
|
|
1222
1480
|
fieldName: a.fieldName || `field-${out.length + 1}`,
|
|
@@ -1236,16 +1494,16 @@ async function extractFormWidgets(page, viewport) {
|
|
|
1236
1494
|
async function flattenOutline(doc, outline) {
|
|
1237
1495
|
if (!outline) return [];
|
|
1238
1496
|
const out = [];
|
|
1239
|
-
async function
|
|
1497
|
+
async function walk2(items, depth) {
|
|
1240
1498
|
for (const item of items) {
|
|
1241
1499
|
const idx = await resolveDestPage(doc, item.dest);
|
|
1242
1500
|
if (idx != null && typeof item.title === "string" && item.title.trim()) {
|
|
1243
1501
|
out.push({ title: item.title.trim(), pageIndex: idx, depth });
|
|
1244
1502
|
}
|
|
1245
|
-
if (item.items?.length) await
|
|
1503
|
+
if (item.items?.length) await walk2(item.items, depth + 1);
|
|
1246
1504
|
}
|
|
1247
1505
|
}
|
|
1248
|
-
await
|
|
1506
|
+
await walk2(outline, 1);
|
|
1249
1507
|
return out;
|
|
1250
1508
|
}
|
|
1251
1509
|
var normTitle = (s) => s.toLowerCase().replace(/[\s\u00a0]+/g, " ").replace(/[^\p{L}\p{N} ]/gu, "").trim();
|
|
@@ -1441,6 +1699,10 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
|
1441
1699
|
if (arr) arr.push(op);
|
|
1442
1700
|
else opBins.set(key, [op]);
|
|
1443
1701
|
}
|
|
1702
|
+
const sizePenalty = (op, fontSize) => {
|
|
1703
|
+
if (!op.fontSize || !fontSize) return 0;
|
|
1704
|
+
return Math.abs(Math.log(op.fontSize / fontSize)) * 6;
|
|
1705
|
+
};
|
|
1444
1706
|
const findOp = (x, y, fontSize) => {
|
|
1445
1707
|
let best = null;
|
|
1446
1708
|
let bestD = Infinity;
|
|
@@ -1450,7 +1712,7 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
|
1450
1712
|
const arr = opBins.get(`${bx + dx},${by + dy}`);
|
|
1451
1713
|
if (!arr) continue;
|
|
1452
1714
|
for (const op of arr) {
|
|
1453
|
-
const d = Math.hypot(op.x - x, op.y - y);
|
|
1715
|
+
const d = Math.hypot(op.x - x, op.y - y) + sizePenalty(op, fontSize);
|
|
1454
1716
|
if (d < bestD) {
|
|
1455
1717
|
bestD = d;
|
|
1456
1718
|
best = op;
|
|
@@ -1460,8 +1722,9 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
|
1460
1722
|
}
|
|
1461
1723
|
if (best) return best;
|
|
1462
1724
|
const tol = Math.max(2, fontSize * 0.6);
|
|
1725
|
+
const sizeOk = (op) => !op.fontSize || !fontSize || op.fontSize / fontSize > 0.6 && op.fontSize / fontSize < 1.7;
|
|
1463
1726
|
for (const op of ops.textOps) {
|
|
1464
|
-
if (Math.abs(op.y - y) > tol) continue;
|
|
1727
|
+
if (!sizeOk(op) || Math.abs(op.y - y) > tol) continue;
|
|
1465
1728
|
const d = Math.abs(op.x - x) + Math.abs(op.y - y) * 4;
|
|
1466
1729
|
if (d < bestD) {
|
|
1467
1730
|
bestD = d;
|
|
@@ -1470,6 +1733,7 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
|
1470
1733
|
}
|
|
1471
1734
|
if (best) return best;
|
|
1472
1735
|
for (const op of ops.textOps) {
|
|
1736
|
+
if (!sizeOk(op)) continue;
|
|
1473
1737
|
const d = Math.hypot(op.x - x, op.y - y);
|
|
1474
1738
|
if (d < bestD) {
|
|
1475
1739
|
bestD = d;
|
|
@@ -1488,6 +1752,7 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
|
1488
1752
|
const conv = viewport.convertToViewportPoint(baseX, baseY);
|
|
1489
1753
|
const vx = safeNum(conv?.[0], 0);
|
|
1490
1754
|
const vy = safeNum(conv?.[1], 0);
|
|
1755
|
+
if (fontSize < 1.5) return;
|
|
1491
1756
|
const op = findOp(vx, vy, fontSize);
|
|
1492
1757
|
const mode = op?.mode ?? 0;
|
|
1493
1758
|
if (mode === 7) return;
|
|
@@ -1498,12 +1763,12 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
|
1498
1763
|
const w = safeNum(it.width, 0);
|
|
1499
1764
|
runs.push({
|
|
1500
1765
|
text: it.str,
|
|
1501
|
-
x: safeNum(vx *
|
|
1502
|
-
y: safeNum(yTop *
|
|
1766
|
+
x: safeNum(vx * PT_TO_MM2, 0),
|
|
1767
|
+
y: safeNum(yTop * PT_TO_MM2, 0),
|
|
1503
1768
|
fontSize: safeNum(fontSize, 10),
|
|
1504
1769
|
fontName: it.fontName,
|
|
1505
|
-
width: safeNum(w *
|
|
1506
|
-
height: safeNum((it.height || fontSize) *
|
|
1770
|
+
width: safeNum(w * PT_TO_MM2, 0),
|
|
1771
|
+
height: safeNum((it.height || fontSize) * PT_TO_MM2, fontSize * PT_TO_MM2),
|
|
1507
1772
|
color: op?.fill || "#000000",
|
|
1508
1773
|
opacity: invisible ? 0 : safeNum(op?.alpha, 1)
|
|
1509
1774
|
});
|
|
@@ -1511,6 +1776,12 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
|
1511
1776
|
runs.sort((a, b) => a.y - b.y || a.x - b.x);
|
|
1512
1777
|
const lines = [];
|
|
1513
1778
|
const Y_TOL = 0.6;
|
|
1779
|
+
const kGlyph = calibrateGlyphWidth(runs);
|
|
1780
|
+
const stretchedSpaces = hasStretchedSpaces(runs, kGlyph);
|
|
1781
|
+
const fontKey = (name) => {
|
|
1782
|
+
const c = fontMap.get(name) || classifyFont(name || "");
|
|
1783
|
+
return `${c.family}|${c.weight || ""}|${c.style || ""}`;
|
|
1784
|
+
};
|
|
1514
1785
|
for (const r of runs) {
|
|
1515
1786
|
if (!r.text.length) continue;
|
|
1516
1787
|
const last = lines[lines.length - 1];
|
|
@@ -1519,24 +1790,48 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
|
1519
1790
|
continue;
|
|
1520
1791
|
}
|
|
1521
1792
|
const sameLine = Math.abs(last.y - r.y) <= Y_TOL;
|
|
1522
|
-
const sameStyle = Math.abs(last.fontSize - r.fontSize) < 0.4 && last.fontName === r.fontName && last.color === r.color && Math.abs(last.opacity - r.opacity) < 0.05;
|
|
1523
|
-
const
|
|
1524
|
-
const
|
|
1525
|
-
|
|
1793
|
+
const sameStyle = Math.abs(last.fontSize - r.fontSize) < 0.4 && (last.fontName === r.fontName || fontKey(last.fontName) === fontKey(r.fontName)) && last.color === r.color && Math.abs(last.opacity - r.opacity) < 0.05;
|
|
1794
|
+
const emMm = r.fontSize * PT_TO_MM2;
|
|
1795
|
+
const extent = (t) => {
|
|
1796
|
+
if (!/\s$/.test(t.text)) return t.width;
|
|
1797
|
+
const em = t.fontSize * PT_TO_MM2;
|
|
1798
|
+
const est = Math.max(1, t.text.trim().length) * em * kGlyph + em * 0.25;
|
|
1799
|
+
if (stretchedSpaces) return Math.min(t.width, est);
|
|
1800
|
+
return t.width > est * 1.4 ? est : t.width;
|
|
1801
|
+
};
|
|
1802
|
+
const gapMm = r.x - (last.x + extent(last));
|
|
1803
|
+
const mergeOk = sameLine && sameStyle && gapMm >= -emMm * 0.5 && gapMm <= emMm * 0.45;
|
|
1526
1804
|
if (mergeOk) {
|
|
1527
1805
|
const lastEndsSpace = /\s$/.test(last.text);
|
|
1528
1806
|
const currStartsSpace = /^\s/.test(r.text);
|
|
1529
1807
|
const sep = gapMm > emMm * 0.08 && !lastEndsSpace && !currStartsSpace ? " " : "";
|
|
1530
1808
|
last.text = last.text + sep + r.text;
|
|
1531
1809
|
const newExtent = r.x - last.x + r.width;
|
|
1532
|
-
last.width = Math.max(last
|
|
1810
|
+
last.width = Math.max(extent(last), newExtent);
|
|
1533
1811
|
} else {
|
|
1534
1812
|
lines.push({ ...r });
|
|
1535
1813
|
}
|
|
1536
1814
|
}
|
|
1537
1815
|
const elements = [];
|
|
1538
|
-
|
|
1539
|
-
|
|
1816
|
+
const tRuns = lines.map((l) => {
|
|
1817
|
+
const cls = fontMap.get(l.fontName) || classifyFont(l.fontName || "");
|
|
1818
|
+
return { text: l.text, x: l.x, y: l.y, width: l.width, height: l.height, fontSize: l.fontSize, fontName: l.fontName, color: l.color, bold: cls.weight === "bold" };
|
|
1819
|
+
});
|
|
1820
|
+
const detected = options.detectTables === false ? [] : detectTables(tRuns, ops.shapes, pageW * PT_TO_MM2);
|
|
1821
|
+
const consumedLines = /* @__PURE__ */ new Set();
|
|
1822
|
+
const consumedShapes = /* @__PURE__ */ new Set();
|
|
1823
|
+
const tableAtLine = /* @__PURE__ */ new Map();
|
|
1824
|
+
for (const t of detected) {
|
|
1825
|
+
for (const k of t.lineIdx) consumedLines.add(k);
|
|
1826
|
+
for (const k of t.shapeIdx) consumedShapes.add(k);
|
|
1827
|
+
tableAtLine.set(Math.min(...t.lineIdx), t.element);
|
|
1828
|
+
}
|
|
1829
|
+
const pageWmm = pageW * PT_TO_MM2, pageHmm = pageH * PT_TO_MM2;
|
|
1830
|
+
ops.shapes.forEach((sh, shapeIdx) => {
|
|
1831
|
+
if (consumedShapes.has(shapeIdx)) return;
|
|
1832
|
+
if (sh.width < 0.3 && sh.height < 0.3) return;
|
|
1833
|
+
if (sh.x + sh.width <= 0 || sh.y + sh.height <= 0 || sh.x >= pageWmm || sh.y >= pageHmm) return;
|
|
1834
|
+
if (sh.kind === "rect" && sh.fill && !sh.stroke && sh.width * sh.height >= pageWmm * pageHmm * 0.9) return;
|
|
1540
1835
|
const shapeType = sh.kind;
|
|
1541
1836
|
const shape = {
|
|
1542
1837
|
type: "shape",
|
|
@@ -1552,7 +1847,7 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
|
1552
1847
|
shape.style = { opacity: Math.round(sh.opacity * 100) / 100 };
|
|
1553
1848
|
}
|
|
1554
1849
|
elements.push(shape);
|
|
1555
|
-
}
|
|
1850
|
+
});
|
|
1556
1851
|
const imgs = await extractImages(page, ops.imagePositions, runtime, dataUrlCache);
|
|
1557
1852
|
for (const { pos, dataUrl } of imgs) {
|
|
1558
1853
|
let resourceKey = resourceKeyByName.get(pos.name);
|
|
@@ -1575,7 +1870,98 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
|
1575
1870
|
fit: "fill"
|
|
1576
1871
|
});
|
|
1577
1872
|
}
|
|
1873
|
+
const sizeChars = /* @__PURE__ */ new Map();
|
|
1578
1874
|
for (const l of lines) {
|
|
1875
|
+
const k = Math.round(l.fontSize * 2) / 2;
|
|
1876
|
+
sizeChars.set(k, (sizeChars.get(k) ?? 0) + l.text.length);
|
|
1877
|
+
}
|
|
1878
|
+
const bodyFontSize = [...sizeChars.entries()].sort((a, b) => b[1] - a[1])[0]?.[0] ?? 0;
|
|
1879
|
+
const rowOf = /* @__PURE__ */ new Map();
|
|
1880
|
+
const rowStartOf = /* @__PURE__ */ new Map();
|
|
1881
|
+
const nextOnRow = /* @__PURE__ */ new Map();
|
|
1882
|
+
{
|
|
1883
|
+
const order = lines.map((_, i) => i).filter((i) => !consumedLines.has(i));
|
|
1884
|
+
for (let a = 0; a < order.length; a++) {
|
|
1885
|
+
const i = order[a], li = lines[i];
|
|
1886
|
+
const tolY = Math.max(0.6, li.fontSize * PT_TO_MM2 * 0.35);
|
|
1887
|
+
let bestNext = -1, bestX = Infinity;
|
|
1888
|
+
for (let b = 0; b < order.length; b++) {
|
|
1889
|
+
const j = order[b], lj = lines[j];
|
|
1890
|
+
if (j === i || Math.abs(lj.y - li.y) > tolY || lj.x <= li.x) continue;
|
|
1891
|
+
if (lj.x < bestX) {
|
|
1892
|
+
bestX = lj.x;
|
|
1893
|
+
bestNext = j;
|
|
1894
|
+
}
|
|
1895
|
+
}
|
|
1896
|
+
if (bestNext >= 0) nextOnRow.set(i, bestNext);
|
|
1897
|
+
}
|
|
1898
|
+
const seen = /* @__PURE__ */ new Set();
|
|
1899
|
+
for (const i of order) {
|
|
1900
|
+
if (seen.has(i)) continue;
|
|
1901
|
+
const row = [i];
|
|
1902
|
+
seen.add(i);
|
|
1903
|
+
let cur = i;
|
|
1904
|
+
while (nextOnRow.has(cur)) {
|
|
1905
|
+
const j = nextOnRow.get(cur), lc = lines[cur], lj = lines[j];
|
|
1906
|
+
const em = Math.min(lc.fontSize, lj.fontSize) * PT_TO_MM2;
|
|
1907
|
+
const gap = lj.x - (lc.x + lc.width);
|
|
1908
|
+
if (gap < -em * 0.3 || gap > em * 0.6) break;
|
|
1909
|
+
row.push(j);
|
|
1910
|
+
seen.add(j);
|
|
1911
|
+
cur = j;
|
|
1912
|
+
}
|
|
1913
|
+
rowOf.set(i, row);
|
|
1914
|
+
for (const j of row) rowStartOf.set(j, i);
|
|
1915
|
+
}
|
|
1916
|
+
}
|
|
1917
|
+
const runStyle = (l) => {
|
|
1918
|
+
const cls = fontMap.get(l.fontName) || classifyFont(l.fontName || "");
|
|
1919
|
+
return { cls, bold: cls.weight === "bold", italic: cls.style === "italic" };
|
|
1920
|
+
};
|
|
1921
|
+
lines.forEach((l, lineIdx) => {
|
|
1922
|
+
const tableEl = tableAtLine.get(lineIdx);
|
|
1923
|
+
if (tableEl) elements.push(tableEl);
|
|
1924
|
+
if (consumedLines.has(lineIdx)) return;
|
|
1925
|
+
const row = rowOf.get(lineIdx);
|
|
1926
|
+
if (!row) return;
|
|
1927
|
+
if (row.length > 1) {
|
|
1928
|
+
const first = lines[row[0]], last = lines[row[row.length - 1]];
|
|
1929
|
+
const base = runStyle(first);
|
|
1930
|
+
const rowEnd = last.x + last.width;
|
|
1931
|
+
const measuredW = Math.max((rowEnd - first.x) * 1.2 + first.fontSize * PT_TO_MM2 * 0.4, first.fontSize * PT_TO_MM2);
|
|
1932
|
+
const nextIdx2 = nextOnRow.get(row[row.length - 1]);
|
|
1933
|
+
const cap2 = nextIdx2 != null ? lines[nextIdx2].x - first.x - first.fontSize * PT_TO_MM2 * 0.3 : pageWmm - first.x;
|
|
1934
|
+
const runs2 = [];
|
|
1935
|
+
row.forEach((idx, k) => {
|
|
1936
|
+
const r = lines[idx];
|
|
1937
|
+
const st = runStyle(r);
|
|
1938
|
+
let text2 = r.text;
|
|
1939
|
+
if (k > 0) {
|
|
1940
|
+
const prev2 = lines[row[k - 1]];
|
|
1941
|
+
const gap = r.x - (prev2.x + prev2.width);
|
|
1942
|
+
if (gap > r.fontSize * PT_TO_MM2 * 0.08 && !/\s$/.test(prev2.text) && !/^\s/.test(text2)) text2 = " " + text2;
|
|
1943
|
+
}
|
|
1944
|
+
const run = { text: text2 };
|
|
1945
|
+
if (st.bold) run.bold = true;
|
|
1946
|
+
if (st.italic) run.italic = true;
|
|
1947
|
+
if (r.color !== "#000000") run.color = r.color;
|
|
1948
|
+
if (Math.abs(r.fontSize - first.fontSize) >= 0.5) run.fontSize = Math.round(r.fontSize * 10) / 10;
|
|
1949
|
+
if (st.cls.family !== base.cls.family) run.fontFamily = st.cls.family;
|
|
1950
|
+
const lk = findLinkForRun2(r);
|
|
1951
|
+
if (lk) run.link = lk.url ? lk.url : lk.destPage != null ? { type: "internal", target: `#page-${lk.destPage + 1}` } : void 0;
|
|
1952
|
+
runs2.push(run);
|
|
1953
|
+
});
|
|
1954
|
+
const style2 = { fontSize: Math.round(first.fontSize * 10) / 10, fontFamily: base.cls.family };
|
|
1955
|
+
if (first.opacity < 0.999) style2.opacity = Math.round(first.opacity * 100) / 100;
|
|
1956
|
+
elements.push({
|
|
1957
|
+
type: "richtext",
|
|
1958
|
+
runs: runs2,
|
|
1959
|
+
position: { x: Math.max(0, Math.round(first.x * 100) / 100), y: Math.max(0, Math.round(Math.min(...row.map((i) => lines[i].y)) * 100) / 100) },
|
|
1960
|
+
width: Math.max(2, Math.round(Math.max(first.fontSize * PT_TO_MM2, Math.min(measuredW, cap2)) * 100) / 100),
|
|
1961
|
+
style: style2
|
|
1962
|
+
});
|
|
1963
|
+
return;
|
|
1964
|
+
}
|
|
1579
1965
|
const cls = fontMap.get(l.fontName) || classifyFont(l.fontName || "");
|
|
1580
1966
|
const style = {
|
|
1581
1967
|
fontSize: Math.round(l.fontSize * 10) / 10,
|
|
@@ -1586,10 +1972,11 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
|
1586
1972
|
if (l.color !== "#000000") style.color = l.color;
|
|
1587
1973
|
if (l.opacity < 0.999) style.opacity = Math.round(l.opacity * 100) / 100;
|
|
1588
1974
|
const link = findLinkForRun2(l);
|
|
1589
|
-
const
|
|
1590
|
-
const measured = Math.max(l.width + l.fontSize * PT_TO_MM * 0.4, l.fontSize * PT_TO_MM);
|
|
1975
|
+
const measured = Math.max(l.width * 1.2 + l.fontSize * PT_TO_MM2 * 0.4, l.fontSize * PT_TO_MM2);
|
|
1591
1976
|
const remaining = Math.max(measured, pageWmm - l.x);
|
|
1592
|
-
const
|
|
1977
|
+
const nextIdx = nextOnRow.get(lineIdx);
|
|
1978
|
+
const cap = nextIdx != null ? Math.max(l.fontSize * PT_TO_MM2, lines[nextIdx].x - l.x - l.fontSize * PT_TO_MM2 * 0.3) : Infinity;
|
|
1979
|
+
const elWidth = Math.min(measured, remaining, cap);
|
|
1593
1980
|
const text = {
|
|
1594
1981
|
type: "text",
|
|
1595
1982
|
content: l.text,
|
|
@@ -1597,18 +1984,26 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
|
1597
1984
|
width: Math.max(2, Math.round(elWidth * 100) / 100),
|
|
1598
1985
|
style
|
|
1599
1986
|
};
|
|
1600
|
-
if (cls.weight === "bold") {
|
|
1601
|
-
|
|
1602
|
-
|
|
1603
|
-
else if (l.fontSize >=
|
|
1987
|
+
if (cls.weight === "bold" && l.text.trim().length <= 120 && !consumedLines.has(lineIdx)) {
|
|
1988
|
+
const ratio = bodyFontSize > 0 ? l.fontSize / bodyFontSize : 1;
|
|
1989
|
+
if (l.fontSize >= 22 || ratio >= 1.8) text.heading = 1;
|
|
1990
|
+
else if (l.fontSize >= 17 || ratio >= 1.35) text.heading = 2;
|
|
1991
|
+
else if (l.fontSize >= 16 || ratio >= 1.2) text.heading = 3;
|
|
1604
1992
|
}
|
|
1605
1993
|
if (text.heading) text.tocEntry = text.content;
|
|
1606
1994
|
if (link) {
|
|
1607
1995
|
if (link.url) text.link = link.url;
|
|
1608
1996
|
else if (link.destPage != null) text.link = { type: "internal", target: `#page-${link.destPage + 1}` };
|
|
1609
1997
|
}
|
|
1998
|
+
const prev = elements[elements.length - 1];
|
|
1999
|
+
if (text.heading && prev && prev.type === "text" && prev.heading === text.heading && !link && !prev.link && Math.abs(prev.style?.fontSize - style.fontSize) < 0.5 && Math.abs(prev.position.x - text.position.x) < 1 && text.position.y - prev.position.y < l.fontSize * PT_TO_MM2 * 2.2 && text.position.y > prev.position.y) {
|
|
2000
|
+
prev.content = `${prev.content} ${text.content}`.replace(/\s+/g, " ");
|
|
2001
|
+
prev.tocEntry = prev.content;
|
|
2002
|
+
prev.width = Math.max(prev.width ?? 0, text.width ?? 0);
|
|
2003
|
+
return;
|
|
2004
|
+
}
|
|
1610
2005
|
elements.push(text);
|
|
1611
|
-
}
|
|
2006
|
+
});
|
|
1612
2007
|
for (const w of formWidgets) {
|
|
1613
2008
|
if (w.pushButton) continue;
|
|
1614
2009
|
const baseEl = {
|
|
@@ -1643,7 +2038,7 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
|
1643
2038
|
}
|
|
1644
2039
|
pages.push({
|
|
1645
2040
|
id: `page-${pi}`,
|
|
1646
|
-
pageSize: { width: Math.round(pageW *
|
|
2041
|
+
pageSize: { width: Math.round(pageW * PT_TO_MM2 * 100) / 100, height: Math.round(pageH * PT_TO_MM2 * 100) / 100 },
|
|
1647
2042
|
margins: { top: 0, right: 0, bottom: 0, left: 0 },
|
|
1648
2043
|
elements
|
|
1649
2044
|
});
|
|
@@ -1796,13 +2191,13 @@ async function importPdfToJdf2(source, title, options = {}) {
|
|
|
1796
2191
|
|
|
1797
2192
|
// src/commands/import-pdf.ts
|
|
1798
2193
|
async function importPdf(inputPath, outputPath, options = {}) {
|
|
1799
|
-
const input =
|
|
1800
|
-
if (!
|
|
2194
|
+
const input = path7.resolve(inputPath);
|
|
2195
|
+
if (!fs7.existsSync(input)) {
|
|
1801
2196
|
console.error(`File not found: ${input}`);
|
|
1802
2197
|
process.exit(1);
|
|
1803
2198
|
}
|
|
1804
2199
|
console.log(`Importing: ${input}`);
|
|
1805
|
-
const title =
|
|
2200
|
+
const title = path7.basename(input, path7.extname(input));
|
|
1806
2201
|
const t0 = Date.now();
|
|
1807
2202
|
const doc = await importPdfToJdf2(input, title, {
|
|
1808
2203
|
password: options.password,
|
|
@@ -1811,7 +2206,7 @@ async function importPdf(inputPath, outputPath, options = {}) {
|
|
|
1811
2206
|
console.log(`Parsed in ${((Date.now() - t0) / 1e3).toFixed(1)}s \u2014 ${doc.pages.length} page(s)`);
|
|
1812
2207
|
let output;
|
|
1813
2208
|
if (outputPath) {
|
|
1814
|
-
output =
|
|
2209
|
+
output = path7.resolve(outputPath);
|
|
1815
2210
|
} else {
|
|
1816
2211
|
const stem = input.replace(/\.pdf$/i, "");
|
|
1817
2212
|
const wantJdfx = !options.forceJson && shouldUseJdfx(doc);
|
|
@@ -1820,11 +2215,11 @@ async function importPdf(inputPath, outputPath, options = {}) {
|
|
|
1820
2215
|
console.log(`Output: ${output}`);
|
|
1821
2216
|
if (output.toLowerCase().endsWith(".jdfx")) {
|
|
1822
2217
|
const { bytes, manifest } = await packJdfx(doc);
|
|
1823
|
-
|
|
2218
|
+
fs7.writeFileSync(output, bytes);
|
|
1824
2219
|
console.log(`
|
|
1825
2220
|
Done! Created ${doc.pages.length} page(s), ${manifest.assets.length} asset(s) bundled`);
|
|
1826
2221
|
} else {
|
|
1827
|
-
|
|
2222
|
+
fs7.writeFileSync(output, JSON.stringify(doc, null, 2));
|
|
1828
2223
|
console.log(`
|
|
1829
2224
|
Done! Created ${doc.pages.length} page(s)`);
|
|
1830
2225
|
}
|
|
@@ -1837,23 +2232,23 @@ var ImportJsonError = class extends Error {
|
|
|
1837
2232
|
}
|
|
1838
2233
|
};
|
|
1839
2234
|
async function importJson(inputPath, outputPath, options = {}) {
|
|
1840
|
-
const input =
|
|
1841
|
-
if (!
|
|
2235
|
+
const input = path7.resolve(inputPath);
|
|
2236
|
+
if (!fs7.existsSync(input)) {
|
|
1842
2237
|
throw new ImportJsonError(`File not found: ${input}`);
|
|
1843
2238
|
}
|
|
1844
2239
|
console.log(`Importing: ${input}`);
|
|
1845
|
-
const raw =
|
|
2240
|
+
const raw = fs7.readFileSync(input, "utf-8");
|
|
1846
2241
|
let parsed;
|
|
1847
2242
|
try {
|
|
1848
2243
|
parsed = JSON.parse(raw);
|
|
1849
2244
|
} catch (e) {
|
|
1850
2245
|
throw new ImportJsonError(`Not valid JSON: ${e.message}`);
|
|
1851
2246
|
}
|
|
1852
|
-
const title =
|
|
2247
|
+
const title = path7.basename(input, path7.extname(input));
|
|
1853
2248
|
const doc = normaliseToJdf(parsed, title);
|
|
1854
2249
|
let output;
|
|
1855
2250
|
if (outputPath) {
|
|
1856
|
-
output =
|
|
2251
|
+
output = path7.resolve(outputPath);
|
|
1857
2252
|
} else {
|
|
1858
2253
|
const stem = input.replace(/\.json$/i, "");
|
|
1859
2254
|
const wantJdfx = !options.forceJson && shouldUseJdfx(doc);
|
|
@@ -1862,11 +2257,11 @@ async function importJson(inputPath, outputPath, options = {}) {
|
|
|
1862
2257
|
console.log(`Output: ${output}`);
|
|
1863
2258
|
if (output.toLowerCase().endsWith(".jdfx")) {
|
|
1864
2259
|
const { bytes, manifest } = await packJdfx(doc);
|
|
1865
|
-
|
|
2260
|
+
fs7.writeFileSync(output, bytes);
|
|
1866
2261
|
console.log(`
|
|
1867
2262
|
Done! Created ${doc.pages.length} page(s), ${manifest.assets.length} asset(s) bundled`);
|
|
1868
2263
|
} else {
|
|
1869
|
-
|
|
2264
|
+
fs7.writeFileSync(output, JSON.stringify(doc, null, 2));
|
|
1870
2265
|
console.log(`
|
|
1871
2266
|
Done! Created ${doc.pages.length} page(s)`);
|
|
1872
2267
|
}
|
|
@@ -1942,6 +2337,47 @@ function wrapElements(elements, title, meta) {
|
|
|
1942
2337
|
]
|
|
1943
2338
|
};
|
|
1944
2339
|
}
|
|
2340
|
+
var DEFAULT_TRANSCRIPT_WINDOW = 45;
|
|
2341
|
+
var fmtTime = (sec) => {
|
|
2342
|
+
const s = Math.max(0, Math.round(sec));
|
|
2343
|
+
const h = Math.floor(s / 3600), m = Math.floor(s % 3600 / 60), r = s % 60;
|
|
2344
|
+
return h ? `${h}:${String(m).padStart(2, "0")}:${String(r).padStart(2, "0")}` : `${String(m).padStart(2, "0")}:${String(r).padStart(2, "0")}`;
|
|
2345
|
+
};
|
|
2346
|
+
function transcriptChunks(el, elementId2, page, crumb, windowSec, maxTokens) {
|
|
2347
|
+
const segs = Array.isArray(el?.transcript?.segments) ? el.transcript.segments : [];
|
|
2348
|
+
if (!segs.length) return [];
|
|
2349
|
+
const chapters = Array.isArray(el?.chapters) ? [...el.chapters].sort((a, b) => a.t - b.t) : [];
|
|
2350
|
+
const chapterAt = (t) => {
|
|
2351
|
+
let cur = null;
|
|
2352
|
+
for (const c of chapters) {
|
|
2353
|
+
if (c.t <= t + 1e-6) cur = c;
|
|
2354
|
+
else break;
|
|
2355
|
+
}
|
|
2356
|
+
return cur;
|
|
2357
|
+
};
|
|
2358
|
+
const out = [];
|
|
2359
|
+
let win = [];
|
|
2360
|
+
const flush = () => {
|
|
2361
|
+
if (!win.length) return;
|
|
2362
|
+
const t0 = win[0].t0, t1 = win[win.length - 1].t1;
|
|
2363
|
+
const body = win.map((sg) => sg.speaker ? `${sg.speaker}: ${sg.text}` : sg.text).join(" ").replace(/\s+/g, " ").trim();
|
|
2364
|
+
const text = `[${fmtTime(t0)}\u2013${fmtTime(t1)}] ${body}`;
|
|
2365
|
+
const chapter = chapterAt(t0);
|
|
2366
|
+
const path9 = [...crumb, ...el.title ? [String(el.title)] : [], ...chapter ? [String(chapter.title)] : []];
|
|
2367
|
+
out.push({ id: `${elementId2}@${Math.round(t0)}`, text, path: path9, page, types: ["video"], tokens: estimateTokens(text), hash: hashText(text), media: { element: elementId2, t0, t1 } });
|
|
2368
|
+
win = [];
|
|
2369
|
+
};
|
|
2370
|
+
for (const sg of segs) {
|
|
2371
|
+
if (typeof sg?.text !== "string" || !sg.text.trim()) continue;
|
|
2372
|
+
const startsNewChapter = win.length && chapterAt(sg.t0) !== chapterAt(win[0].t0);
|
|
2373
|
+
const spansWindow = win.length && sg.t1 - win[0].t0 > windowSec;
|
|
2374
|
+
const overBudget = win.length && estimateTokens(win.map((w) => w.text).join(" ") + sg.text) > maxTokens;
|
|
2375
|
+
if (startsNewChapter || spansWindow || overBudget) flush();
|
|
2376
|
+
win.push(sg);
|
|
2377
|
+
}
|
|
2378
|
+
flush();
|
|
2379
|
+
return out;
|
|
2380
|
+
}
|
|
1945
2381
|
var DEFAULT_MAX_TOKENS = 512;
|
|
1946
2382
|
function estimateTokens(text) {
|
|
1947
2383
|
return Math.ceil(text.length / 4);
|
|
@@ -1964,11 +2400,11 @@ function serializeElement(el) {
|
|
|
1964
2400
|
case "richtext":
|
|
1965
2401
|
return (e.runs || []).map((r) => r.text ?? "").join("").trim();
|
|
1966
2402
|
case "list": {
|
|
1967
|
-
const
|
|
2403
|
+
const walk2 = (items, depth = 0) => (items || []).flatMap((it) => {
|
|
1968
2404
|
const line = " ".repeat(depth) + "- " + String(it.content ?? "").trim();
|
|
1969
|
-
return it.children?.length ? [line, ...
|
|
2405
|
+
return it.children?.length ? [line, ...walk2(it.children, depth + 1)] : [line];
|
|
1970
2406
|
});
|
|
1971
|
-
return
|
|
2407
|
+
return walk2(e.items).join("\n");
|
|
1972
2408
|
}
|
|
1973
2409
|
case "table": {
|
|
1974
2410
|
const headers = e.headers ?? e.columns?.map((c) => c.header || "").filter((h) => h) ?? [];
|
|
@@ -1999,6 +2435,8 @@ function serializeElement(el) {
|
|
|
1999
2435
|
return `${e.checked ? "[x]" : "[ ]"} ${e.label ?? ""}`.trim();
|
|
2000
2436
|
case "image":
|
|
2001
2437
|
return e.alt ? `[image: ${e.alt}]` : "";
|
|
2438
|
+
case "video":
|
|
2439
|
+
return e.title ? `[video: ${e.title}]` : "";
|
|
2002
2440
|
case "toc":
|
|
2003
2441
|
case "shape":
|
|
2004
2442
|
case "signature":
|
|
@@ -2038,8 +2476,15 @@ function makeChunk(group, breadcrumb) {
|
|
|
2038
2476
|
function chunkDocument(doc, options = {}) {
|
|
2039
2477
|
const strategy = options.strategy ?? "section";
|
|
2040
2478
|
const maxTokens = options.maxTokens ?? DEFAULT_MAX_TOKENS;
|
|
2479
|
+
const windowSec = options.transcriptWindowSec ?? DEFAULT_TRANSCRIPT_WINDOW;
|
|
2041
2480
|
const flat = flatten(doc);
|
|
2042
2481
|
const chunks = [];
|
|
2482
|
+
const withTranscripts = (group, crumb2, c) => {
|
|
2483
|
+
if (c) chunks.push(c);
|
|
2484
|
+
for (const f of group) {
|
|
2485
|
+
if (f.el.type === "video") chunks.push(...transcriptChunks(f.el, f.id, f.page, crumb2, windowSec, maxTokens));
|
|
2486
|
+
}
|
|
2487
|
+
};
|
|
2043
2488
|
if (strategy === "element") {
|
|
2044
2489
|
const crumb2 = [];
|
|
2045
2490
|
for (const f of flat) {
|
|
@@ -2048,8 +2493,7 @@ function chunkDocument(doc, options = {}) {
|
|
|
2048
2493
|
crumb2.length = Math.max(0, lvl - 1);
|
|
2049
2494
|
crumb2[lvl - 1] = serializeElement(f.el);
|
|
2050
2495
|
}
|
|
2051
|
-
|
|
2052
|
-
if (c) chunks.push(c);
|
|
2496
|
+
withTranscripts([f], crumb2, makeChunk([f], crumb2));
|
|
2053
2497
|
}
|
|
2054
2498
|
return chunks;
|
|
2055
2499
|
}
|
|
@@ -2058,8 +2502,7 @@ function chunkDocument(doc, options = {}) {
|
|
|
2058
2502
|
let buf2 = [];
|
|
2059
2503
|
let bufTokens = 0;
|
|
2060
2504
|
const flush = () => {
|
|
2061
|
-
|
|
2062
|
-
if (c) chunks.push(c);
|
|
2505
|
+
withTranscripts(buf2, crumb2, makeChunk(buf2, crumb2));
|
|
2063
2506
|
buf2 = [];
|
|
2064
2507
|
bufTokens = 0;
|
|
2065
2508
|
};
|
|
@@ -2086,22 +2529,21 @@ function chunkDocument(doc, options = {}) {
|
|
|
2086
2529
|
for (const f of buf) {
|
|
2087
2530
|
const t = estimateTokens(serializeElement(f.el));
|
|
2088
2531
|
if (subTokens + t > maxTokens && sub.length > 0) {
|
|
2089
|
-
|
|
2090
|
-
if (c2) chunks.push(c2);
|
|
2532
|
+
withTranscripts(sub, crumb, makeChunk(sub, crumb));
|
|
2091
2533
|
sub = [];
|
|
2092
2534
|
subTokens = 0;
|
|
2093
2535
|
}
|
|
2094
2536
|
sub.push(f);
|
|
2095
2537
|
subTokens += t;
|
|
2096
2538
|
}
|
|
2097
|
-
|
|
2098
|
-
if (c) chunks.push(c);
|
|
2539
|
+
withTranscripts(sub, crumb, makeChunk(sub, crumb));
|
|
2099
2540
|
buf = [];
|
|
2100
2541
|
};
|
|
2101
2542
|
for (const f of flat) {
|
|
2102
2543
|
const lvl = headingLevel(f.el);
|
|
2103
2544
|
if (lvl != null) {
|
|
2104
|
-
|
|
2545
|
+
const onlyHeadings = buf.length > 0 && buf.every((b) => headingLevel(b.el) != null);
|
|
2546
|
+
if (!onlyHeadings) flushSection();
|
|
2105
2547
|
crumb.length = Math.max(0, lvl - 1);
|
|
2106
2548
|
crumb[lvl - 1] = serializeElement(f.el);
|
|
2107
2549
|
}
|
|
@@ -2112,34 +2554,34 @@ function chunkDocument(doc, options = {}) {
|
|
|
2112
2554
|
}
|
|
2113
2555
|
async function loadJdf(filePath) {
|
|
2114
2556
|
if (filePath.toLowerCase().endsWith(".jdfx")) {
|
|
2115
|
-
const zip = await JSZip.loadAsync(
|
|
2557
|
+
const zip = await JSZip.loadAsync(fs7.readFileSync(filePath));
|
|
2116
2558
|
const docFile = zip.file(JDFX_DOCUMENT_PATH);
|
|
2117
2559
|
if (!docFile) throw new Error(`Bundle missing ${JDFX_DOCUMENT_PATH}`);
|
|
2118
2560
|
return JSON.parse(await docFile.async("string"));
|
|
2119
2561
|
}
|
|
2120
|
-
return JSON.parse(
|
|
2562
|
+
return JSON.parse(fs7.readFileSync(filePath, "utf-8"));
|
|
2121
2563
|
}
|
|
2122
2564
|
async function chunkFile(inputPath, opts = {}) {
|
|
2123
|
-
const input =
|
|
2124
|
-
if (!
|
|
2565
|
+
const input = path7.resolve(inputPath);
|
|
2566
|
+
if (!fs7.existsSync(input)) throw new Error(`File not found: ${input}`);
|
|
2125
2567
|
const doc = await loadJdf(input);
|
|
2126
2568
|
const strategy = opts.strategy ?? "section";
|
|
2127
|
-
const chunks = chunkDocument(doc, { strategy, maxTokens: opts.maxTokens });
|
|
2569
|
+
const chunks = chunkDocument(doc, { strategy, maxTokens: opts.maxTokens, transcriptWindowSec: opts.transcriptWindowSec });
|
|
2128
2570
|
const format = opts.format ?? "jsonl";
|
|
2129
2571
|
console.log(`Chunking: ${input}`);
|
|
2130
2572
|
console.log(`Strategy: ${strategy}${opts.maxTokens ? ` (max ${opts.maxTokens} tokens)` : ""}`);
|
|
2131
2573
|
if (format === "inline") {
|
|
2132
|
-
const out = opts.output ?
|
|
2574
|
+
const out = opts.output ? path7.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".jdf");
|
|
2133
2575
|
const withIndex = { ...doc, index: { chunker: `jdf-${strategy}-v1`, chunks } };
|
|
2134
|
-
|
|
2576
|
+
fs7.writeFileSync(out, JSON.stringify(withIndex, null, 2));
|
|
2135
2577
|
console.log(`Output: ${out} (${chunks.length} chunks in "index" block)`);
|
|
2136
2578
|
} else if (format === "json") {
|
|
2137
|
-
const out = opts.output ?
|
|
2138
|
-
|
|
2579
|
+
const out = opts.output ? path7.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".chunks.json");
|
|
2580
|
+
fs7.writeFileSync(out, JSON.stringify(chunks, null, 2));
|
|
2139
2581
|
console.log(`Output: ${out} (${chunks.length} chunks)`);
|
|
2140
2582
|
} else {
|
|
2141
|
-
const out = opts.output ?
|
|
2142
|
-
|
|
2583
|
+
const out = opts.output ? path7.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".chunks.jsonl");
|
|
2584
|
+
fs7.writeFileSync(out, chunks.map((c) => JSON.stringify(c)).join("\n") + "\n");
|
|
2143
2585
|
console.log(`Output: ${out} (${chunks.length} chunks)`);
|
|
2144
2586
|
}
|
|
2145
2587
|
const totalTokens = chunks.reduce((a, c) => a + c.tokens, 0);
|
|
@@ -2311,17 +2753,17 @@ async function embedOpenAI(model, inputs) {
|
|
|
2311
2753
|
}
|
|
2312
2754
|
async function loadJdf2(filePath) {
|
|
2313
2755
|
if (filePath.toLowerCase().endsWith(".jdfx")) {
|
|
2314
|
-
const zip = await JSZip.loadAsync(
|
|
2756
|
+
const zip = await JSZip.loadAsync(fs7.readFileSync(filePath));
|
|
2315
2757
|
const docFile = zip.file(JDFX_DOCUMENT_PATH);
|
|
2316
2758
|
if (!docFile) throw new Error(`Bundle missing ${JDFX_DOCUMENT_PATH}`);
|
|
2317
2759
|
return JSON.parse(await docFile.async("string"));
|
|
2318
2760
|
}
|
|
2319
|
-
return JSON.parse(
|
|
2761
|
+
return JSON.parse(fs7.readFileSync(filePath, "utf-8"));
|
|
2320
2762
|
}
|
|
2321
2763
|
function loadCache(cachePath) {
|
|
2322
2764
|
try {
|
|
2323
|
-
if (!
|
|
2324
|
-
return JSON.parse(
|
|
2765
|
+
if (!fs7.existsSync(cachePath)) return null;
|
|
2766
|
+
return JSON.parse(fs7.readFileSync(cachePath, "utf-8"));
|
|
2325
2767
|
} catch {
|
|
2326
2768
|
return null;
|
|
2327
2769
|
}
|
|
@@ -2332,15 +2774,15 @@ function batched(items, size) {
|
|
|
2332
2774
|
return out;
|
|
2333
2775
|
}
|
|
2334
2776
|
async function embedFile(inputPath, opts = {}) {
|
|
2335
|
-
const input =
|
|
2336
|
-
if (!
|
|
2777
|
+
const input = path7.resolve(inputPath);
|
|
2778
|
+
if (!fs7.existsSync(input)) throw new Error(`File not found: ${input}`);
|
|
2337
2779
|
const provider = opts.provider ?? "ollama";
|
|
2338
2780
|
const model = opts.model ?? DEFAULT_MODEL[provider];
|
|
2339
2781
|
const strategy = opts.strategy ?? "section";
|
|
2340
2782
|
const doc = await loadJdf2(input);
|
|
2341
|
-
const chunks = chunkDocument(doc, { strategy, maxTokens: opts.maxTokens });
|
|
2342
|
-
const output = opts.output ?
|
|
2343
|
-
const cachePath = opts.cache ?
|
|
2783
|
+
const chunks = chunkDocument(doc, { strategy, maxTokens: opts.maxTokens, transcriptWindowSec: opts.transcriptWindowSec });
|
|
2784
|
+
const output = opts.output ? path7.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".embeddings.json");
|
|
2785
|
+
const cachePath = opts.cache ? path7.resolve(opts.cache) : output;
|
|
2344
2786
|
console.log(`Embedding: ${input}`);
|
|
2345
2787
|
console.log(`Provider: ${provider} / ${model}${provider === "ollama" ? " (local \u2014 no data leaves this machine)" : " (remote API)"}`);
|
|
2346
2788
|
console.log(`Strategy: ${strategy} \u2192 ${chunks.length} chunks`);
|
|
@@ -2379,11 +2821,295 @@ async function embedFile(inputPath, opts = {}) {
|
|
|
2379
2821
|
chunker: `jdf-${strategy}-v1`,
|
|
2380
2822
|
vectors
|
|
2381
2823
|
};
|
|
2382
|
-
|
|
2824
|
+
fs7.writeFileSync(output, JSON.stringify(sidecar));
|
|
2383
2825
|
console.log(`
|
|
2384
2826
|
Done! ${Object.keys(vectors).length} vectors (${dims}-dim) \u2192 ${output}`);
|
|
2385
2827
|
return sidecar;
|
|
2386
2828
|
}
|
|
2829
|
+
var toSec = (ts) => {
|
|
2830
|
+
const m = ts.trim().replace(",", ".").match(/^(?:(\d+):)?(\d{1,2}):(\d{2}(?:\.\d+)?)$/);
|
|
2831
|
+
if (!m) throw new Error(`bad timestamp "${ts}"`);
|
|
2832
|
+
return (m[1] ? Number(m[1]) * 3600 : 0) + Number(m[2]) * 60 + Number(m[3]);
|
|
2833
|
+
};
|
|
2834
|
+
function parseSubtitles(text, filename = "") {
|
|
2835
|
+
const trimmed = text.replace(/^/, "").trim();
|
|
2836
|
+
if (trimmed.startsWith("{") || trimmed.startsWith("[")) {
|
|
2837
|
+
const j = JSON.parse(trimmed);
|
|
2838
|
+
const arr = Array.isArray(j) ? j : Array.isArray(j.segments) ? j.segments : [];
|
|
2839
|
+
return arr.map((sg) => ({
|
|
2840
|
+
t0: Number(sg.t0 ?? sg.start ?? sg.from ?? 0),
|
|
2841
|
+
t1: Number(sg.t1 ?? sg.end ?? sg.to ?? 0),
|
|
2842
|
+
text: String(sg.text ?? "").trim(),
|
|
2843
|
+
...sg.speaker ? { speaker: String(sg.speaker) } : {}
|
|
2844
|
+
})).filter((sg) => sg.text);
|
|
2845
|
+
}
|
|
2846
|
+
const segs = [];
|
|
2847
|
+
for (const block of trimmed.split(/\r?\n\r?\n+/)) {
|
|
2848
|
+
const lines = block.split(/\r?\n/).filter((l) => l.trim() !== "" && l.trim() !== "WEBVTT");
|
|
2849
|
+
const ti = lines.findIndex((l) => l.includes("-->"));
|
|
2850
|
+
if (ti < 0) continue;
|
|
2851
|
+
const [a, b] = lines[ti].split("-->").map((x) => x.trim().split(/\s+/)[0]);
|
|
2852
|
+
const body = lines.slice(ti + 1).join(" ").replace(/<[^>]+>/g, "").replace(/\s+/g, " ").trim();
|
|
2853
|
+
if (!body) continue;
|
|
2854
|
+
segs.push({ t0: toSec(a), t1: toSec(b), text: body });
|
|
2855
|
+
}
|
|
2856
|
+
if (!segs.length) throw new Error(`no cues found in ${filename || "subtitle input"} (expected SRT, WebVTT or JSON segments)`);
|
|
2857
|
+
return segs;
|
|
2858
|
+
}
|
|
2859
|
+
function parseChapters(text) {
|
|
2860
|
+
const t = text.trim();
|
|
2861
|
+
if (t.startsWith("[")) return JSON.parse(t).map((c) => ({ t: Number(c.t ?? c.start ?? 0), title: String(c.title ?? "") }));
|
|
2862
|
+
return t.split(/\r?\n/).map((l) => l.trim()).filter(Boolean).map((l) => {
|
|
2863
|
+
const m = l.match(/^(\S+)\s+(.+)$/);
|
|
2864
|
+
if (!m) throw new Error(`bad chapter line "${l}" (expected "mm:ss Title")`);
|
|
2865
|
+
return { t: toSec(m[1]), title: m[2].trim() };
|
|
2866
|
+
});
|
|
2867
|
+
}
|
|
2868
|
+
async function loadDoc(file) {
|
|
2869
|
+
if (file.toLowerCase().endsWith(".jdfx")) {
|
|
2870
|
+
const zip = await JSZip.loadAsync(fs7.readFileSync(file));
|
|
2871
|
+
const f = zip.file(JDFX_DOCUMENT_PATH);
|
|
2872
|
+
if (!f) throw new Error(`Bundle missing ${JDFX_DOCUMENT_PATH}`);
|
|
2873
|
+
const doc = JSON.parse(await f.async("string"));
|
|
2874
|
+
const manifest = zip.file("manifest.json") ? JSON.parse(await zip.file("manifest.json").async("string")) : { assets: [] };
|
|
2875
|
+
for (const a of manifest.assets ?? []) {
|
|
2876
|
+
const af = zip.file(a.path);
|
|
2877
|
+
if (!af) continue;
|
|
2878
|
+
const data = (await af.async("nodebuffer")).toString("base64");
|
|
2879
|
+
const res = { src: "embedded", mimeType: a.mimeType, data };
|
|
2880
|
+
doc.resources ??= {};
|
|
2881
|
+
if (/^video\//i.test(a.mimeType || "")) (doc.resources.videos ??= {})[a.id] = res;
|
|
2882
|
+
else (doc.resources.images ??= {})[a.id] = res;
|
|
2883
|
+
}
|
|
2884
|
+
return { doc, bundle: true, zip };
|
|
2885
|
+
}
|
|
2886
|
+
return { doc: JSON.parse(fs7.readFileSync(file, "utf-8")), bundle: false };
|
|
2887
|
+
}
|
|
2888
|
+
function findVideos(doc) {
|
|
2889
|
+
const out = [];
|
|
2890
|
+
const walk2 = (els, page) => {
|
|
2891
|
+
for (const el of els ?? []) {
|
|
2892
|
+
if (el?.type === "video") out.push({ el, page, index: out.length });
|
|
2893
|
+
if (el?.elements) walk2(el.elements, page);
|
|
2894
|
+
}
|
|
2895
|
+
};
|
|
2896
|
+
doc.pages.forEach((p, i) => walk2(p.elements, i + 1));
|
|
2897
|
+
return out;
|
|
2898
|
+
}
|
|
2899
|
+
async function clipToTempFile(doc, el, docDir) {
|
|
2900
|
+
const tmp = path7.join(fs7.mkdtempSync(path7.join(__require("os").tmpdir(), "jdf-transcribe-")), "clip.mp4");
|
|
2901
|
+
const res = el.resource ? doc.resources?.videos?.[el.resource] ?? doc.resources?.images?.[el.resource] : void 0;
|
|
2902
|
+
if (res?.data) {
|
|
2903
|
+
fs7.writeFileSync(tmp, Buffer.from(res.data.replace(/^data:[^,]*,/, ""), "base64"));
|
|
2904
|
+
return tmp;
|
|
2905
|
+
}
|
|
2906
|
+
if (res?.path) return path7.resolve(docDir, res.path);
|
|
2907
|
+
const src = el.src;
|
|
2908
|
+
if (!src) return null;
|
|
2909
|
+
if (src.startsWith("data:")) {
|
|
2910
|
+
fs7.writeFileSync(tmp, Buffer.from(src.replace(/^data:[^,]*,/, ""), "base64"));
|
|
2911
|
+
return tmp;
|
|
2912
|
+
}
|
|
2913
|
+
if (/^https?:\/\//i.test(src)) {
|
|
2914
|
+
const r = await fetch(src);
|
|
2915
|
+
if (!r.ok) throw new Error(`download failed ${r.status}: ${src}`);
|
|
2916
|
+
fs7.writeFileSync(tmp, Buffer.from(await r.arrayBuffer()));
|
|
2917
|
+
return tmp;
|
|
2918
|
+
}
|
|
2919
|
+
const local = path7.resolve(docDir, src);
|
|
2920
|
+
return fs7.existsSync(local) ? local : null;
|
|
2921
|
+
}
|
|
2922
|
+
function whisperCli(clip, model, language, prompt2) {
|
|
2923
|
+
const ffmpeg = spawnSync("ffmpeg", ["-version"]);
|
|
2924
|
+
if (ffmpeg.error) throw new Error("ffmpeg not found \u2014 needed to extract audio for whisper-cli (brew install ffmpeg)");
|
|
2925
|
+
const wav = clip.replace(/\.[^.]+$/, "") + ".16k.wav";
|
|
2926
|
+
const ex = spawnSync("ffmpeg", ["-y", "-i", clip, "-vn", "-ac", "1", "-ar", "16000", "-f", "wav", wav], { encoding: "utf-8" });
|
|
2927
|
+
if (ex.status !== 0) throw new Error(`ffmpeg failed: ${ex.stderr.slice(-400)}`);
|
|
2928
|
+
const args = ["-f", wav, "-oj", "-of", wav.replace(/\.wav$/, "")];
|
|
2929
|
+
if (model) args.push("-m", model);
|
|
2930
|
+
if (language) args.push("-l", language);
|
|
2931
|
+
if (prompt2) args.push("--prompt", prompt2);
|
|
2932
|
+
const run = spawnSync("whisper-cli", args, { encoding: "utf-8" });
|
|
2933
|
+
if (run.error) throw new Error("whisper-cli not found \u2014 install whisper.cpp (brew install whisper-cpp) or use --from / --provider openai");
|
|
2934
|
+
if (run.status !== 0) throw new Error(`whisper-cli failed: ${run.stderr.slice(-400)}`);
|
|
2935
|
+
const j = JSON.parse(fs7.readFileSync(wav.replace(/\.wav$/, "") + ".json", "utf-8"));
|
|
2936
|
+
const segs = j.transcription ?? j.segments ?? [];
|
|
2937
|
+
const ms = (x) => typeof x === "number" ? x / 1e3 : toSec(String(x).replace(",", "."));
|
|
2938
|
+
return segs.map((sg) => ({ t0: ms(sg.offsets?.from ?? sg.start), t1: ms(sg.offsets?.to ?? sg.end), text: String(sg.text ?? "").trim() })).filter((sg) => sg.text);
|
|
2939
|
+
}
|
|
2940
|
+
async function openaiTranscribe(clip, model, language, prompt2) {
|
|
2941
|
+
const key = process.env.OPENAI_API_KEY;
|
|
2942
|
+
if (!key) throw new Error("OPENAI_API_KEY is not set");
|
|
2943
|
+
const base = process.env.OPENAI_BASE_URL || "https://api.openai.com/v1";
|
|
2944
|
+
const form = new FormData();
|
|
2945
|
+
form.append("file", new Blob([fs7.readFileSync(clip)]), path7.basename(clip));
|
|
2946
|
+
form.append("model", model || "whisper-1");
|
|
2947
|
+
form.append("response_format", "verbose_json");
|
|
2948
|
+
form.append("timestamp_granularities[]", "segment");
|
|
2949
|
+
if (language) form.append("language", language);
|
|
2950
|
+
if (prompt2) form.append("prompt", prompt2);
|
|
2951
|
+
const r = await fetch(`${base}/audio/transcriptions`, { method: "POST", headers: { Authorization: `Bearer ${key}` }, body: form });
|
|
2952
|
+
if (!r.ok) throw new Error(`OpenAI transcription failed ${r.status}: ${(await r.text()).slice(0, 300)}`);
|
|
2953
|
+
const j = await r.json();
|
|
2954
|
+
return (j.segments ?? []).map((sg) => ({ t0: Number(sg.start), t1: Number(sg.end), text: String(sg.text).trim() })).filter((sg) => sg.text);
|
|
2955
|
+
}
|
|
2956
|
+
async function transcribeFile(inputPath, opts = {}) {
|
|
2957
|
+
const input = path7.resolve(inputPath);
|
|
2958
|
+
if (!fs7.existsSync(input)) throw new Error(`File not found: ${input}`);
|
|
2959
|
+
const { doc, bundle } = await loadDoc(input);
|
|
2960
|
+
const videos = findVideos(doc);
|
|
2961
|
+
if (!videos.length) throw new Error("document has no video element");
|
|
2962
|
+
let target = videos[0];
|
|
2963
|
+
if (opts.element != null) {
|
|
2964
|
+
const byId = videos.find((v) => v.el.id === opts.element);
|
|
2965
|
+
const byIdx = /^\d+$/.test(opts.element) ? videos[Number(opts.element)] : void 0;
|
|
2966
|
+
target = byId ?? byIdx ?? (() => {
|
|
2967
|
+
throw new Error(`no video element "${opts.element}" (have: ${videos.map((v) => v.el.id ?? `#${v.index}`).join(", ")})`);
|
|
2968
|
+
})();
|
|
2969
|
+
} else if (videos.length > 1) {
|
|
2970
|
+
throw new Error(`document has ${videos.length} videos \u2014 pick one with --element <id|index>`);
|
|
2971
|
+
}
|
|
2972
|
+
let segments;
|
|
2973
|
+
let source;
|
|
2974
|
+
if (opts.from) {
|
|
2975
|
+
segments = parseSubtitles(fs7.readFileSync(path7.resolve(opts.from), "utf-8"), opts.from);
|
|
2976
|
+
source = `${path7.extname(opts.from).slice(1).toLowerCase() || "file"}-import`;
|
|
2977
|
+
} else {
|
|
2978
|
+
const provider = opts.provider ?? "whisper-cli";
|
|
2979
|
+
const clip = await clipToTempFile(doc, target.el, path7.dirname(input));
|
|
2980
|
+
if (!clip) throw new Error("could not locate the clip bytes (no bundled asset, data URL, local path or http URL) \u2014 use --from to import subtitles instead");
|
|
2981
|
+
segments = provider === "openai" ? await openaiTranscribe(clip, opts.model, opts.language, opts.prompt) : whisperCli(clip, opts.model, opts.language, opts.prompt);
|
|
2982
|
+
source = provider === "openai" ? `openai:${opts.model || "whisper-1"}` : `whisper-cli${opts.model ? ":" + path7.basename(opts.model) : ""}`;
|
|
2983
|
+
}
|
|
2984
|
+
segments.sort((a, b) => a.t0 - b.t0);
|
|
2985
|
+
const transcript = { ...opts.language ? { language: opts.language } : {}, source, created: (/* @__PURE__ */ new Date()).toISOString(), segments };
|
|
2986
|
+
target.el.transcript = transcript;
|
|
2987
|
+
if (opts.chapters) target.el.chapters = parseChapters(fs7.readFileSync(path7.resolve(opts.chapters), "utf-8"));
|
|
2988
|
+
if (!target.el.id) target.el.id = `video-${target.index + 1}`;
|
|
2989
|
+
const output = opts.output ? path7.resolve(opts.output) : input;
|
|
2990
|
+
if (output.toLowerCase().endsWith(".jdfx") || bundle && !opts.output) {
|
|
2991
|
+
const { bytes } = await packJdfx(doc);
|
|
2992
|
+
fs7.writeFileSync(output, bytes);
|
|
2993
|
+
} else {
|
|
2994
|
+
fs7.writeFileSync(output, JSON.stringify(doc, null, 2));
|
|
2995
|
+
}
|
|
2996
|
+
const dur = segments.length ? segments[segments.length - 1].t1 : 0;
|
|
2997
|
+
console.log(`Transcribed: ${path7.basename(input)} \u2192 element "${target.el.id}" (${segments.length} segments, ${Math.round(dur)} s, source ${source})`);
|
|
2998
|
+
if (target.el.chapters) console.log(`Chapters: ${target.el.chapters.length}`);
|
|
2999
|
+
console.log(`Output: ${output}
|
|
3000
|
+
Next: jdf chunk ${path7.basename(output)} # transcript \u2192 time-windowed chunks with media.t0/t1`);
|
|
3001
|
+
return transcript;
|
|
3002
|
+
}
|
|
3003
|
+
var CONFIG_NAME = "jdf.rag.json";
|
|
3004
|
+
var OUT_DIR = ".jdf-rag";
|
|
3005
|
+
function walk(dir, acc = []) {
|
|
3006
|
+
for (const ent of fs7.readdirSync(dir, { withFileTypes: true })) {
|
|
3007
|
+
if (ent.name === "node_modules" || ent.name === OUT_DIR || ent.name.startsWith(".")) continue;
|
|
3008
|
+
const p = path7.join(dir, ent.name);
|
|
3009
|
+
if (ent.isDirectory()) walk(p, acc);
|
|
3010
|
+
else if (/\.(jdf|jdfx)$/i.test(ent.name)) acc.push(p);
|
|
3011
|
+
}
|
|
3012
|
+
return acc.sort();
|
|
3013
|
+
}
|
|
3014
|
+
async function readDoc(file) {
|
|
3015
|
+
if (file.toLowerCase().endsWith(".jdfx")) {
|
|
3016
|
+
const zip = await JSZip.loadAsync(fs7.readFileSync(file));
|
|
3017
|
+
return JSON.parse(await zip.file(JDFX_DOCUMENT_PATH).async("string"));
|
|
3018
|
+
}
|
|
3019
|
+
return JSON.parse(fs7.readFileSync(file, "utf-8"));
|
|
3020
|
+
}
|
|
3021
|
+
function videosIn(doc) {
|
|
3022
|
+
const out = [];
|
|
3023
|
+
const w = (els) => {
|
|
3024
|
+
for (const el of els ?? []) {
|
|
3025
|
+
if (el?.type === "video") out.push({ id: el.id, hasTranscript: !!el.transcript?.segments?.length });
|
|
3026
|
+
if (el?.elements) w(el.elements);
|
|
3027
|
+
}
|
|
3028
|
+
};
|
|
3029
|
+
for (const p of doc.pages ?? []) w(p.elements);
|
|
3030
|
+
return out;
|
|
3031
|
+
}
|
|
3032
|
+
async function ragFolder(dirPath, cli = {}) {
|
|
3033
|
+
const dir = path7.resolve(dirPath);
|
|
3034
|
+
if (!fs7.existsSync(dir) || !fs7.statSync(dir).isDirectory()) throw new Error(`Not a directory: ${dir}`);
|
|
3035
|
+
const cfgPath = path7.join(dir, CONFIG_NAME);
|
|
3036
|
+
const cfg = fs7.existsSync(cfgPath) ? JSON.parse(fs7.readFileSync(cfgPath, "utf-8")) : {};
|
|
3037
|
+
const opts = { ...cfg, ...Object.fromEntries(Object.entries(cli).filter(([, v]) => v !== void 0)) };
|
|
3038
|
+
const provider = opts.provider ?? "ollama";
|
|
3039
|
+
const transcribe = opts.transcribe ?? "none";
|
|
3040
|
+
const outDir = path7.resolve(opts.out ?? path7.join(dir, OUT_DIR));
|
|
3041
|
+
const files = walk(dir);
|
|
3042
|
+
console.log(`jdf rag: ${dir}
|
|
3043
|
+
files: ${files.length} (.jdf/.jdfx)${fs7.existsSync(cfgPath) ? `
|
|
3044
|
+
config: ${CONFIG_NAME}` : ""}
|
|
3045
|
+
embeddings: ${opts.noEmbed ? "skipped (--no-embed)" : `${provider}${opts.model ? " / " + opts.model : ""}`}
|
|
3046
|
+
transcribe: ${transcribe}${opts.dryRun ? "\n DRY RUN \u2014 nothing written" : ""}
|
|
3047
|
+
`);
|
|
3048
|
+
if (!files.length) {
|
|
3049
|
+
console.log("Nothing to do.");
|
|
3050
|
+
return;
|
|
3051
|
+
}
|
|
3052
|
+
const manifest = { dir, created: (/* @__PURE__ */ new Date()).toISOString(), provider: opts.noEmbed ? null : provider, model: opts.model ?? null, strategy: opts.strategy ?? "section", transcribe, files: [], totals: { files: files.length, chunks: 0, videoChunks: 0, videos: 0, transcribed: 0, untranscribed: 0 } };
|
|
3053
|
+
const indexLines = [];
|
|
3054
|
+
for (const file of files) {
|
|
3055
|
+
const rel = path7.relative(dir, file);
|
|
3056
|
+
const doc = await readDoc(file);
|
|
3057
|
+
const vids = videosIn(doc);
|
|
3058
|
+
let transcribedHere = 0;
|
|
3059
|
+
for (const v of vids) {
|
|
3060
|
+
if (v.hasTranscript) continue;
|
|
3061
|
+
if (transcribe === "none") {
|
|
3062
|
+
manifest.totals.untranscribed++;
|
|
3063
|
+
continue;
|
|
3064
|
+
}
|
|
3065
|
+
if (opts.dryRun) {
|
|
3066
|
+
transcribedHere++;
|
|
3067
|
+
continue;
|
|
3068
|
+
}
|
|
3069
|
+
try {
|
|
3070
|
+
await transcribeFile(file, { provider: transcribe, element: v.id, model: opts.transcribeModel, language: opts.language, prompt: opts.prompt });
|
|
3071
|
+
transcribedHere++;
|
|
3072
|
+
} catch (e) {
|
|
3073
|
+
console.warn(` ! ${rel}: transcription failed for video ${v.id ?? "#?"}: ${e.message}`);
|
|
3074
|
+
manifest.totals.untranscribed++;
|
|
3075
|
+
}
|
|
3076
|
+
}
|
|
3077
|
+
manifest.totals.videos += vids.length;
|
|
3078
|
+
manifest.totals.transcribed += transcribedHere;
|
|
3079
|
+
if (opts.dryRun) {
|
|
3080
|
+
manifest.files.push({ file: rel, videos: vids.length, wouldTranscribe: transcribedHere });
|
|
3081
|
+
continue;
|
|
3082
|
+
}
|
|
3083
|
+
const chunkOpts = { strategy: opts.strategy, maxTokens: opts.maxTokens, transcriptWindowSec: opts.transcriptWindowSec };
|
|
3084
|
+
const chunkOut = path7.join(outDir, "chunks", rel.replace(/\.(jdf|jdfx)$/i, ".chunks.jsonl"));
|
|
3085
|
+
fs7.mkdirSync(path7.dirname(chunkOut), { recursive: true });
|
|
3086
|
+
let chunks;
|
|
3087
|
+
if (opts.noEmbed) {
|
|
3088
|
+
chunks = await chunkFile(file, { ...chunkOpts, format: "jsonl", output: chunkOut });
|
|
3089
|
+
} else {
|
|
3090
|
+
const side = await embedFile(file, { ...chunkOpts, provider, model: opts.model, incremental: true });
|
|
3091
|
+
chunks = await chunkFile(file, { ...chunkOpts, format: "jsonl", output: chunkOut });
|
|
3092
|
+
manifest.files.push({ file: rel, chunks: chunks.length, vectors: Object.keys(side.vectors).length, sidecar: path7.relative(dir, file.replace(/\.(jdf|jdfx)$/i, ".embeddings.json")), videos: vids.length, transcribed: transcribedHere });
|
|
3093
|
+
}
|
|
3094
|
+
if (opts.noEmbed) manifest.files.push({ file: rel, chunks: chunks.length, videos: vids.length, transcribed: transcribedHere });
|
|
3095
|
+
for (const c of chunks) {
|
|
3096
|
+
indexLines.push(JSON.stringify({ file: rel, ...c }));
|
|
3097
|
+
manifest.totals.chunks++;
|
|
3098
|
+
if (c.media) manifest.totals.videoChunks++;
|
|
3099
|
+
}
|
|
3100
|
+
}
|
|
3101
|
+
if (!opts.dryRun) {
|
|
3102
|
+
fs7.mkdirSync(outDir, { recursive: true });
|
|
3103
|
+
fs7.writeFileSync(path7.join(outDir, "index.jsonl"), indexLines.join("\n") + (indexLines.length ? "\n" : ""));
|
|
3104
|
+
fs7.writeFileSync(path7.join(outDir, "manifest.json"), JSON.stringify(manifest, null, 2) + "\n");
|
|
3105
|
+
}
|
|
3106
|
+
const t = manifest.totals;
|
|
3107
|
+
console.log(`
|
|
3108
|
+
Done. ${t.files} files \u2192 ${t.chunks} chunks (${t.videoChunks} from video transcripts); videos ${t.videos}, transcribed now ${t.transcribed}, still without transcript ${t.untranscribed}${t.untranscribed && transcribe === "none" ? " (pass --transcribe whisper-cli|openai, or jdf transcribe --from subs.srt)" : ""}.`);
|
|
3109
|
+
if (!opts.dryRun) console.log(`Index: ${path7.join(outDir, "index.jsonl")}
|
|
3110
|
+
Report: ${path7.join(outDir, "manifest.json")}${opts.noEmbed ? "" : `
|
|
3111
|
+
Vectors: one <file>.embeddings.json next to each document (incremental \u2014 re-run any time)`}`);
|
|
3112
|
+
}
|
|
2387
3113
|
|
|
2388
3114
|
// src/index.ts
|
|
2389
3115
|
var HELP = `jdf \u2014 JSON Document Format CLI
|
|
@@ -2395,12 +3121,18 @@ The CLI exists for these workflows:
|
|
|
2395
3121
|
into a validated .jdf (or .jdfx) you can ship.
|
|
2396
3122
|
\u2022 JDF \u2192 chunks turn a document into retrieval-ready chunks (RAG).
|
|
2397
3123
|
\u2022 JDF \u2192 vectors embed those chunks, incrementally, for a vector store.
|
|
3124
|
+
\u2022 video \u2192 text attach a time-stamped transcript to a video element so
|
|
3125
|
+
RAG retrieves "video at 02:13", not just "a video".
|
|
3126
|
+
\u2022 folder \u2192 index one command over a directory of .jdf/.jdfx: transcribe,
|
|
3127
|
+
chunk, embed incrementally, write .jdf-rag/index.jsonl.
|
|
2398
3128
|
|
|
2399
3129
|
Usage:
|
|
2400
3130
|
jdf validate <file.jdf>
|
|
2401
3131
|
jdf convert <file.{pdf,json,md}> [-o output.{jdf,jdfx}] [--json] [--password PW] [--drop-invisible-text]
|
|
2402
3132
|
jdf chunk <file.{jdf,jdfx}> [--strategy section|element|fixed] [--format jsonl|json|inline] [--max-tokens N] [-o out]
|
|
2403
3133
|
jdf embed <file.{jdf,jdfx}> [--provider ollama|openai] [--model NAME] [--strategy \u2026] [--incremental] [-o out]
|
|
3134
|
+
jdf transcribe <file.{jdf,jdfx}> [--from subs.srt|.vtt|.json] [--provider whisper-cli|openai] [--element ID] [--chapters FILE] [-o out]
|
|
3135
|
+
jdf rag <dir> [--provider ollama|openai] [--model NAME] [--transcribe none|whisper-cli|openai] [--no-embed] [--dry-run] [--out DIR]
|
|
2404
3136
|
jdf --help
|
|
2405
3137
|
|
|
2406
3138
|
Commands:
|
|
@@ -2408,6 +3140,8 @@ Commands:
|
|
|
2408
3140
|
convert Convert a PDF, JSON, or Markdown file into JDF (alias: import)
|
|
2409
3141
|
chunk Split a JDF document into retrieval-ready chunks (offline, deterministic)
|
|
2410
3142
|
embed Compute embeddings for the chunks (local via Ollama by default)
|
|
3143
|
+
transcribe Store time-stamped text on a video element (import SRT/VTT/JSON, or run Whisper)
|
|
3144
|
+
rag Make a whole folder retrieval-ready (finds .jdf/.jdfx, transcribes, chunks, embeds, indexes)
|
|
2411
3145
|
|
|
2412
3146
|
Flags:
|
|
2413
3147
|
-o, --output <path> Explicit output path
|
|
@@ -2424,7 +3158,20 @@ Flags:
|
|
|
2424
3158
|
--provider <p> embed: ollama (default, local) | openai (remote API)
|
|
2425
3159
|
--model <name> embed: model id (default: nomic-embed-text / text-embedding-3-small)
|
|
2426
3160
|
--incremental embed: skip chunks whose content hash is unchanged
|
|
3161
|
+
--cache <path> embed: sidecar to reuse vectors from (default: the
|
|
3162
|
+
output path itself)
|
|
2427
3163
|
--no-auto-start embed(ollama): don't auto-launch Ollama via Docker
|
|
3164
|
+
--from <file> transcribe: import subtitles (.srt / .vtt / JSON segments) \u2014 offline, no model
|
|
3165
|
+
--element <id|n> transcribe: which video element (id, or 0-based index); default the only one
|
|
3166
|
+
--chapters <file> transcribe: JSON [{t,title}] or "mm:ss Title" lines \u2192 chapter breadcrumbs
|
|
3167
|
+
--language <tag> transcribe: BCP-47 language hint for Whisper
|
|
3168
|
+
--prompt <text> transcribe: Whisper vocabulary hint (names, acronyms) \u2014 not an instruction
|
|
3169
|
+
--window <sec> chunk/embed/rag: transcript window per video chunk (default 45)
|
|
3170
|
+
--transcribe <p> rag: none (default) | whisper-cli | openai \u2014 for videos that have no transcript yet
|
|
3171
|
+
--no-embed rag: chunk + index only
|
|
3172
|
+
--dry-run rag: list what would happen, write nothing
|
|
3173
|
+
--out <dir> rag: index folder (default <dir>/.jdf-rag)
|
|
3174
|
+
rag reads defaults from <dir>/jdf.rag.json (same keys as the flags; flags win)
|
|
2428
3175
|
|
|
2429
3176
|
Environment (embed):
|
|
2430
3177
|
ollama: OLLAMA_HOST (default http://localhost:11434)
|
|
@@ -2438,8 +3185,11 @@ Examples:
|
|
|
2438
3185
|
jdf chunk report.jdf --format inline # embed the chunk index into the .jdf
|
|
2439
3186
|
jdf embed report.jdf # local embeddings via Ollama (auto-setup)
|
|
2440
3187
|
jdf embed report.jdf --provider openai --incremental
|
|
3188
|
+
jdf transcribe talk.jdfx --from talk.srt --chapters chapters.txt # then: jdf chunk talk.jdfx
|
|
3189
|
+
jdf transcribe talk.jdfx --provider openai --language en --prompt "JDF, jdfx, Ollama"
|
|
3190
|
+
jdf rag ./knowledge-base --transcribe openai # whole folder \u2192 .jdf-rag/index.jsonl
|
|
2441
3191
|
`;
|
|
2442
|
-
var BOOLEAN_FLAGS = /* @__PURE__ */ new Set(["help", "h", "json", "verbose", "skip-validate", "incremental", "no-auto-start", "drop-invisible-text"]);
|
|
3192
|
+
var BOOLEAN_FLAGS = /* @__PURE__ */ new Set(["help", "h", "json", "verbose", "skip-validate", "incremental", "no-auto-start", "drop-invisible-text", "no-embed", "dry-run"]);
|
|
2443
3193
|
function parseArgs(argv) {
|
|
2444
3194
|
const positional = [];
|
|
2445
3195
|
const flags = {};
|
|
@@ -2543,14 +3293,55 @@ async function main() {
|
|
|
2543
3293
|
strategy: typeof flags.strategy === "string" ? flags.strategy : void 0,
|
|
2544
3294
|
format: typeof flags.format === "string" ? flags.format : void 0,
|
|
2545
3295
|
maxTokens: typeof flags["max-tokens"] === "string" ? parseInt(flags["max-tokens"], 10) : void 0,
|
|
3296
|
+
transcriptWindowSec: typeof flags.window === "string" ? parseInt(flags.window, 10) : void 0,
|
|
2546
3297
|
output: typeof flags.output === "string" ? flags.output : void 0
|
|
2547
3298
|
});
|
|
2548
3299
|
process.exit(0);
|
|
2549
3300
|
}
|
|
3301
|
+
case "transcribe": {
|
|
3302
|
+
const input = positional[0];
|
|
3303
|
+
if (!input) {
|
|
3304
|
+
console.error("Usage: jdf transcribe <file.{jdf,jdfx}> [--from subs.srt|.vtt|.json] [--provider whisper-cli|openai] [--model M] [--language tag] [--element id|n] [--chapters file] [-o out]");
|
|
3305
|
+
process.exit(1);
|
|
3306
|
+
}
|
|
3307
|
+
await transcribeFile(input, {
|
|
3308
|
+
from: typeof flags.from === "string" ? flags.from : void 0,
|
|
3309
|
+
provider: typeof flags.provider === "string" ? flags.provider : void 0,
|
|
3310
|
+
model: typeof flags.model === "string" ? flags.model : void 0,
|
|
3311
|
+
language: typeof flags.language === "string" ? flags.language : void 0,
|
|
3312
|
+
prompt: typeof flags.prompt === "string" ? flags.prompt : void 0,
|
|
3313
|
+
element: typeof flags.element === "string" ? flags.element : void 0,
|
|
3314
|
+
chapters: typeof flags.chapters === "string" ? flags.chapters : void 0,
|
|
3315
|
+
output: typeof flags.output === "string" ? flags.output : void 0
|
|
3316
|
+
});
|
|
3317
|
+
process.exit(0);
|
|
3318
|
+
}
|
|
3319
|
+
case "rag": {
|
|
3320
|
+
const input = positional[0];
|
|
3321
|
+
if (!input) {
|
|
3322
|
+
console.error("Usage: jdf rag <dir> [--provider ollama|openai] [--model NAME] [--strategy \u2026] [--window sec] [--transcribe none|whisper-cli|openai] [--language tag] [--prompt text] [--no-embed] [--dry-run] [--out DIR]");
|
|
3323
|
+
process.exit(1);
|
|
3324
|
+
}
|
|
3325
|
+
await ragFolder(input, {
|
|
3326
|
+
provider: typeof flags.provider === "string" ? flags.provider : void 0,
|
|
3327
|
+
model: typeof flags.model === "string" ? flags.model : void 0,
|
|
3328
|
+
strategy: typeof flags.strategy === "string" ? flags.strategy : void 0,
|
|
3329
|
+
maxTokens: typeof flags["max-tokens"] === "string" ? parseInt(flags["max-tokens"], 10) : void 0,
|
|
3330
|
+
transcriptWindowSec: typeof flags.window === "string" ? parseInt(flags.window, 10) : void 0,
|
|
3331
|
+
transcribe: typeof flags.transcribe === "string" ? flags.transcribe : void 0,
|
|
3332
|
+
transcribeModel: typeof flags["transcribe-model"] === "string" ? flags["transcribe-model"] : void 0,
|
|
3333
|
+
language: typeof flags.language === "string" ? flags.language : void 0,
|
|
3334
|
+
prompt: typeof flags.prompt === "string" ? flags.prompt : void 0,
|
|
3335
|
+
noEmbed: flags["no-embed"] === true,
|
|
3336
|
+
dryRun: flags["dry-run"] === true,
|
|
3337
|
+
out: typeof flags.out === "string" ? flags.out : void 0
|
|
3338
|
+
});
|
|
3339
|
+
process.exit(0);
|
|
3340
|
+
}
|
|
2550
3341
|
case "embed": {
|
|
2551
3342
|
const input = positional[0];
|
|
2552
3343
|
if (!input) {
|
|
2553
|
-
console.error("Usage: jdf embed <file.{jdf,jdfx}> [--provider ollama|openai] [--model NAME] [--strategy \u2026] [--incremental] [--no-auto-start] [-o out]");
|
|
3344
|
+
console.error("Usage: jdf embed <file.{jdf,jdfx}> [--provider ollama|openai] [--model NAME] [--strategy \u2026] [--incremental] [--cache prev.embeddings.json] [--no-auto-start] [-o out]");
|
|
2554
3345
|
process.exit(1);
|
|
2555
3346
|
}
|
|
2556
3347
|
await embedFile(input, {
|
|
@@ -2559,8 +3350,10 @@ async function main() {
|
|
|
2559
3350
|
strategy: typeof flags.strategy === "string" ? flags.strategy : void 0,
|
|
2560
3351
|
maxTokens: typeof flags["max-tokens"] === "string" ? parseInt(flags["max-tokens"], 10) : void 0,
|
|
2561
3352
|
incremental: flags.incremental === true,
|
|
3353
|
+
transcriptWindowSec: typeof flags.window === "string" ? parseInt(flags.window, 10) : void 0,
|
|
2562
3354
|
autoStart: flags["no-auto-start"] !== true,
|
|
2563
|
-
output: typeof flags.output === "string" ? flags.output : void 0
|
|
3355
|
+
output: typeof flags.output === "string" ? flags.output : void 0,
|
|
3356
|
+
cache: typeof flags.cache === "string" ? flags.cache : void 0
|
|
2564
3357
|
});
|
|
2565
3358
|
process.exit(0);
|
|
2566
3359
|
}
|