@uurtech/jdf-cli 0.1.26 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.js +858 -156
- package/dist/jdf-schema.json +33 -2
- package/package.json +1 -1
package/dist/index.js
CHANGED
|
@@ -1,13 +1,20 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
|
-
import
|
|
3
|
-
import
|
|
2
|
+
import fs7 from 'fs';
|
|
3
|
+
import path7 from 'path';
|
|
4
4
|
import { fileURLToPath } from 'url';
|
|
5
5
|
import Ajv from 'ajv';
|
|
6
6
|
import addFormats from 'ajv-formats';
|
|
7
7
|
import JSZip from 'jszip';
|
|
8
8
|
import crypto, { createHash } from 'crypto';
|
|
9
9
|
import { readFile } from 'fs/promises';
|
|
10
|
-
import { execFileSync } from 'child_process';
|
|
10
|
+
import { execFileSync, spawnSync } from 'child_process';
|
|
11
|
+
|
|
12
|
+
var __require = /* @__PURE__ */ ((x) => typeof require !== "undefined" ? require : typeof Proxy !== "undefined" ? new Proxy(x, {
|
|
13
|
+
get: (a, b) => (typeof require !== "undefined" ? require : a)[b]
|
|
14
|
+
}) : x)(function(x) {
|
|
15
|
+
if (typeof require !== "undefined") return require.apply(this, arguments);
|
|
16
|
+
throw Error('Dynamic require of "' + x + '" is not supported');
|
|
17
|
+
});
|
|
11
18
|
|
|
12
19
|
// ../../packages/jdf-core/src/manifest.ts
|
|
13
20
|
var JDFX_MANIFEST_VERSION = "1.0.0";
|
|
@@ -16,17 +23,17 @@ var JDFX_MANIFEST_PATH = "manifest.json";
|
|
|
16
23
|
var JDFX_ASSET_DIR = "assets";
|
|
17
24
|
|
|
18
25
|
// src/commands/validate.ts
|
|
19
|
-
var __dirname$1 =
|
|
26
|
+
var __dirname$1 = path7.dirname(fileURLToPath(import.meta.url));
|
|
20
27
|
function resolveSchemaPath() {
|
|
21
|
-
const bundled =
|
|
22
|
-
if (
|
|
23
|
-
const dev =
|
|
28
|
+
const bundled = path7.resolve(__dirname$1, "jdf-schema.json");
|
|
29
|
+
if (fs7.existsSync(bundled)) return bundled;
|
|
30
|
+
const dev = path7.resolve(__dirname$1, "../../../../spec/jdf-schema.json");
|
|
24
31
|
return dev;
|
|
25
32
|
}
|
|
26
33
|
var SCHEMA_PATH = resolveSchemaPath();
|
|
27
34
|
async function loadDocument(filePath) {
|
|
28
35
|
if (filePath.toLowerCase().endsWith(".jdfx")) {
|
|
29
|
-
const zip = await JSZip.loadAsync(
|
|
36
|
+
const zip = await JSZip.loadAsync(fs7.readFileSync(filePath));
|
|
30
37
|
const docFile = zip.file(JDFX_DOCUMENT_PATH);
|
|
31
38
|
if (!docFile) {
|
|
32
39
|
console.error(`\u2717 Bundle missing ${JDFX_DOCUMENT_PATH}`);
|
|
@@ -51,11 +58,11 @@ async function loadDocument(filePath) {
|
|
|
51
58
|
}
|
|
52
59
|
return { doc, bundle: { manifest, assetCount } };
|
|
53
60
|
}
|
|
54
|
-
return { doc: JSON.parse(
|
|
61
|
+
return { doc: JSON.parse(fs7.readFileSync(filePath, "utf-8")) };
|
|
55
62
|
}
|
|
56
63
|
async function validate(file) {
|
|
57
|
-
const filePath =
|
|
58
|
-
if (!
|
|
64
|
+
const filePath = path7.resolve(file);
|
|
65
|
+
if (!fs7.existsSync(filePath)) {
|
|
59
66
|
console.error(`File not found: ${filePath}`);
|
|
60
67
|
return false;
|
|
61
68
|
}
|
|
@@ -68,11 +75,11 @@ async function validate(file) {
|
|
|
68
75
|
}
|
|
69
76
|
if (!loaded) return false;
|
|
70
77
|
const { doc, bundle } = loaded;
|
|
71
|
-
if (!
|
|
78
|
+
if (!fs7.existsSync(SCHEMA_PATH)) {
|
|
72
79
|
console.error(`Schema not found at ${SCHEMA_PATH}`);
|
|
73
80
|
return false;
|
|
74
81
|
}
|
|
75
|
-
const schema = JSON.parse(
|
|
82
|
+
const schema = JSON.parse(fs7.readFileSync(SCHEMA_PATH, "utf-8"));
|
|
76
83
|
const ajv = new Ajv({ allErrors: true, strict: false });
|
|
77
84
|
addFormats(ajv);
|
|
78
85
|
const validateFn = ajv.compile(schema);
|
|
@@ -81,7 +88,7 @@ async function validate(file) {
|
|
|
81
88
|
const d = doc;
|
|
82
89
|
const pageCount = Array.isArray(d.pages) ? d.pages.length : 0;
|
|
83
90
|
const elCount = Array.isArray(d.pages) ? d.pages.reduce((acc, p) => acc + (Array.isArray(p?.elements) ? p.elements.length : 0), 0) : 0;
|
|
84
|
-
console.log(`\u2713 Valid: ${
|
|
91
|
+
console.log(`\u2713 Valid: ${path7.basename(filePath)}`);
|
|
85
92
|
console.log(` Format: ${d.$jdf}${bundle ? " (jdfx bundle)" : ""}`);
|
|
86
93
|
console.log(` Title: ${d.meta?.title}`);
|
|
87
94
|
console.log(` Pages: ${pageCount}`);
|
|
@@ -94,7 +101,7 @@ async function validate(file) {
|
|
|
94
101
|
}
|
|
95
102
|
return true;
|
|
96
103
|
}
|
|
97
|
-
console.error(`\u2717 Invalid: ${
|
|
104
|
+
console.error(`\u2717 Invalid: ${path7.basename(filePath)}`);
|
|
98
105
|
for (const err of validateFn.errors || []) {
|
|
99
106
|
const loc = err.instancePath || "(root)";
|
|
100
107
|
console.error(` ${loc} \u2014 ${err.message}`);
|
|
@@ -109,15 +116,15 @@ function hashBytes(bytes) {
|
|
|
109
116
|
return createHash("sha1").update(bytes).digest("hex").slice(0, 16);
|
|
110
117
|
}
|
|
111
118
|
function rewriteResourceRefs(doc, oldKey, newKey) {
|
|
112
|
-
function
|
|
119
|
+
function walk2(els) {
|
|
113
120
|
if (!els) return;
|
|
114
121
|
for (const el of els) {
|
|
115
122
|
if (el?.resource === oldKey) el.resource = newKey;
|
|
116
|
-
if (el?.elements)
|
|
117
|
-
if (el?.children)
|
|
123
|
+
if (el?.elements) walk2(el.elements);
|
|
124
|
+
if (el?.children) walk2(el.children);
|
|
118
125
|
}
|
|
119
126
|
}
|
|
120
|
-
for (const page of doc.pages || [])
|
|
127
|
+
for (const page of doc.pages || []) walk2(page.elements);
|
|
121
128
|
}
|
|
122
129
|
function extractAssets(doc) {
|
|
123
130
|
const assets = [];
|
|
@@ -149,10 +156,10 @@ function extractAssets(doc) {
|
|
|
149
156
|
assets.push({ id, bytes, mimeType, ext });
|
|
150
157
|
return id;
|
|
151
158
|
}
|
|
152
|
-
function
|
|
159
|
+
function walk2(els) {
|
|
153
160
|
if (!els) return;
|
|
154
161
|
for (const el of els) {
|
|
155
|
-
if (el?.type === "image" && typeof el.src === "string" && el.src.startsWith("data:")) {
|
|
162
|
+
if ((el?.type === "image" || el?.type === "video") && typeof el.src === "string" && el.src.startsWith("data:")) {
|
|
156
163
|
const m = el.src.match(/^data:([^;]+);base64,(.*)$/);
|
|
157
164
|
if (m) {
|
|
158
165
|
const mimeType = m[1];
|
|
@@ -162,19 +169,21 @@ function extractAssets(doc) {
|
|
|
162
169
|
el.resource = id;
|
|
163
170
|
}
|
|
164
171
|
}
|
|
165
|
-
if (el?.elements)
|
|
166
|
-
if (el?.children)
|
|
172
|
+
if (el?.elements) walk2(el.elements);
|
|
173
|
+
if (el?.children) walk2(el.children);
|
|
167
174
|
}
|
|
168
175
|
}
|
|
169
|
-
for (const page of cloned.pages || [])
|
|
170
|
-
|
|
171
|
-
|
|
176
|
+
for (const page of cloned.pages || []) walk2(page.elements);
|
|
177
|
+
for (const bucket of ["images", "videos"]) {
|
|
178
|
+
const store = cloned.resources?.[bucket];
|
|
179
|
+
if (!store) continue;
|
|
180
|
+
for (const [key, res] of Object.entries(store)) {
|
|
172
181
|
if (!res || typeof res !== "object" || !("data" in res) || !res.data) continue;
|
|
173
182
|
const data = String(res.data);
|
|
174
183
|
const m = data.match(/^data:([^;]+);base64,(.*)$/);
|
|
175
184
|
const b64 = m ? m[2] : data;
|
|
176
|
-
const mimeType = m ? m[1] : res.mimeType || "image/png";
|
|
177
|
-
const ext = mimeType.split("/")[1]?.replace("jpeg", "jpg") || "bin";
|
|
185
|
+
const mimeType = m ? m[1] : res.mimeType || (bucket === "videos" ? "video/mp4" : "image/png");
|
|
186
|
+
const ext = mimeType.split("/")[1]?.replace("jpeg", "jpg").replace("quicktime", "mov") || "bin";
|
|
178
187
|
const bytes = decodeBase64(b64);
|
|
179
188
|
const h = hashBytes(bytes);
|
|
180
189
|
let canonicalId = hashToId.get(h);
|
|
@@ -188,10 +197,10 @@ function extractAssets(doc) {
|
|
|
188
197
|
delete updated.data;
|
|
189
198
|
updated.src = "embedded";
|
|
190
199
|
if (canonicalId !== key) {
|
|
191
|
-
delete
|
|
200
|
+
delete store[key];
|
|
192
201
|
rewriteResourceRefs(cloned, key, canonicalId);
|
|
193
202
|
} else {
|
|
194
|
-
|
|
203
|
+
store[key] = updated;
|
|
195
204
|
}
|
|
196
205
|
}
|
|
197
206
|
}
|
|
@@ -221,21 +230,22 @@ async function packJdfx(doc) {
|
|
|
221
230
|
return { bytes, manifest };
|
|
222
231
|
}
|
|
223
232
|
function shouldUseJdfx(doc) {
|
|
224
|
-
const
|
|
225
|
-
|
|
226
|
-
|
|
233
|
+
for (const store of [doc.resources?.images ?? {}, doc.resources?.videos ?? {}]) {
|
|
234
|
+
for (const v of Object.values(store)) {
|
|
235
|
+
if (v && typeof v === "object" && "data" in v && v.data) return true;
|
|
236
|
+
}
|
|
227
237
|
}
|
|
228
|
-
function
|
|
238
|
+
function walk2(els) {
|
|
229
239
|
if (!els) return false;
|
|
230
240
|
for (const el of els) {
|
|
231
|
-
if (el?.type === "image" && typeof el.src === "string" && el.src.startsWith("data:")) return true;
|
|
232
|
-
if (el?.elements &&
|
|
233
|
-
if (el?.children &&
|
|
241
|
+
if ((el?.type === "image" || el?.type === "video") && typeof el.src === "string" && el.src.startsWith("data:")) return true;
|
|
242
|
+
if (el?.elements && walk2(el.elements)) return true;
|
|
243
|
+
if (el?.children && walk2(el.children)) return true;
|
|
234
244
|
}
|
|
235
245
|
return false;
|
|
236
246
|
}
|
|
237
247
|
for (const page of doc.pages || []) {
|
|
238
|
-
if (
|
|
248
|
+
if (walk2(page.elements)) return true;
|
|
239
249
|
}
|
|
240
250
|
return false;
|
|
241
251
|
}
|
|
@@ -294,13 +304,13 @@ function stripInline(text) {
|
|
|
294
304
|
return parseInline(text).map((r) => r.text).join("");
|
|
295
305
|
}
|
|
296
306
|
async function importMarkdown(inputPath, outputPath) {
|
|
297
|
-
const input =
|
|
307
|
+
const input = path7.resolve(inputPath);
|
|
298
308
|
console.log(`Importing: ${input}`);
|
|
299
|
-
const content =
|
|
300
|
-
const doc = convertMarkdownToJdf(content,
|
|
309
|
+
const content = fs7.readFileSync(input, "utf-8");
|
|
310
|
+
const doc = convertMarkdownToJdf(content, path7.basename(input, path7.extname(input)), path7.dirname(input));
|
|
301
311
|
let output;
|
|
302
312
|
if (outputPath) {
|
|
303
|
-
output =
|
|
313
|
+
output = path7.resolve(outputPath);
|
|
304
314
|
} else {
|
|
305
315
|
const stem = input.replace(/\.(md|markdown)$/i, "");
|
|
306
316
|
output = stem + (shouldUseJdfx(doc) ? ".jdfx" : ".jdf");
|
|
@@ -308,11 +318,11 @@ async function importMarkdown(inputPath, outputPath) {
|
|
|
308
318
|
console.log(`Output: ${output}`);
|
|
309
319
|
if (output.toLowerCase().endsWith(".jdfx")) {
|
|
310
320
|
const { bytes, manifest } = await packJdfx(doc);
|
|
311
|
-
|
|
321
|
+
fs7.writeFileSync(output, bytes);
|
|
312
322
|
console.log(`
|
|
313
323
|
Done! Created ${doc.pages.length} page(s), ${manifest.assets.length} asset(s) bundled`);
|
|
314
324
|
} else {
|
|
315
|
-
|
|
325
|
+
fs7.writeFileSync(output, JSON.stringify(doc, null, 2));
|
|
316
326
|
console.log(`
|
|
317
327
|
Done! Created ${doc.pages.length} page(s)`);
|
|
318
328
|
}
|
|
@@ -329,10 +339,10 @@ var MIME_BY_EXT2 = {
|
|
|
329
339
|
};
|
|
330
340
|
function resolveImageSrc(src, baseDir) {
|
|
331
341
|
if (/^(https?:|data:|file:)/i.test(src)) return src;
|
|
332
|
-
const abs =
|
|
342
|
+
const abs = path7.isAbsolute(src) ? src : path7.resolve(baseDir, src);
|
|
333
343
|
try {
|
|
334
|
-
const bytes =
|
|
335
|
-
const ext =
|
|
344
|
+
const bytes = fs7.readFileSync(abs);
|
|
345
|
+
const ext = path7.extname(abs).slice(1).toLowerCase();
|
|
336
346
|
const mime = MIME_BY_EXT2[ext] || "application/octet-stream";
|
|
337
347
|
return `data:${mime};base64,${bytes.toString("base64")}`;
|
|
338
348
|
} catch {
|
|
@@ -613,8 +623,232 @@ function convertMarkdownToJdf(md, title, baseDir = process.cwd()) {
|
|
|
613
623
|
};
|
|
614
624
|
}
|
|
615
625
|
|
|
616
|
-
// ../../packages/jdf-pdf-import/src/
|
|
626
|
+
// ../../packages/jdf-pdf-import/src/tables.ts
|
|
617
627
|
var PT_TO_MM = 0.352778;
|
|
628
|
+
function calibrateGlyphWidth(runs) {
|
|
629
|
+
const ks = [];
|
|
630
|
+
for (const r of runs) {
|
|
631
|
+
const t = r.text;
|
|
632
|
+
if (/\s$/.test(t) || t.trim().length < 3 || r.width <= 0) continue;
|
|
633
|
+
ks.push(r.width / (t.length * r.fontSize * PT_TO_MM));
|
|
634
|
+
}
|
|
635
|
+
if (ks.length < 3) return 0.55;
|
|
636
|
+
ks.sort((a, b) => a - b);
|
|
637
|
+
return Math.min(0.7, Math.max(0.4, ks[Math.floor(ks.length / 2)]));
|
|
638
|
+
}
|
|
639
|
+
function hasStretchedSpaces(runs, k) {
|
|
640
|
+
let n = 0, wide = 0;
|
|
641
|
+
for (const r of runs) {
|
|
642
|
+
if (!/\s$/.test(r.text) || r.text.trim().length === 0) continue;
|
|
643
|
+
const em = r.fontSize * PT_TO_MM;
|
|
644
|
+
const est = r.text.trim().length * em * k + em * 0.25;
|
|
645
|
+
n++;
|
|
646
|
+
if (r.width > est * 1.4) wide++;
|
|
647
|
+
}
|
|
648
|
+
return n >= 4 && wide / n >= 0.3;
|
|
649
|
+
}
|
|
650
|
+
function textExtent(r, k, stretched) {
|
|
651
|
+
const em = r.fontSize * PT_TO_MM;
|
|
652
|
+
if (!/\s$/.test(r.text)) return Math.max(em * 0.5, r.width);
|
|
653
|
+
const chars = Math.max(1, r.text.trim().length);
|
|
654
|
+
const est = chars * em * k + em * 0.25;
|
|
655
|
+
if (stretched) return Math.max(em * 0.5, Math.min(r.width, est));
|
|
656
|
+
return Math.max(em * 0.5, r.width > est * 1.4 ? est : r.width);
|
|
657
|
+
}
|
|
658
|
+
function groupRows(runs, skip) {
|
|
659
|
+
const k = calibrateGlyphWidth(runs);
|
|
660
|
+
const stretched = hasStretchedSpaces(runs, k);
|
|
661
|
+
const idx = runs.map((_, i) => i).filter((i) => !skip(runs[i]) && runs[i].text.trim().length > 0);
|
|
662
|
+
idx.sort((a, b) => runs[a].y - runs[b].y || runs[a].x - runs[b].x);
|
|
663
|
+
const rows = [];
|
|
664
|
+
for (const i of idx) {
|
|
665
|
+
const r = runs[i];
|
|
666
|
+
const tol = Math.max(0.8, r.fontSize * PT_TO_MM * 0.35);
|
|
667
|
+
const last = rows[rows.length - 1];
|
|
668
|
+
const cell = { run: r, idx: i, x0: r.x, x1: r.x + textExtent(r, k, stretched) };
|
|
669
|
+
if (last && Math.abs(last.y - r.y) <= tol) {
|
|
670
|
+
last.cells.push(cell);
|
|
671
|
+
last.h = Math.max(last.h, r.height);
|
|
672
|
+
} else {
|
|
673
|
+
rows.push({ y: r.y, h: r.height, cells: [cell] });
|
|
674
|
+
}
|
|
675
|
+
}
|
|
676
|
+
for (const row of rows) {
|
|
677
|
+
row.cells.sort((a, b) => a.x0 - b.x0);
|
|
678
|
+
const merged = [];
|
|
679
|
+
for (const c of row.cells) {
|
|
680
|
+
const last = merged[merged.length - 1];
|
|
681
|
+
const em = c.run.fontSize * PT_TO_MM;
|
|
682
|
+
if (last && c.x0 - last.x1 <= em * 1) {
|
|
683
|
+
last.x1 = Math.max(last.x1, c.x1);
|
|
684
|
+
last.run = { ...last.run, text: `${last.run.text.replace(/\s+$/, "")} ${c.run.text.replace(/^\s+/, "")}`, width: last.x1 - last.x0 };
|
|
685
|
+
last.extra = [...last.extra ?? [], c.idx];
|
|
686
|
+
} else merged.push({ ...c });
|
|
687
|
+
}
|
|
688
|
+
row.cells = merged;
|
|
689
|
+
}
|
|
690
|
+
return rows;
|
|
691
|
+
}
|
|
692
|
+
function columnBands(rows) {
|
|
693
|
+
const cells = rows.flatMap((r) => r.cells);
|
|
694
|
+
const sorted = cells.slice().sort((a, b) => a.x0 - b.x0);
|
|
695
|
+
const bands = [];
|
|
696
|
+
for (const c of sorted) {
|
|
697
|
+
const last = bands[bands.length - 1];
|
|
698
|
+
if (last && c.x0 <= last.x1 - 0.2) {
|
|
699
|
+
last.x1 = Math.max(last.x1, c.x1);
|
|
700
|
+
last.members.push(c);
|
|
701
|
+
} else bands.push({ x0: c.x0, x1: c.x1, members: [c] });
|
|
702
|
+
}
|
|
703
|
+
for (const b of bands) {
|
|
704
|
+
const rowsSeen = /* @__PURE__ */ new Set();
|
|
705
|
+
for (const m of b.members) {
|
|
706
|
+
const row = rows.find((r) => r.cells.includes(m));
|
|
707
|
+
if (rowsSeen.has(row)) return null;
|
|
708
|
+
rowsSeen.add(row);
|
|
709
|
+
}
|
|
710
|
+
}
|
|
711
|
+
return bands.map(({ x0, x1 }) => ({ x0, x1 }));
|
|
712
|
+
}
|
|
713
|
+
var numeric = (s) => /^[\s$€£¥+\-−–]*[\d.,]+\s*(%|ms|s|k|m|b|M|K|B|x|×)?\s*(\/\w+)?$/i.test(s.trim()) || /^[+\-−]?\d/.test(s.trim()) && /\d$/.test(s.trim().replace(/[%)]$/, ""));
|
|
714
|
+
function detectTables(runs, shapes, pageWidthMm) {
|
|
715
|
+
const out = [];
|
|
716
|
+
const used = /* @__PURE__ */ new Set();
|
|
717
|
+
const rows = groupRows(runs, () => false);
|
|
718
|
+
const pageW = pageWidthMm;
|
|
719
|
+
let i = 0;
|
|
720
|
+
while (i < rows.length) {
|
|
721
|
+
if (rows[i].cells.length < 2) {
|
|
722
|
+
i++;
|
|
723
|
+
continue;
|
|
724
|
+
}
|
|
725
|
+
let j = i;
|
|
726
|
+
let cur = columnBands([rows[i]]);
|
|
727
|
+
let best = null;
|
|
728
|
+
while (j + 1 < rows.length && cur) {
|
|
729
|
+
const next = rows[j + 1];
|
|
730
|
+
const gap = next.y - (rows[j].y + rows[j].h);
|
|
731
|
+
const rowH = Math.max(rows[j].h, next.h);
|
|
732
|
+
if (gap > rowH * 2.2) break;
|
|
733
|
+
if (next.cells.length === 1) {
|
|
734
|
+
const c = next.cells[0];
|
|
735
|
+
const inBand = cur.findIndex((b) => c.x0 < b.x1 - 0.2 && c.x1 > b.x0 + 0.2);
|
|
736
|
+
const spansSeveral = cur.filter((b) => c.x0 < b.x1 - 0.2 && c.x1 > b.x0 + 0.2).length > 1;
|
|
737
|
+
if (inBand <= 0 || spansSeveral) break;
|
|
738
|
+
j++;
|
|
739
|
+
continue;
|
|
740
|
+
}
|
|
741
|
+
const nb = columnBands(rows.slice(i, j + 2));
|
|
742
|
+
if (!nb || nb.length < 2) break;
|
|
743
|
+
if (nb.length > cur.length && j - i >= 2) break;
|
|
744
|
+
cur = nb;
|
|
745
|
+
j++;
|
|
746
|
+
if (cur.length >= 2) best = { j, bands: cur };
|
|
747
|
+
}
|
|
748
|
+
const multiRows = best ? rows.slice(i, best.j + 1).filter((r) => r.cells.length >= 2).length : 0;
|
|
749
|
+
const blockRows = best ? rows.slice(i, best.j + 1) : [];
|
|
750
|
+
const bbox = blockRows.length ? {
|
|
751
|
+
x0: Math.min(...best.bands.map((b) => b.x0)) - 5,
|
|
752
|
+
x1: Math.max(...best.bands.map((b) => b.x1)) + 5,
|
|
753
|
+
// Cell padding puts backgrounds/borders well above the first baseline and below the last.
|
|
754
|
+
y0: blockRows[0].y - blockRows[0].h * 2.5,
|
|
755
|
+
y1: blockRows[blockRows.length - 1].y + blockRows[blockRows.length - 1].h * 3
|
|
756
|
+
} : null;
|
|
757
|
+
const gridShapes = bbox ? shapes.map((s, k) => ({ s, k })).filter(({ s }) => s.x >= bbox.x0 - 1 && s.x + s.width <= bbox.x1 + 1 && s.y >= bbox.y0 - 1 && s.y + s.height <= bbox.y1 + 1 && (s.kind === "line" || s.kind === "rect")) : [];
|
|
758
|
+
const hasLattice = gridShapes.length >= 3;
|
|
759
|
+
if (!best || multiRows < (hasLattice ? 2 : 3) || best.bands.length < 2) {
|
|
760
|
+
i++;
|
|
761
|
+
continue;
|
|
762
|
+
}
|
|
763
|
+
const bands = best.bands;
|
|
764
|
+
const cellText = (row, b) => row.cells.filter((c) => c.x0 < bands[b].x1 - 0.2 && c.x1 > bands[b].x0 + 0.2).map((c) => c.run.text.trim()).join(" ").trim();
|
|
765
|
+
const grid = [];
|
|
766
|
+
const lineIdx = [];
|
|
767
|
+
for (const row of blockRows) {
|
|
768
|
+
if (row.cells.length === 1 && grid.length) {
|
|
769
|
+
const c = row.cells[0];
|
|
770
|
+
const b = bands.findIndex((bb) => c.x0 < bb.x1 - 0.2 && c.x1 > bb.x0 + 0.2);
|
|
771
|
+
if (b > 0) {
|
|
772
|
+
grid[grid.length - 1][b] = (grid[grid.length - 1][b] + " " + c.run.text.trim()).trim();
|
|
773
|
+
lineIdx.push(c.idx, ...c.extra ?? []);
|
|
774
|
+
continue;
|
|
775
|
+
}
|
|
776
|
+
}
|
|
777
|
+
grid.push(bands.map((_, b) => cellText(row, b)));
|
|
778
|
+
for (const c of row.cells) {
|
|
779
|
+
lineIdx.push(c.idx);
|
|
780
|
+
for (const k of c.extra ?? []) lineIdx.push(k);
|
|
781
|
+
}
|
|
782
|
+
}
|
|
783
|
+
if (lineIdx.some((k) => used.has(k))) {
|
|
784
|
+
i = best.j + 1;
|
|
785
|
+
continue;
|
|
786
|
+
}
|
|
787
|
+
const first = blockRows[0];
|
|
788
|
+
const tableW = bbox.x1 - bbox.x0;
|
|
789
|
+
const rowFill = (row) => {
|
|
790
|
+
const cy = row.y + row.h * 0.5;
|
|
791
|
+
const rects = gridShapes.filter(({ s }) => s.kind === "rect" && s.fill && s.fill.toLowerCase() !== "#ffffff" && s.height >= row.h * 0.6 && s.height < row.h * 4.5 && s.y <= cy && s.y + s.height >= cy);
|
|
792
|
+
const covered = rects.reduce((a, { s }) => a + s.width, 0);
|
|
793
|
+
if (!rects.length || covered < tableW * 0.5) return null;
|
|
794
|
+
const counts = /* @__PURE__ */ new Map();
|
|
795
|
+
for (const { s } of rects) counts.set(s.fill, (counts.get(s.fill) ?? 0) + s.width);
|
|
796
|
+
const fill = [...counts.entries()].sort((a, b) => b[1] - a[1])[0][0];
|
|
797
|
+
return { fill, rects };
|
|
798
|
+
};
|
|
799
|
+
const headerBg = rowFill(first);
|
|
800
|
+
const firstBold = first.cells.every((c) => c.run.bold);
|
|
801
|
+
const bodyFills = blockRows.slice(1).map(rowFill);
|
|
802
|
+
const headerDistinct = !!headerBg && !bodyFills.every((f) => f?.fill === headerBg.fill);
|
|
803
|
+
const isHeader = headerDistinct || firstBold && !blockRows.slice(1).every((r) => r.cells.every((c) => c.run.bold));
|
|
804
|
+
const columns = bands.map((b, k) => {
|
|
805
|
+
const vals = grid.slice(isHeader ? 1 : 0).map((r) => r[k]).filter(Boolean);
|
|
806
|
+
const numericShare = vals.length ? vals.filter(numeric).length / vals.length : 0;
|
|
807
|
+
const col = { width: Math.round((b.x1 - b.x0) * 10) / 10 };
|
|
808
|
+
if (numericShare >= 0.7) col.align = "right";
|
|
809
|
+
return col;
|
|
810
|
+
});
|
|
811
|
+
const x0 = Math.max(0, bands[0].x0 - 2.5);
|
|
812
|
+
const x1 = Math.min(pageW, bands[bands.length - 1].x1 + 2.5);
|
|
813
|
+
for (let k = 0; k < bands.length; k++) {
|
|
814
|
+
const left = k === 0 ? x0 : (bands[k - 1].x1 + bands[k].x0) / 2;
|
|
815
|
+
const right = k === bands.length - 1 ? x1 : (bands[k].x1 + bands[k + 1].x0) / 2;
|
|
816
|
+
columns[k].width = Math.round((right - left) * 10) / 10;
|
|
817
|
+
}
|
|
818
|
+
const rowFills = (isHeader ? bodyFills : [headerBg, ...bodyFills]).map((f) => f?.fill ?? null);
|
|
819
|
+
const odd = rowFills.filter((_, k) => k % 2 === 1), even = rowFills.filter((_, k) => k % 2 === 0);
|
|
820
|
+
const altColor = odd.length && odd[0] && odd.every((f) => f === odd[0]) && even.every((f) => f !== odd[0]) ? odd[0] : void 0;
|
|
821
|
+
const borderShape = gridShapes.find(({ s }) => s.kind === "line" || s.kind === "rect" && (s.height < 0.6 || s.width < 0.6) && (s.fill || s.stroke));
|
|
822
|
+
const borderColor = borderShape ? borderShape.s.stroke || borderShape.s.fill : void 0;
|
|
823
|
+
const fontSize = Math.round(first.cells[0].run.fontSize * 10) / 10;
|
|
824
|
+
const y0 = headerBg ? Math.min(...headerBg.rects.map(({ s }) => s.y)) : first.y - first.h * 0.5;
|
|
825
|
+
const element = {
|
|
826
|
+
type: "table",
|
|
827
|
+
position: { x: Math.round(x0 * 100) / 100, y: Math.round(Math.max(0, y0) * 100) / 100 },
|
|
828
|
+
width: Math.round((x1 - x0) * 100) / 100,
|
|
829
|
+
columns,
|
|
830
|
+
rows: isHeader ? grid.slice(1) : grid,
|
|
831
|
+
style: { fontSize }
|
|
832
|
+
};
|
|
833
|
+
if (isHeader) {
|
|
834
|
+
element.headers = grid[0];
|
|
835
|
+
const hs = { fontWeight: "bold" };
|
|
836
|
+
if (headerBg) hs.backgroundColor = headerBg.fill;
|
|
837
|
+
const hc = first.cells[0].run.color;
|
|
838
|
+
if (hc && hc !== "#000000") hs.color = hc;
|
|
839
|
+
element.headerStyle = hs;
|
|
840
|
+
}
|
|
841
|
+
if (altColor) element.alternatingRowColor = altColor;
|
|
842
|
+
element.borders = borderColor ? { outer: true, inner: true, color: borderColor, width: 1 } : false;
|
|
843
|
+
for (const k of lineIdx) used.add(k);
|
|
844
|
+
out.push({ element, lineIdx, shapeIdx: gridShapes.map(({ k }) => k) });
|
|
845
|
+
i = best.j + 1;
|
|
846
|
+
}
|
|
847
|
+
return out;
|
|
848
|
+
}
|
|
849
|
+
|
|
850
|
+
// ../../packages/jdf-pdf-import/src/core.ts
|
|
851
|
+
var PT_TO_MM2 = 0.352778;
|
|
618
852
|
function classifyFont(name) {
|
|
619
853
|
const n = (name || "").toLowerCase();
|
|
620
854
|
const bold = /bold|black|heavy|semibold|demibold|extrabold/.test(n);
|
|
@@ -724,19 +958,36 @@ async function walkOps(page, OPS, viewport) {
|
|
|
724
958
|
const minY = Math.min(...ys), maxY = Math.max(...ys);
|
|
725
959
|
imagePositions.push({
|
|
726
960
|
name,
|
|
727
|
-
x: minX *
|
|
728
|
-
y: minY *
|
|
729
|
-
w: (maxX - minX) *
|
|
730
|
-
h: (maxY - minY) *
|
|
961
|
+
x: minX * PT_TO_MM2,
|
|
962
|
+
y: minY * PT_TO_MM2,
|
|
963
|
+
w: (maxX - minX) * PT_TO_MM2,
|
|
964
|
+
h: (maxY - minY) * PT_TO_MM2,
|
|
731
965
|
inline,
|
|
732
966
|
maskFill
|
|
733
967
|
});
|
|
734
968
|
}
|
|
735
969
|
let pathSegments = [];
|
|
736
970
|
let pathRect = null;
|
|
971
|
+
let pathRects = [];
|
|
737
972
|
let pathStart = null;
|
|
738
973
|
let pathLast = null;
|
|
739
974
|
function flushPath(isFill, isStroke) {
|
|
975
|
+
for (const r of pathRects) {
|
|
976
|
+
const tl = toViewport(r.x, r.y + r.h);
|
|
977
|
+
const br = toViewport(r.x + r.w, r.y);
|
|
978
|
+
shapes.push({
|
|
979
|
+
kind: "rect",
|
|
980
|
+
x: Math.min(tl.x, br.x) * PT_TO_MM2,
|
|
981
|
+
y: Math.min(tl.y, br.y) * PT_TO_MM2,
|
|
982
|
+
width: Math.abs(br.x - tl.x) * PT_TO_MM2,
|
|
983
|
+
height: Math.abs(br.y - tl.y) * PT_TO_MM2,
|
|
984
|
+
fill: isFill ? gs.fill : void 0,
|
|
985
|
+
stroke: isStroke ? gs.stroke : void 0,
|
|
986
|
+
strokeWidth: isStroke ? gs.lineWidth * PT_TO_MM2 : void 0,
|
|
987
|
+
opacity: isFill ? gs.fillAlpha : gs.strokeAlpha
|
|
988
|
+
});
|
|
989
|
+
}
|
|
990
|
+
pathRects = [];
|
|
740
991
|
if (pathRect) {
|
|
741
992
|
const tl = toViewport(pathRect.x, pathRect.y + pathRect.h);
|
|
742
993
|
const br = toViewport(pathRect.x + pathRect.w, pathRect.y);
|
|
@@ -746,13 +997,13 @@ async function walkOps(page, OPS, viewport) {
|
|
|
746
997
|
const h = Math.abs(br.y - tl.y);
|
|
747
998
|
shapes.push({
|
|
748
999
|
kind: "rect",
|
|
749
|
-
x: x *
|
|
750
|
-
y: y *
|
|
751
|
-
width: w *
|
|
752
|
-
height: h *
|
|
1000
|
+
x: x * PT_TO_MM2,
|
|
1001
|
+
y: y * PT_TO_MM2,
|
|
1002
|
+
width: w * PT_TO_MM2,
|
|
1003
|
+
height: h * PT_TO_MM2,
|
|
753
1004
|
fill: isFill ? gs.fill : void 0,
|
|
754
1005
|
stroke: isStroke ? gs.stroke : void 0,
|
|
755
|
-
strokeWidth: isStroke ? gs.lineWidth *
|
|
1006
|
+
strokeWidth: isStroke ? gs.lineWidth * PT_TO_MM2 : void 0,
|
|
756
1007
|
opacity: isFill ? gs.fillAlpha : gs.strokeAlpha
|
|
757
1008
|
});
|
|
758
1009
|
} else if (pathSegments.length === 2 && pathSegments[0].type === "M" && pathSegments[1].type === "L") {
|
|
@@ -764,35 +1015,35 @@ async function walkOps(page, OPS, viewport) {
|
|
|
764
1015
|
const minY = Math.min(va.y, vb.y);
|
|
765
1016
|
const maxX = Math.max(va.x, vb.x);
|
|
766
1017
|
const maxY = Math.max(va.y, vb.y);
|
|
767
|
-
const x1Local = (va.x - minX) *
|
|
768
|
-
const y1Local = (va.y - minY) *
|
|
769
|
-
const x2Local = (vb.x - minX) *
|
|
770
|
-
const y2Local = (vb.y - minY) *
|
|
771
|
-
const wLocal = Math.max(0.05, (maxX - minX) *
|
|
772
|
-
const hLocal = Math.max(0.05, (maxY - minY) *
|
|
1018
|
+
const x1Local = (va.x - minX) * PT_TO_MM2;
|
|
1019
|
+
const y1Local = (va.y - minY) * PT_TO_MM2;
|
|
1020
|
+
const x2Local = (vb.x - minX) * PT_TO_MM2;
|
|
1021
|
+
const y2Local = (vb.y - minY) * PT_TO_MM2;
|
|
1022
|
+
const wLocal = Math.max(0.05, (maxX - minX) * PT_TO_MM2);
|
|
1023
|
+
const hLocal = Math.max(0.05, (maxY - minY) * PT_TO_MM2);
|
|
773
1024
|
const dx = Math.abs(va.x - vb.x);
|
|
774
1025
|
const dy = Math.abs(va.y - vb.y);
|
|
775
1026
|
const axisAligned = dx < 0.5 || dy < 0.5;
|
|
776
1027
|
if (axisAligned) {
|
|
777
1028
|
shapes.push({
|
|
778
1029
|
kind: "line",
|
|
779
|
-
x: minX *
|
|
780
|
-
y: minY *
|
|
1030
|
+
x: minX * PT_TO_MM2,
|
|
1031
|
+
y: minY * PT_TO_MM2,
|
|
781
1032
|
width: wLocal,
|
|
782
1033
|
height: hLocal,
|
|
783
1034
|
stroke: isStroke ? gs.stroke : void 0,
|
|
784
|
-
strokeWidth: isStroke ? gs.lineWidth *
|
|
1035
|
+
strokeWidth: isStroke ? gs.lineWidth * PT_TO_MM2 : void 0,
|
|
785
1036
|
opacity: gs.strokeAlpha
|
|
786
1037
|
});
|
|
787
1038
|
} else {
|
|
788
1039
|
shapes.push({
|
|
789
1040
|
kind: "path",
|
|
790
|
-
x: minX *
|
|
791
|
-
y: minY *
|
|
1041
|
+
x: minX * PT_TO_MM2,
|
|
1042
|
+
y: minY * PT_TO_MM2,
|
|
792
1043
|
width: wLocal,
|
|
793
1044
|
height: hLocal,
|
|
794
1045
|
stroke: isStroke ? gs.stroke : void 0,
|
|
795
|
-
strokeWidth: isStroke ? gs.lineWidth *
|
|
1046
|
+
strokeWidth: isStroke ? gs.lineWidth * PT_TO_MM2 : void 0,
|
|
796
1047
|
opacity: gs.strokeAlpha,
|
|
797
1048
|
path: `M ${x1Local.toFixed(2)} ${y1Local.toFixed(2)} L ${x2Local.toFixed(2)} ${y2Local.toFixed(2)}`
|
|
798
1049
|
});
|
|
@@ -825,20 +1076,20 @@ async function walkOps(page, OPS, viewport) {
|
|
|
825
1076
|
if (seg.type === "Z") return "Z";
|
|
826
1077
|
const p = [];
|
|
827
1078
|
for (let i = 0; i < seg.pts.length; i += 2) {
|
|
828
|
-
p.push(((seg.pts[i] - minX) *
|
|
829
|
-
p.push(((seg.pts[i + 1] - minY) *
|
|
1079
|
+
p.push(((seg.pts[i] - minX) * PT_TO_MM2).toFixed(2));
|
|
1080
|
+
p.push(((seg.pts[i + 1] - minY) * PT_TO_MM2).toFixed(2));
|
|
830
1081
|
}
|
|
831
1082
|
return `${seg.type} ${p.join(" ")}`;
|
|
832
1083
|
}).join(" ");
|
|
833
1084
|
shapes.push({
|
|
834
1085
|
kind: "path",
|
|
835
|
-
x: minX *
|
|
836
|
-
y: minY *
|
|
837
|
-
width: bw *
|
|
838
|
-
height: bh *
|
|
1086
|
+
x: minX * PT_TO_MM2,
|
|
1087
|
+
y: minY * PT_TO_MM2,
|
|
1088
|
+
width: bw * PT_TO_MM2,
|
|
1089
|
+
height: bh * PT_TO_MM2,
|
|
839
1090
|
fill: isFill ? gs.fill : void 0,
|
|
840
1091
|
stroke: isStroke ? gs.stroke : void 0,
|
|
841
|
-
strokeWidth: isStroke ? gs.lineWidth *
|
|
1092
|
+
strokeWidth: isStroke ? gs.lineWidth * PT_TO_MM2 : void 0,
|
|
842
1093
|
opacity: isFill ? gs.fillAlpha : gs.strokeAlpha,
|
|
843
1094
|
path: d
|
|
844
1095
|
});
|
|
@@ -1000,6 +1251,12 @@ async function walkOps(page, OPS, viewport) {
|
|
|
1000
1251
|
} else if (op === OPS.closePath) {
|
|
1001
1252
|
pathSegments.push({ type: "Z", pts: [] });
|
|
1002
1253
|
if (pathStart) pathLast = { ...pathStart };
|
|
1254
|
+
} else if (op === OPS.rectangle) {
|
|
1255
|
+
const x = pathArgs[ai], y = pathArgs[ai + 1], w = pathArgs[ai + 2], h = pathArgs[ai + 3];
|
|
1256
|
+
ai += 4;
|
|
1257
|
+
const p1 = tx(gs.ctm, x, y);
|
|
1258
|
+
const p3 = tx(gs.ctm, x + w, y + h);
|
|
1259
|
+
pathRects.push({ x: Math.min(p1.x, p3.x), y: Math.min(p1.y, p3.y), w: Math.abs(p3.x - p1.x), h: Math.abs(p3.y - p1.y) });
|
|
1003
1260
|
}
|
|
1004
1261
|
}
|
|
1005
1262
|
} else if (fn === OPS.fill || fn === OPS.stroke || fn === OPS.fillStroke || fn === OPS.eoFill || fn === OPS.eoFillStroke || fn === OPS.closeFillStroke || fn === OPS.closeStroke || fn === OPS.closeEOFillStroke) {
|
|
@@ -1008,6 +1265,7 @@ async function walkOps(page, OPS, viewport) {
|
|
|
1008
1265
|
flushPath(isFill, isStroke);
|
|
1009
1266
|
} else if (fn === OPS.endPath || fn === OPS.clip || fn === OPS.eoClip) {
|
|
1010
1267
|
pathSegments = [];
|
|
1268
|
+
pathRects = [];
|
|
1011
1269
|
pathRect = null;
|
|
1012
1270
|
pathStart = null;
|
|
1013
1271
|
pathLast = null;
|
|
@@ -1172,10 +1430,10 @@ async function extractLinks(doc, page, viewport) {
|
|
|
1172
1430
|
const xMax = Math.max(c1.x, c2.x);
|
|
1173
1431
|
const yMax = Math.max(c1.y, c2.y);
|
|
1174
1432
|
const rectMm = {
|
|
1175
|
-
x: xMin *
|
|
1176
|
-
y: yMin *
|
|
1177
|
-
w: (xMax - xMin) *
|
|
1178
|
-
h: (yMax - yMin) *
|
|
1433
|
+
x: xMin * PT_TO_MM2,
|
|
1434
|
+
y: yMin * PT_TO_MM2,
|
|
1435
|
+
w: (xMax - xMin) * PT_TO_MM2,
|
|
1436
|
+
h: (yMax - yMin) * PT_TO_MM2
|
|
1179
1437
|
};
|
|
1180
1438
|
const url = a.url || a.unsafeUrl;
|
|
1181
1439
|
const destPage = url ? void 0 : await resolveDestPage(doc, a.dest);
|
|
@@ -1213,10 +1471,10 @@ async function extractFormWidgets(page, viewport) {
|
|
|
1213
1471
|
})).filter((o) => o.value !== "") : [];
|
|
1214
1472
|
out.push({
|
|
1215
1473
|
rectMm: {
|
|
1216
|
-
x: xMin *
|
|
1217
|
-
y: yMin *
|
|
1218
|
-
w: (xMax - xMin) *
|
|
1219
|
-
h: (yMax - yMin) *
|
|
1474
|
+
x: xMin * PT_TO_MM2,
|
|
1475
|
+
y: yMin * PT_TO_MM2,
|
|
1476
|
+
w: (xMax - xMin) * PT_TO_MM2,
|
|
1477
|
+
h: (yMax - yMin) * PT_TO_MM2
|
|
1220
1478
|
},
|
|
1221
1479
|
fieldType: a.fieldType || "",
|
|
1222
1480
|
fieldName: a.fieldName || `field-${out.length + 1}`,
|
|
@@ -1236,16 +1494,16 @@ async function extractFormWidgets(page, viewport) {
|
|
|
1236
1494
|
async function flattenOutline(doc, outline) {
|
|
1237
1495
|
if (!outline) return [];
|
|
1238
1496
|
const out = [];
|
|
1239
|
-
async function
|
|
1497
|
+
async function walk2(items, depth) {
|
|
1240
1498
|
for (const item of items) {
|
|
1241
1499
|
const idx = await resolveDestPage(doc, item.dest);
|
|
1242
1500
|
if (idx != null && typeof item.title === "string" && item.title.trim()) {
|
|
1243
1501
|
out.push({ title: item.title.trim(), pageIndex: idx, depth });
|
|
1244
1502
|
}
|
|
1245
|
-
if (item.items?.length) await
|
|
1503
|
+
if (item.items?.length) await walk2(item.items, depth + 1);
|
|
1246
1504
|
}
|
|
1247
1505
|
}
|
|
1248
|
-
await
|
|
1506
|
+
await walk2(outline, 1);
|
|
1249
1507
|
return out;
|
|
1250
1508
|
}
|
|
1251
1509
|
var normTitle = (s) => s.toLowerCase().replace(/[\s\u00a0]+/g, " ").replace(/[^\p{L}\p{N} ]/gu, "").trim();
|
|
@@ -1498,12 +1756,12 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
|
1498
1756
|
const w = safeNum(it.width, 0);
|
|
1499
1757
|
runs.push({
|
|
1500
1758
|
text: it.str,
|
|
1501
|
-
x: safeNum(vx *
|
|
1502
|
-
y: safeNum(yTop *
|
|
1759
|
+
x: safeNum(vx * PT_TO_MM2, 0),
|
|
1760
|
+
y: safeNum(yTop * PT_TO_MM2, 0),
|
|
1503
1761
|
fontSize: safeNum(fontSize, 10),
|
|
1504
1762
|
fontName: it.fontName,
|
|
1505
|
-
width: safeNum(w *
|
|
1506
|
-
height: safeNum((it.height || fontSize) *
|
|
1763
|
+
width: safeNum(w * PT_TO_MM2, 0),
|
|
1764
|
+
height: safeNum((it.height || fontSize) * PT_TO_MM2, fontSize * PT_TO_MM2),
|
|
1507
1765
|
color: op?.fill || "#000000",
|
|
1508
1766
|
opacity: invisible ? 0 : safeNum(op?.alpha, 1)
|
|
1509
1767
|
});
|
|
@@ -1511,6 +1769,12 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
|
1511
1769
|
runs.sort((a, b) => a.y - b.y || a.x - b.x);
|
|
1512
1770
|
const lines = [];
|
|
1513
1771
|
const Y_TOL = 0.6;
|
|
1772
|
+
const kGlyph = calibrateGlyphWidth(runs);
|
|
1773
|
+
const stretchedSpaces = hasStretchedSpaces(runs, kGlyph);
|
|
1774
|
+
const fontKey = (name) => {
|
|
1775
|
+
const c = fontMap.get(name) || classifyFont(name || "");
|
|
1776
|
+
return `${c.family}|${c.weight || ""}|${c.style || ""}`;
|
|
1777
|
+
};
|
|
1514
1778
|
for (const r of runs) {
|
|
1515
1779
|
if (!r.text.length) continue;
|
|
1516
1780
|
const last = lines[lines.length - 1];
|
|
@@ -1519,24 +1783,48 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
|
1519
1783
|
continue;
|
|
1520
1784
|
}
|
|
1521
1785
|
const sameLine = Math.abs(last.y - r.y) <= Y_TOL;
|
|
1522
|
-
const sameStyle = Math.abs(last.fontSize - r.fontSize) < 0.4 && last.fontName === r.fontName && last.color === r.color && Math.abs(last.opacity - r.opacity) < 0.05;
|
|
1523
|
-
const
|
|
1524
|
-
const
|
|
1525
|
-
|
|
1786
|
+
const sameStyle = Math.abs(last.fontSize - r.fontSize) < 0.4 && (last.fontName === r.fontName || fontKey(last.fontName) === fontKey(r.fontName)) && last.color === r.color && Math.abs(last.opacity - r.opacity) < 0.05;
|
|
1787
|
+
const emMm = r.fontSize * PT_TO_MM2;
|
|
1788
|
+
const extent = (t) => {
|
|
1789
|
+
if (!/\s$/.test(t.text)) return t.width;
|
|
1790
|
+
const em = t.fontSize * PT_TO_MM2;
|
|
1791
|
+
const est = Math.max(1, t.text.trim().length) * em * kGlyph + em * 0.25;
|
|
1792
|
+
if (stretchedSpaces) return Math.min(t.width, est);
|
|
1793
|
+
return t.width > est * 1.4 ? est : t.width;
|
|
1794
|
+
};
|
|
1795
|
+
const gapMm = r.x - (last.x + extent(last));
|
|
1796
|
+
const mergeOk = sameLine && sameStyle && gapMm >= -emMm * 0.5 && gapMm <= emMm * 0.45;
|
|
1526
1797
|
if (mergeOk) {
|
|
1527
1798
|
const lastEndsSpace = /\s$/.test(last.text);
|
|
1528
1799
|
const currStartsSpace = /^\s/.test(r.text);
|
|
1529
1800
|
const sep = gapMm > emMm * 0.08 && !lastEndsSpace && !currStartsSpace ? " " : "";
|
|
1530
1801
|
last.text = last.text + sep + r.text;
|
|
1531
1802
|
const newExtent = r.x - last.x + r.width;
|
|
1532
|
-
last.width = Math.max(last
|
|
1803
|
+
last.width = Math.max(extent(last), newExtent);
|
|
1533
1804
|
} else {
|
|
1534
1805
|
lines.push({ ...r });
|
|
1535
1806
|
}
|
|
1536
1807
|
}
|
|
1537
1808
|
const elements = [];
|
|
1538
|
-
|
|
1539
|
-
|
|
1809
|
+
const tRuns = lines.map((l) => {
|
|
1810
|
+
const cls = fontMap.get(l.fontName) || classifyFont(l.fontName || "");
|
|
1811
|
+
return { text: l.text, x: l.x, y: l.y, width: l.width, height: l.height, fontSize: l.fontSize, fontName: l.fontName, color: l.color, bold: cls.weight === "bold" };
|
|
1812
|
+
});
|
|
1813
|
+
const detected = options.detectTables === false ? [] : detectTables(tRuns, ops.shapes, pageW * PT_TO_MM2);
|
|
1814
|
+
const consumedLines = /* @__PURE__ */ new Set();
|
|
1815
|
+
const consumedShapes = /* @__PURE__ */ new Set();
|
|
1816
|
+
const tableAtLine = /* @__PURE__ */ new Map();
|
|
1817
|
+
for (const t of detected) {
|
|
1818
|
+
for (const k of t.lineIdx) consumedLines.add(k);
|
|
1819
|
+
for (const k of t.shapeIdx) consumedShapes.add(k);
|
|
1820
|
+
tableAtLine.set(Math.min(...t.lineIdx), t.element);
|
|
1821
|
+
}
|
|
1822
|
+
const pageWmm = pageW * PT_TO_MM2, pageHmm = pageH * PT_TO_MM2;
|
|
1823
|
+
ops.shapes.forEach((sh, shapeIdx) => {
|
|
1824
|
+
if (consumedShapes.has(shapeIdx)) return;
|
|
1825
|
+
if (sh.width < 0.3 && sh.height < 0.3) return;
|
|
1826
|
+
if (sh.x + sh.width <= 0 || sh.y + sh.height <= 0 || sh.x >= pageWmm || sh.y >= pageHmm) return;
|
|
1827
|
+
if (sh.kind === "rect" && sh.fill && !sh.stroke && sh.width * sh.height >= pageWmm * pageHmm * 0.9) return;
|
|
1540
1828
|
const shapeType = sh.kind;
|
|
1541
1829
|
const shape = {
|
|
1542
1830
|
type: "shape",
|
|
@@ -1552,7 +1840,7 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
|
1552
1840
|
shape.style = { opacity: Math.round(sh.opacity * 100) / 100 };
|
|
1553
1841
|
}
|
|
1554
1842
|
elements.push(shape);
|
|
1555
|
-
}
|
|
1843
|
+
});
|
|
1556
1844
|
const imgs = await extractImages(page, ops.imagePositions, runtime, dataUrlCache);
|
|
1557
1845
|
for (const { pos, dataUrl } of imgs) {
|
|
1558
1846
|
let resourceKey = resourceKeyByName.get(pos.name);
|
|
@@ -1575,7 +1863,16 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
|
1575
1863
|
fit: "fill"
|
|
1576
1864
|
});
|
|
1577
1865
|
}
|
|
1866
|
+
const sizeChars = /* @__PURE__ */ new Map();
|
|
1578
1867
|
for (const l of lines) {
|
|
1868
|
+
const k = Math.round(l.fontSize * 2) / 2;
|
|
1869
|
+
sizeChars.set(k, (sizeChars.get(k) ?? 0) + l.text.length);
|
|
1870
|
+
}
|
|
1871
|
+
const bodyFontSize = [...sizeChars.entries()].sort((a, b) => b[1] - a[1])[0]?.[0] ?? 0;
|
|
1872
|
+
lines.forEach((l, lineIdx) => {
|
|
1873
|
+
const tableEl = tableAtLine.get(lineIdx);
|
|
1874
|
+
if (tableEl) elements.push(tableEl);
|
|
1875
|
+
if (consumedLines.has(lineIdx)) return;
|
|
1579
1876
|
const cls = fontMap.get(l.fontName) || classifyFont(l.fontName || "");
|
|
1580
1877
|
const style = {
|
|
1581
1878
|
fontSize: Math.round(l.fontSize * 10) / 10,
|
|
@@ -1586,8 +1883,7 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
|
1586
1883
|
if (l.color !== "#000000") style.color = l.color;
|
|
1587
1884
|
if (l.opacity < 0.999) style.opacity = Math.round(l.opacity * 100) / 100;
|
|
1588
1885
|
const link = findLinkForRun2(l);
|
|
1589
|
-
const
|
|
1590
|
-
const measured = Math.max(l.width + l.fontSize * PT_TO_MM * 0.4, l.fontSize * PT_TO_MM);
|
|
1886
|
+
const measured = Math.max(l.width + l.fontSize * PT_TO_MM2 * 0.4, l.fontSize * PT_TO_MM2);
|
|
1591
1887
|
const remaining = Math.max(measured, pageWmm - l.x);
|
|
1592
1888
|
const elWidth = Math.min(measured, remaining);
|
|
1593
1889
|
const text = {
|
|
@@ -1597,18 +1893,26 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
|
1597
1893
|
width: Math.max(2, Math.round(elWidth * 100) / 100),
|
|
1598
1894
|
style
|
|
1599
1895
|
};
|
|
1600
|
-
if (cls.weight === "bold") {
|
|
1601
|
-
|
|
1602
|
-
|
|
1603
|
-
else if (l.fontSize >=
|
|
1896
|
+
if (cls.weight === "bold" && l.text.trim().length <= 120 && !consumedLines.has(lineIdx)) {
|
|
1897
|
+
const ratio = bodyFontSize > 0 ? l.fontSize / bodyFontSize : 1;
|
|
1898
|
+
if (l.fontSize >= 22 || ratio >= 1.8) text.heading = 1;
|
|
1899
|
+
else if (l.fontSize >= 17 || ratio >= 1.35) text.heading = 2;
|
|
1900
|
+
else if (l.fontSize >= 16 || ratio >= 1.2) text.heading = 3;
|
|
1604
1901
|
}
|
|
1605
1902
|
if (text.heading) text.tocEntry = text.content;
|
|
1606
1903
|
if (link) {
|
|
1607
1904
|
if (link.url) text.link = link.url;
|
|
1608
1905
|
else if (link.destPage != null) text.link = { type: "internal", target: `#page-${link.destPage + 1}` };
|
|
1609
1906
|
}
|
|
1907
|
+
const prev = elements[elements.length - 1];
|
|
1908
|
+
if (text.heading && prev && prev.type === "text" && prev.heading === text.heading && !link && !prev.link && Math.abs(prev.style?.fontSize - style.fontSize) < 0.5 && Math.abs(prev.position.x - text.position.x) < 1 && text.position.y - prev.position.y < l.fontSize * PT_TO_MM2 * 2.2 && text.position.y > prev.position.y) {
|
|
1909
|
+
prev.content = `${prev.content} ${text.content}`.replace(/\s+/g, " ");
|
|
1910
|
+
prev.tocEntry = prev.content;
|
|
1911
|
+
prev.width = Math.max(prev.width ?? 0, text.width ?? 0);
|
|
1912
|
+
return;
|
|
1913
|
+
}
|
|
1610
1914
|
elements.push(text);
|
|
1611
|
-
}
|
|
1915
|
+
});
|
|
1612
1916
|
for (const w of formWidgets) {
|
|
1613
1917
|
if (w.pushButton) continue;
|
|
1614
1918
|
const baseEl = {
|
|
@@ -1643,7 +1947,7 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
|
1643
1947
|
}
|
|
1644
1948
|
pages.push({
|
|
1645
1949
|
id: `page-${pi}`,
|
|
1646
|
-
pageSize: { width: Math.round(pageW *
|
|
1950
|
+
pageSize: { width: Math.round(pageW * PT_TO_MM2 * 100) / 100, height: Math.round(pageH * PT_TO_MM2 * 100) / 100 },
|
|
1647
1951
|
margins: { top: 0, right: 0, bottom: 0, left: 0 },
|
|
1648
1952
|
elements
|
|
1649
1953
|
});
|
|
@@ -1796,13 +2100,13 @@ async function importPdfToJdf2(source, title, options = {}) {
|
|
|
1796
2100
|
|
|
1797
2101
|
// src/commands/import-pdf.ts
|
|
1798
2102
|
async function importPdf(inputPath, outputPath, options = {}) {
|
|
1799
|
-
const input =
|
|
1800
|
-
if (!
|
|
2103
|
+
const input = path7.resolve(inputPath);
|
|
2104
|
+
if (!fs7.existsSync(input)) {
|
|
1801
2105
|
console.error(`File not found: ${input}`);
|
|
1802
2106
|
process.exit(1);
|
|
1803
2107
|
}
|
|
1804
2108
|
console.log(`Importing: ${input}`);
|
|
1805
|
-
const title =
|
|
2109
|
+
const title = path7.basename(input, path7.extname(input));
|
|
1806
2110
|
const t0 = Date.now();
|
|
1807
2111
|
const doc = await importPdfToJdf2(input, title, {
|
|
1808
2112
|
password: options.password,
|
|
@@ -1811,7 +2115,7 @@ async function importPdf(inputPath, outputPath, options = {}) {
|
|
|
1811
2115
|
console.log(`Parsed in ${((Date.now() - t0) / 1e3).toFixed(1)}s \u2014 ${doc.pages.length} page(s)`);
|
|
1812
2116
|
let output;
|
|
1813
2117
|
if (outputPath) {
|
|
1814
|
-
output =
|
|
2118
|
+
output = path7.resolve(outputPath);
|
|
1815
2119
|
} else {
|
|
1816
2120
|
const stem = input.replace(/\.pdf$/i, "");
|
|
1817
2121
|
const wantJdfx = !options.forceJson && shouldUseJdfx(doc);
|
|
@@ -1820,11 +2124,11 @@ async function importPdf(inputPath, outputPath, options = {}) {
|
|
|
1820
2124
|
console.log(`Output: ${output}`);
|
|
1821
2125
|
if (output.toLowerCase().endsWith(".jdfx")) {
|
|
1822
2126
|
const { bytes, manifest } = await packJdfx(doc);
|
|
1823
|
-
|
|
2127
|
+
fs7.writeFileSync(output, bytes);
|
|
1824
2128
|
console.log(`
|
|
1825
2129
|
Done! Created ${doc.pages.length} page(s), ${manifest.assets.length} asset(s) bundled`);
|
|
1826
2130
|
} else {
|
|
1827
|
-
|
|
2131
|
+
fs7.writeFileSync(output, JSON.stringify(doc, null, 2));
|
|
1828
2132
|
console.log(`
|
|
1829
2133
|
Done! Created ${doc.pages.length} page(s)`);
|
|
1830
2134
|
}
|
|
@@ -1837,23 +2141,23 @@ var ImportJsonError = class extends Error {
|
|
|
1837
2141
|
}
|
|
1838
2142
|
};
|
|
1839
2143
|
async function importJson(inputPath, outputPath, options = {}) {
|
|
1840
|
-
const input =
|
|
1841
|
-
if (!
|
|
2144
|
+
const input = path7.resolve(inputPath);
|
|
2145
|
+
if (!fs7.existsSync(input)) {
|
|
1842
2146
|
throw new ImportJsonError(`File not found: ${input}`);
|
|
1843
2147
|
}
|
|
1844
2148
|
console.log(`Importing: ${input}`);
|
|
1845
|
-
const raw =
|
|
2149
|
+
const raw = fs7.readFileSync(input, "utf-8");
|
|
1846
2150
|
let parsed;
|
|
1847
2151
|
try {
|
|
1848
2152
|
parsed = JSON.parse(raw);
|
|
1849
2153
|
} catch (e) {
|
|
1850
2154
|
throw new ImportJsonError(`Not valid JSON: ${e.message}`);
|
|
1851
2155
|
}
|
|
1852
|
-
const title =
|
|
2156
|
+
const title = path7.basename(input, path7.extname(input));
|
|
1853
2157
|
const doc = normaliseToJdf(parsed, title);
|
|
1854
2158
|
let output;
|
|
1855
2159
|
if (outputPath) {
|
|
1856
|
-
output =
|
|
2160
|
+
output = path7.resolve(outputPath);
|
|
1857
2161
|
} else {
|
|
1858
2162
|
const stem = input.replace(/\.json$/i, "");
|
|
1859
2163
|
const wantJdfx = !options.forceJson && shouldUseJdfx(doc);
|
|
@@ -1862,11 +2166,11 @@ async function importJson(inputPath, outputPath, options = {}) {
|
|
|
1862
2166
|
console.log(`Output: ${output}`);
|
|
1863
2167
|
if (output.toLowerCase().endsWith(".jdfx")) {
|
|
1864
2168
|
const { bytes, manifest } = await packJdfx(doc);
|
|
1865
|
-
|
|
2169
|
+
fs7.writeFileSync(output, bytes);
|
|
1866
2170
|
console.log(`
|
|
1867
2171
|
Done! Created ${doc.pages.length} page(s), ${manifest.assets.length} asset(s) bundled`);
|
|
1868
2172
|
} else {
|
|
1869
|
-
|
|
2173
|
+
fs7.writeFileSync(output, JSON.stringify(doc, null, 2));
|
|
1870
2174
|
console.log(`
|
|
1871
2175
|
Done! Created ${doc.pages.length} page(s)`);
|
|
1872
2176
|
}
|
|
@@ -1942,6 +2246,47 @@ function wrapElements(elements, title, meta) {
|
|
|
1942
2246
|
]
|
|
1943
2247
|
};
|
|
1944
2248
|
}
|
|
2249
|
+
var DEFAULT_TRANSCRIPT_WINDOW = 45;
|
|
2250
|
+
var fmtTime = (sec) => {
|
|
2251
|
+
const s = Math.max(0, Math.round(sec));
|
|
2252
|
+
const h = Math.floor(s / 3600), m = Math.floor(s % 3600 / 60), r = s % 60;
|
|
2253
|
+
return h ? `${h}:${String(m).padStart(2, "0")}:${String(r).padStart(2, "0")}` : `${String(m).padStart(2, "0")}:${String(r).padStart(2, "0")}`;
|
|
2254
|
+
};
|
|
2255
|
+
function transcriptChunks(el, elementId2, page, crumb, windowSec, maxTokens) {
|
|
2256
|
+
const segs = Array.isArray(el?.transcript?.segments) ? el.transcript.segments : [];
|
|
2257
|
+
if (!segs.length) return [];
|
|
2258
|
+
const chapters = Array.isArray(el?.chapters) ? [...el.chapters].sort((a, b) => a.t - b.t) : [];
|
|
2259
|
+
const chapterAt = (t) => {
|
|
2260
|
+
let cur = null;
|
|
2261
|
+
for (const c of chapters) {
|
|
2262
|
+
if (c.t <= t + 1e-6) cur = c;
|
|
2263
|
+
else break;
|
|
2264
|
+
}
|
|
2265
|
+
return cur;
|
|
2266
|
+
};
|
|
2267
|
+
const out = [];
|
|
2268
|
+
let win = [];
|
|
2269
|
+
const flush = () => {
|
|
2270
|
+
if (!win.length) return;
|
|
2271
|
+
const t0 = win[0].t0, t1 = win[win.length - 1].t1;
|
|
2272
|
+
const body = win.map((sg) => sg.speaker ? `${sg.speaker}: ${sg.text}` : sg.text).join(" ").replace(/\s+/g, " ").trim();
|
|
2273
|
+
const text = `[${fmtTime(t0)}\u2013${fmtTime(t1)}] ${body}`;
|
|
2274
|
+
const chapter = chapterAt(t0);
|
|
2275
|
+
const path9 = [...crumb, ...el.title ? [String(el.title)] : [], ...chapter ? [String(chapter.title)] : []];
|
|
2276
|
+
out.push({ id: `${elementId2}@${Math.round(t0)}`, text, path: path9, page, types: ["video"], tokens: estimateTokens(text), hash: hashText(text), media: { element: elementId2, t0, t1 } });
|
|
2277
|
+
win = [];
|
|
2278
|
+
};
|
|
2279
|
+
for (const sg of segs) {
|
|
2280
|
+
if (typeof sg?.text !== "string" || !sg.text.trim()) continue;
|
|
2281
|
+
const startsNewChapter = win.length && chapterAt(sg.t0) !== chapterAt(win[0].t0);
|
|
2282
|
+
const spansWindow = win.length && sg.t1 - win[0].t0 > windowSec;
|
|
2283
|
+
const overBudget = win.length && estimateTokens(win.map((w) => w.text).join(" ") + sg.text) > maxTokens;
|
|
2284
|
+
if (startsNewChapter || spansWindow || overBudget) flush();
|
|
2285
|
+
win.push(sg);
|
|
2286
|
+
}
|
|
2287
|
+
flush();
|
|
2288
|
+
return out;
|
|
2289
|
+
}
|
|
1945
2290
|
var DEFAULT_MAX_TOKENS = 512;
|
|
1946
2291
|
function estimateTokens(text) {
|
|
1947
2292
|
return Math.ceil(text.length / 4);
|
|
@@ -1964,11 +2309,11 @@ function serializeElement(el) {
|
|
|
1964
2309
|
case "richtext":
|
|
1965
2310
|
return (e.runs || []).map((r) => r.text ?? "").join("").trim();
|
|
1966
2311
|
case "list": {
|
|
1967
|
-
const
|
|
2312
|
+
const walk2 = (items, depth = 0) => (items || []).flatMap((it) => {
|
|
1968
2313
|
const line = " ".repeat(depth) + "- " + String(it.content ?? "").trim();
|
|
1969
|
-
return it.children?.length ? [line, ...
|
|
2314
|
+
return it.children?.length ? [line, ...walk2(it.children, depth + 1)] : [line];
|
|
1970
2315
|
});
|
|
1971
|
-
return
|
|
2316
|
+
return walk2(e.items).join("\n");
|
|
1972
2317
|
}
|
|
1973
2318
|
case "table": {
|
|
1974
2319
|
const headers = e.headers ?? e.columns?.map((c) => c.header || "").filter((h) => h) ?? [];
|
|
@@ -1999,6 +2344,8 @@ function serializeElement(el) {
|
|
|
1999
2344
|
return `${e.checked ? "[x]" : "[ ]"} ${e.label ?? ""}`.trim();
|
|
2000
2345
|
case "image":
|
|
2001
2346
|
return e.alt ? `[image: ${e.alt}]` : "";
|
|
2347
|
+
case "video":
|
|
2348
|
+
return e.title ? `[video: ${e.title}]` : "";
|
|
2002
2349
|
case "toc":
|
|
2003
2350
|
case "shape":
|
|
2004
2351
|
case "signature":
|
|
@@ -2038,8 +2385,15 @@ function makeChunk(group, breadcrumb) {
|
|
|
2038
2385
|
function chunkDocument(doc, options = {}) {
|
|
2039
2386
|
const strategy = options.strategy ?? "section";
|
|
2040
2387
|
const maxTokens = options.maxTokens ?? DEFAULT_MAX_TOKENS;
|
|
2388
|
+
const windowSec = options.transcriptWindowSec ?? DEFAULT_TRANSCRIPT_WINDOW;
|
|
2041
2389
|
const flat = flatten(doc);
|
|
2042
2390
|
const chunks = [];
|
|
2391
|
+
const withTranscripts = (group, crumb2, c) => {
|
|
2392
|
+
if (c) chunks.push(c);
|
|
2393
|
+
for (const f of group) {
|
|
2394
|
+
if (f.el.type === "video") chunks.push(...transcriptChunks(f.el, f.id, f.page, crumb2, windowSec, maxTokens));
|
|
2395
|
+
}
|
|
2396
|
+
};
|
|
2043
2397
|
if (strategy === "element") {
|
|
2044
2398
|
const crumb2 = [];
|
|
2045
2399
|
for (const f of flat) {
|
|
@@ -2048,8 +2402,7 @@ function chunkDocument(doc, options = {}) {
|
|
|
2048
2402
|
crumb2.length = Math.max(0, lvl - 1);
|
|
2049
2403
|
crumb2[lvl - 1] = serializeElement(f.el);
|
|
2050
2404
|
}
|
|
2051
|
-
|
|
2052
|
-
if (c) chunks.push(c);
|
|
2405
|
+
withTranscripts([f], crumb2, makeChunk([f], crumb2));
|
|
2053
2406
|
}
|
|
2054
2407
|
return chunks;
|
|
2055
2408
|
}
|
|
@@ -2058,8 +2411,7 @@ function chunkDocument(doc, options = {}) {
|
|
|
2058
2411
|
let buf2 = [];
|
|
2059
2412
|
let bufTokens = 0;
|
|
2060
2413
|
const flush = () => {
|
|
2061
|
-
|
|
2062
|
-
if (c) chunks.push(c);
|
|
2414
|
+
withTranscripts(buf2, crumb2, makeChunk(buf2, crumb2));
|
|
2063
2415
|
buf2 = [];
|
|
2064
2416
|
bufTokens = 0;
|
|
2065
2417
|
};
|
|
@@ -2086,22 +2438,21 @@ function chunkDocument(doc, options = {}) {
|
|
|
2086
2438
|
for (const f of buf) {
|
|
2087
2439
|
const t = estimateTokens(serializeElement(f.el));
|
|
2088
2440
|
if (subTokens + t > maxTokens && sub.length > 0) {
|
|
2089
|
-
|
|
2090
|
-
if (c2) chunks.push(c2);
|
|
2441
|
+
withTranscripts(sub, crumb, makeChunk(sub, crumb));
|
|
2091
2442
|
sub = [];
|
|
2092
2443
|
subTokens = 0;
|
|
2093
2444
|
}
|
|
2094
2445
|
sub.push(f);
|
|
2095
2446
|
subTokens += t;
|
|
2096
2447
|
}
|
|
2097
|
-
|
|
2098
|
-
if (c) chunks.push(c);
|
|
2448
|
+
withTranscripts(sub, crumb, makeChunk(sub, crumb));
|
|
2099
2449
|
buf = [];
|
|
2100
2450
|
};
|
|
2101
2451
|
for (const f of flat) {
|
|
2102
2452
|
const lvl = headingLevel(f.el);
|
|
2103
2453
|
if (lvl != null) {
|
|
2104
|
-
|
|
2454
|
+
const onlyHeadings = buf.length > 0 && buf.every((b) => headingLevel(b.el) != null);
|
|
2455
|
+
if (!onlyHeadings) flushSection();
|
|
2105
2456
|
crumb.length = Math.max(0, lvl - 1);
|
|
2106
2457
|
crumb[lvl - 1] = serializeElement(f.el);
|
|
2107
2458
|
}
|
|
@@ -2112,34 +2463,34 @@ function chunkDocument(doc, options = {}) {
|
|
|
2112
2463
|
}
|
|
2113
2464
|
async function loadJdf(filePath) {
|
|
2114
2465
|
if (filePath.toLowerCase().endsWith(".jdfx")) {
|
|
2115
|
-
const zip = await JSZip.loadAsync(
|
|
2466
|
+
const zip = await JSZip.loadAsync(fs7.readFileSync(filePath));
|
|
2116
2467
|
const docFile = zip.file(JDFX_DOCUMENT_PATH);
|
|
2117
2468
|
if (!docFile) throw new Error(`Bundle missing ${JDFX_DOCUMENT_PATH}`);
|
|
2118
2469
|
return JSON.parse(await docFile.async("string"));
|
|
2119
2470
|
}
|
|
2120
|
-
return JSON.parse(
|
|
2471
|
+
return JSON.parse(fs7.readFileSync(filePath, "utf-8"));
|
|
2121
2472
|
}
|
|
2122
2473
|
async function chunkFile(inputPath, opts = {}) {
|
|
2123
|
-
const input =
|
|
2124
|
-
if (!
|
|
2474
|
+
const input = path7.resolve(inputPath);
|
|
2475
|
+
if (!fs7.existsSync(input)) throw new Error(`File not found: ${input}`);
|
|
2125
2476
|
const doc = await loadJdf(input);
|
|
2126
2477
|
const strategy = opts.strategy ?? "section";
|
|
2127
|
-
const chunks = chunkDocument(doc, { strategy, maxTokens: opts.maxTokens });
|
|
2478
|
+
const chunks = chunkDocument(doc, { strategy, maxTokens: opts.maxTokens, transcriptWindowSec: opts.transcriptWindowSec });
|
|
2128
2479
|
const format = opts.format ?? "jsonl";
|
|
2129
2480
|
console.log(`Chunking: ${input}`);
|
|
2130
2481
|
console.log(`Strategy: ${strategy}${opts.maxTokens ? ` (max ${opts.maxTokens} tokens)` : ""}`);
|
|
2131
2482
|
if (format === "inline") {
|
|
2132
|
-
const out = opts.output ?
|
|
2483
|
+
const out = opts.output ? path7.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".jdf");
|
|
2133
2484
|
const withIndex = { ...doc, index: { chunker: `jdf-${strategy}-v1`, chunks } };
|
|
2134
|
-
|
|
2485
|
+
fs7.writeFileSync(out, JSON.stringify(withIndex, null, 2));
|
|
2135
2486
|
console.log(`Output: ${out} (${chunks.length} chunks in "index" block)`);
|
|
2136
2487
|
} else if (format === "json") {
|
|
2137
|
-
const out = opts.output ?
|
|
2138
|
-
|
|
2488
|
+
const out = opts.output ? path7.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".chunks.json");
|
|
2489
|
+
fs7.writeFileSync(out, JSON.stringify(chunks, null, 2));
|
|
2139
2490
|
console.log(`Output: ${out} (${chunks.length} chunks)`);
|
|
2140
2491
|
} else {
|
|
2141
|
-
const out = opts.output ?
|
|
2142
|
-
|
|
2492
|
+
const out = opts.output ? path7.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".chunks.jsonl");
|
|
2493
|
+
fs7.writeFileSync(out, chunks.map((c) => JSON.stringify(c)).join("\n") + "\n");
|
|
2143
2494
|
console.log(`Output: ${out} (${chunks.length} chunks)`);
|
|
2144
2495
|
}
|
|
2145
2496
|
const totalTokens = chunks.reduce((a, c) => a + c.tokens, 0);
|
|
@@ -2311,17 +2662,17 @@ async function embedOpenAI(model, inputs) {
|
|
|
2311
2662
|
}
|
|
2312
2663
|
async function loadJdf2(filePath) {
|
|
2313
2664
|
if (filePath.toLowerCase().endsWith(".jdfx")) {
|
|
2314
|
-
const zip = await JSZip.loadAsync(
|
|
2665
|
+
const zip = await JSZip.loadAsync(fs7.readFileSync(filePath));
|
|
2315
2666
|
const docFile = zip.file(JDFX_DOCUMENT_PATH);
|
|
2316
2667
|
if (!docFile) throw new Error(`Bundle missing ${JDFX_DOCUMENT_PATH}`);
|
|
2317
2668
|
return JSON.parse(await docFile.async("string"));
|
|
2318
2669
|
}
|
|
2319
|
-
return JSON.parse(
|
|
2670
|
+
return JSON.parse(fs7.readFileSync(filePath, "utf-8"));
|
|
2320
2671
|
}
|
|
2321
2672
|
function loadCache(cachePath) {
|
|
2322
2673
|
try {
|
|
2323
|
-
if (!
|
|
2324
|
-
return JSON.parse(
|
|
2674
|
+
if (!fs7.existsSync(cachePath)) return null;
|
|
2675
|
+
return JSON.parse(fs7.readFileSync(cachePath, "utf-8"));
|
|
2325
2676
|
} catch {
|
|
2326
2677
|
return null;
|
|
2327
2678
|
}
|
|
@@ -2332,15 +2683,15 @@ function batched(items, size) {
|
|
|
2332
2683
|
return out;
|
|
2333
2684
|
}
|
|
2334
2685
|
async function embedFile(inputPath, opts = {}) {
|
|
2335
|
-
const input =
|
|
2336
|
-
if (!
|
|
2686
|
+
const input = path7.resolve(inputPath);
|
|
2687
|
+
if (!fs7.existsSync(input)) throw new Error(`File not found: ${input}`);
|
|
2337
2688
|
const provider = opts.provider ?? "ollama";
|
|
2338
2689
|
const model = opts.model ?? DEFAULT_MODEL[provider];
|
|
2339
2690
|
const strategy = opts.strategy ?? "section";
|
|
2340
2691
|
const doc = await loadJdf2(input);
|
|
2341
|
-
const chunks = chunkDocument(doc, { strategy, maxTokens: opts.maxTokens });
|
|
2342
|
-
const output = opts.output ?
|
|
2343
|
-
const cachePath = opts.cache ?
|
|
2692
|
+
const chunks = chunkDocument(doc, { strategy, maxTokens: opts.maxTokens, transcriptWindowSec: opts.transcriptWindowSec });
|
|
2693
|
+
const output = opts.output ? path7.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".embeddings.json");
|
|
2694
|
+
const cachePath = opts.cache ? path7.resolve(opts.cache) : output;
|
|
2344
2695
|
console.log(`Embedding: ${input}`);
|
|
2345
2696
|
console.log(`Provider: ${provider} / ${model}${provider === "ollama" ? " (local \u2014 no data leaves this machine)" : " (remote API)"}`);
|
|
2346
2697
|
console.log(`Strategy: ${strategy} \u2192 ${chunks.length} chunks`);
|
|
@@ -2379,11 +2730,295 @@ async function embedFile(inputPath, opts = {}) {
|
|
|
2379
2730
|
chunker: `jdf-${strategy}-v1`,
|
|
2380
2731
|
vectors
|
|
2381
2732
|
};
|
|
2382
|
-
|
|
2733
|
+
fs7.writeFileSync(output, JSON.stringify(sidecar));
|
|
2383
2734
|
console.log(`
|
|
2384
2735
|
Done! ${Object.keys(vectors).length} vectors (${dims}-dim) \u2192 ${output}`);
|
|
2385
2736
|
return sidecar;
|
|
2386
2737
|
}
|
|
2738
|
+
var toSec = (ts) => {
|
|
2739
|
+
const m = ts.trim().replace(",", ".").match(/^(?:(\d+):)?(\d{1,2}):(\d{2}(?:\.\d+)?)$/);
|
|
2740
|
+
if (!m) throw new Error(`bad timestamp "${ts}"`);
|
|
2741
|
+
return (m[1] ? Number(m[1]) * 3600 : 0) + Number(m[2]) * 60 + Number(m[3]);
|
|
2742
|
+
};
|
|
2743
|
+
function parseSubtitles(text, filename = "") {
|
|
2744
|
+
const trimmed = text.replace(/^/, "").trim();
|
|
2745
|
+
if (trimmed.startsWith("{") || trimmed.startsWith("[")) {
|
|
2746
|
+
const j = JSON.parse(trimmed);
|
|
2747
|
+
const arr = Array.isArray(j) ? j : Array.isArray(j.segments) ? j.segments : [];
|
|
2748
|
+
return arr.map((sg) => ({
|
|
2749
|
+
t0: Number(sg.t0 ?? sg.start ?? sg.from ?? 0),
|
|
2750
|
+
t1: Number(sg.t1 ?? sg.end ?? sg.to ?? 0),
|
|
2751
|
+
text: String(sg.text ?? "").trim(),
|
|
2752
|
+
...sg.speaker ? { speaker: String(sg.speaker) } : {}
|
|
2753
|
+
})).filter((sg) => sg.text);
|
|
2754
|
+
}
|
|
2755
|
+
const segs = [];
|
|
2756
|
+
for (const block of trimmed.split(/\r?\n\r?\n+/)) {
|
|
2757
|
+
const lines = block.split(/\r?\n/).filter((l) => l.trim() !== "" && l.trim() !== "WEBVTT");
|
|
2758
|
+
const ti = lines.findIndex((l) => l.includes("-->"));
|
|
2759
|
+
if (ti < 0) continue;
|
|
2760
|
+
const [a, b] = lines[ti].split("-->").map((x) => x.trim().split(/\s+/)[0]);
|
|
2761
|
+
const body = lines.slice(ti + 1).join(" ").replace(/<[^>]+>/g, "").replace(/\s+/g, " ").trim();
|
|
2762
|
+
if (!body) continue;
|
|
2763
|
+
segs.push({ t0: toSec(a), t1: toSec(b), text: body });
|
|
2764
|
+
}
|
|
2765
|
+
if (!segs.length) throw new Error(`no cues found in ${filename || "subtitle input"} (expected SRT, WebVTT or JSON segments)`);
|
|
2766
|
+
return segs;
|
|
2767
|
+
}
|
|
2768
|
+
function parseChapters(text) {
|
|
2769
|
+
const t = text.trim();
|
|
2770
|
+
if (t.startsWith("[")) return JSON.parse(t).map((c) => ({ t: Number(c.t ?? c.start ?? 0), title: String(c.title ?? "") }));
|
|
2771
|
+
return t.split(/\r?\n/).map((l) => l.trim()).filter(Boolean).map((l) => {
|
|
2772
|
+
const m = l.match(/^(\S+)\s+(.+)$/);
|
|
2773
|
+
if (!m) throw new Error(`bad chapter line "${l}" (expected "mm:ss Title")`);
|
|
2774
|
+
return { t: toSec(m[1]), title: m[2].trim() };
|
|
2775
|
+
});
|
|
2776
|
+
}
|
|
2777
|
+
async function loadDoc(file) {
|
|
2778
|
+
if (file.toLowerCase().endsWith(".jdfx")) {
|
|
2779
|
+
const zip = await JSZip.loadAsync(fs7.readFileSync(file));
|
|
2780
|
+
const f = zip.file(JDFX_DOCUMENT_PATH);
|
|
2781
|
+
if (!f) throw new Error(`Bundle missing ${JDFX_DOCUMENT_PATH}`);
|
|
2782
|
+
const doc = JSON.parse(await f.async("string"));
|
|
2783
|
+
const manifest = zip.file("manifest.json") ? JSON.parse(await zip.file("manifest.json").async("string")) : { assets: [] };
|
|
2784
|
+
for (const a of manifest.assets ?? []) {
|
|
2785
|
+
const af = zip.file(a.path);
|
|
2786
|
+
if (!af) continue;
|
|
2787
|
+
const data = (await af.async("nodebuffer")).toString("base64");
|
|
2788
|
+
const res = { src: "embedded", mimeType: a.mimeType, data };
|
|
2789
|
+
doc.resources ??= {};
|
|
2790
|
+
if (/^video\//i.test(a.mimeType || "")) (doc.resources.videos ??= {})[a.id] = res;
|
|
2791
|
+
else (doc.resources.images ??= {})[a.id] = res;
|
|
2792
|
+
}
|
|
2793
|
+
return { doc, bundle: true, zip };
|
|
2794
|
+
}
|
|
2795
|
+
return { doc: JSON.parse(fs7.readFileSync(file, "utf-8")), bundle: false };
|
|
2796
|
+
}
|
|
2797
|
+
function findVideos(doc) {
|
|
2798
|
+
const out = [];
|
|
2799
|
+
const walk2 = (els, page) => {
|
|
2800
|
+
for (const el of els ?? []) {
|
|
2801
|
+
if (el?.type === "video") out.push({ el, page, index: out.length });
|
|
2802
|
+
if (el?.elements) walk2(el.elements, page);
|
|
2803
|
+
}
|
|
2804
|
+
};
|
|
2805
|
+
doc.pages.forEach((p, i) => walk2(p.elements, i + 1));
|
|
2806
|
+
return out;
|
|
2807
|
+
}
|
|
2808
|
+
async function clipToTempFile(doc, el, docDir) {
|
|
2809
|
+
const tmp = path7.join(fs7.mkdtempSync(path7.join(__require("os").tmpdir(), "jdf-transcribe-")), "clip.mp4");
|
|
2810
|
+
const res = el.resource ? doc.resources?.videos?.[el.resource] ?? doc.resources?.images?.[el.resource] : void 0;
|
|
2811
|
+
if (res?.data) {
|
|
2812
|
+
fs7.writeFileSync(tmp, Buffer.from(res.data.replace(/^data:[^,]*,/, ""), "base64"));
|
|
2813
|
+
return tmp;
|
|
2814
|
+
}
|
|
2815
|
+
if (res?.path) return path7.resolve(docDir, res.path);
|
|
2816
|
+
const src = el.src;
|
|
2817
|
+
if (!src) return null;
|
|
2818
|
+
if (src.startsWith("data:")) {
|
|
2819
|
+
fs7.writeFileSync(tmp, Buffer.from(src.replace(/^data:[^,]*,/, ""), "base64"));
|
|
2820
|
+
return tmp;
|
|
2821
|
+
}
|
|
2822
|
+
if (/^https?:\/\//i.test(src)) {
|
|
2823
|
+
const r = await fetch(src);
|
|
2824
|
+
if (!r.ok) throw new Error(`download failed ${r.status}: ${src}`);
|
|
2825
|
+
fs7.writeFileSync(tmp, Buffer.from(await r.arrayBuffer()));
|
|
2826
|
+
return tmp;
|
|
2827
|
+
}
|
|
2828
|
+
const local = path7.resolve(docDir, src);
|
|
2829
|
+
return fs7.existsSync(local) ? local : null;
|
|
2830
|
+
}
|
|
2831
|
+
function whisperCli(clip, model, language, prompt2) {
|
|
2832
|
+
const ffmpeg = spawnSync("ffmpeg", ["-version"]);
|
|
2833
|
+
if (ffmpeg.error) throw new Error("ffmpeg not found \u2014 needed to extract audio for whisper-cli (brew install ffmpeg)");
|
|
2834
|
+
const wav = clip.replace(/\.[^.]+$/, "") + ".16k.wav";
|
|
2835
|
+
const ex = spawnSync("ffmpeg", ["-y", "-i", clip, "-vn", "-ac", "1", "-ar", "16000", "-f", "wav", wav], { encoding: "utf-8" });
|
|
2836
|
+
if (ex.status !== 0) throw new Error(`ffmpeg failed: ${ex.stderr.slice(-400)}`);
|
|
2837
|
+
const args = ["-f", wav, "-oj", "-of", wav.replace(/\.wav$/, "")];
|
|
2838
|
+
if (model) args.push("-m", model);
|
|
2839
|
+
if (language) args.push("-l", language);
|
|
2840
|
+
if (prompt2) args.push("--prompt", prompt2);
|
|
2841
|
+
const run = spawnSync("whisper-cli", args, { encoding: "utf-8" });
|
|
2842
|
+
if (run.error) throw new Error("whisper-cli not found \u2014 install whisper.cpp (brew install whisper-cpp) or use --from / --provider openai");
|
|
2843
|
+
if (run.status !== 0) throw new Error(`whisper-cli failed: ${run.stderr.slice(-400)}`);
|
|
2844
|
+
const j = JSON.parse(fs7.readFileSync(wav.replace(/\.wav$/, "") + ".json", "utf-8"));
|
|
2845
|
+
const segs = j.transcription ?? j.segments ?? [];
|
|
2846
|
+
const ms = (x) => typeof x === "number" ? x / 1e3 : toSec(String(x).replace(",", "."));
|
|
2847
|
+
return segs.map((sg) => ({ t0: ms(sg.offsets?.from ?? sg.start), t1: ms(sg.offsets?.to ?? sg.end), text: String(sg.text ?? "").trim() })).filter((sg) => sg.text);
|
|
2848
|
+
}
|
|
2849
|
+
async function openaiTranscribe(clip, model, language, prompt2) {
|
|
2850
|
+
const key = process.env.OPENAI_API_KEY;
|
|
2851
|
+
if (!key) throw new Error("OPENAI_API_KEY is not set");
|
|
2852
|
+
const base = process.env.OPENAI_BASE_URL || "https://api.openai.com/v1";
|
|
2853
|
+
const form = new FormData();
|
|
2854
|
+
form.append("file", new Blob([fs7.readFileSync(clip)]), path7.basename(clip));
|
|
2855
|
+
form.append("model", model || "whisper-1");
|
|
2856
|
+
form.append("response_format", "verbose_json");
|
|
2857
|
+
form.append("timestamp_granularities[]", "segment");
|
|
2858
|
+
if (language) form.append("language", language);
|
|
2859
|
+
if (prompt2) form.append("prompt", prompt2);
|
|
2860
|
+
const r = await fetch(`${base}/audio/transcriptions`, { method: "POST", headers: { Authorization: `Bearer ${key}` }, body: form });
|
|
2861
|
+
if (!r.ok) throw new Error(`OpenAI transcription failed ${r.status}: ${(await r.text()).slice(0, 300)}`);
|
|
2862
|
+
const j = await r.json();
|
|
2863
|
+
return (j.segments ?? []).map((sg) => ({ t0: Number(sg.start), t1: Number(sg.end), text: String(sg.text).trim() })).filter((sg) => sg.text);
|
|
2864
|
+
}
|
|
2865
|
+
async function transcribeFile(inputPath, opts = {}) {
|
|
2866
|
+
const input = path7.resolve(inputPath);
|
|
2867
|
+
if (!fs7.existsSync(input)) throw new Error(`File not found: ${input}`);
|
|
2868
|
+
const { doc, bundle } = await loadDoc(input);
|
|
2869
|
+
const videos = findVideos(doc);
|
|
2870
|
+
if (!videos.length) throw new Error("document has no video element");
|
|
2871
|
+
let target = videos[0];
|
|
2872
|
+
if (opts.element != null) {
|
|
2873
|
+
const byId = videos.find((v) => v.el.id === opts.element);
|
|
2874
|
+
const byIdx = /^\d+$/.test(opts.element) ? videos[Number(opts.element)] : void 0;
|
|
2875
|
+
target = byId ?? byIdx ?? (() => {
|
|
2876
|
+
throw new Error(`no video element "${opts.element}" (have: ${videos.map((v) => v.el.id ?? `#${v.index}`).join(", ")})`);
|
|
2877
|
+
})();
|
|
2878
|
+
} else if (videos.length > 1) {
|
|
2879
|
+
throw new Error(`document has ${videos.length} videos \u2014 pick one with --element <id|index>`);
|
|
2880
|
+
}
|
|
2881
|
+
let segments;
|
|
2882
|
+
let source;
|
|
2883
|
+
if (opts.from) {
|
|
2884
|
+
segments = parseSubtitles(fs7.readFileSync(path7.resolve(opts.from), "utf-8"), opts.from);
|
|
2885
|
+
source = `${path7.extname(opts.from).slice(1).toLowerCase() || "file"}-import`;
|
|
2886
|
+
} else {
|
|
2887
|
+
const provider = opts.provider ?? "whisper-cli";
|
|
2888
|
+
const clip = await clipToTempFile(doc, target.el, path7.dirname(input));
|
|
2889
|
+
if (!clip) throw new Error("could not locate the clip bytes (no bundled asset, data URL, local path or http URL) \u2014 use --from to import subtitles instead");
|
|
2890
|
+
segments = provider === "openai" ? await openaiTranscribe(clip, opts.model, opts.language, opts.prompt) : whisperCli(clip, opts.model, opts.language, opts.prompt);
|
|
2891
|
+
source = provider === "openai" ? `openai:${opts.model || "whisper-1"}` : `whisper-cli${opts.model ? ":" + path7.basename(opts.model) : ""}`;
|
|
2892
|
+
}
|
|
2893
|
+
segments.sort((a, b) => a.t0 - b.t0);
|
|
2894
|
+
const transcript = { ...opts.language ? { language: opts.language } : {}, source, created: (/* @__PURE__ */ new Date()).toISOString(), segments };
|
|
2895
|
+
target.el.transcript = transcript;
|
|
2896
|
+
if (opts.chapters) target.el.chapters = parseChapters(fs7.readFileSync(path7.resolve(opts.chapters), "utf-8"));
|
|
2897
|
+
if (!target.el.id) target.el.id = `video-${target.index + 1}`;
|
|
2898
|
+
const output = opts.output ? path7.resolve(opts.output) : input;
|
|
2899
|
+
if (output.toLowerCase().endsWith(".jdfx") || bundle && !opts.output) {
|
|
2900
|
+
const { bytes } = await packJdfx(doc);
|
|
2901
|
+
fs7.writeFileSync(output, bytes);
|
|
2902
|
+
} else {
|
|
2903
|
+
fs7.writeFileSync(output, JSON.stringify(doc, null, 2));
|
|
2904
|
+
}
|
|
2905
|
+
const dur = segments.length ? segments[segments.length - 1].t1 : 0;
|
|
2906
|
+
console.log(`Transcribed: ${path7.basename(input)} \u2192 element "${target.el.id}" (${segments.length} segments, ${Math.round(dur)} s, source ${source})`);
|
|
2907
|
+
if (target.el.chapters) console.log(`Chapters: ${target.el.chapters.length}`);
|
|
2908
|
+
console.log(`Output: ${output}
|
|
2909
|
+
Next: jdf chunk ${path7.basename(output)} # transcript \u2192 time-windowed chunks with media.t0/t1`);
|
|
2910
|
+
return transcript;
|
|
2911
|
+
}
|
|
2912
|
+
var CONFIG_NAME = "jdf.rag.json";
|
|
2913
|
+
var OUT_DIR = ".jdf-rag";
|
|
2914
|
+
function walk(dir, acc = []) {
|
|
2915
|
+
for (const ent of fs7.readdirSync(dir, { withFileTypes: true })) {
|
|
2916
|
+
if (ent.name === "node_modules" || ent.name === OUT_DIR || ent.name.startsWith(".")) continue;
|
|
2917
|
+
const p = path7.join(dir, ent.name);
|
|
2918
|
+
if (ent.isDirectory()) walk(p, acc);
|
|
2919
|
+
else if (/\.(jdf|jdfx)$/i.test(ent.name)) acc.push(p);
|
|
2920
|
+
}
|
|
2921
|
+
return acc.sort();
|
|
2922
|
+
}
|
|
2923
|
+
async function readDoc(file) {
|
|
2924
|
+
if (file.toLowerCase().endsWith(".jdfx")) {
|
|
2925
|
+
const zip = await JSZip.loadAsync(fs7.readFileSync(file));
|
|
2926
|
+
return JSON.parse(await zip.file(JDFX_DOCUMENT_PATH).async("string"));
|
|
2927
|
+
}
|
|
2928
|
+
return JSON.parse(fs7.readFileSync(file, "utf-8"));
|
|
2929
|
+
}
|
|
2930
|
+
function videosIn(doc) {
|
|
2931
|
+
const out = [];
|
|
2932
|
+
const w = (els) => {
|
|
2933
|
+
for (const el of els ?? []) {
|
|
2934
|
+
if (el?.type === "video") out.push({ id: el.id, hasTranscript: !!el.transcript?.segments?.length });
|
|
2935
|
+
if (el?.elements) w(el.elements);
|
|
2936
|
+
}
|
|
2937
|
+
};
|
|
2938
|
+
for (const p of doc.pages ?? []) w(p.elements);
|
|
2939
|
+
return out;
|
|
2940
|
+
}
|
|
2941
|
+
async function ragFolder(dirPath, cli = {}) {
|
|
2942
|
+
const dir = path7.resolve(dirPath);
|
|
2943
|
+
if (!fs7.existsSync(dir) || !fs7.statSync(dir).isDirectory()) throw new Error(`Not a directory: ${dir}`);
|
|
2944
|
+
const cfgPath = path7.join(dir, CONFIG_NAME);
|
|
2945
|
+
const cfg = fs7.existsSync(cfgPath) ? JSON.parse(fs7.readFileSync(cfgPath, "utf-8")) : {};
|
|
2946
|
+
const opts = { ...cfg, ...Object.fromEntries(Object.entries(cli).filter(([, v]) => v !== void 0)) };
|
|
2947
|
+
const provider = opts.provider ?? "ollama";
|
|
2948
|
+
const transcribe = opts.transcribe ?? "none";
|
|
2949
|
+
const outDir = path7.resolve(opts.out ?? path7.join(dir, OUT_DIR));
|
|
2950
|
+
const files = walk(dir);
|
|
2951
|
+
console.log(`jdf rag: ${dir}
|
|
2952
|
+
files: ${files.length} (.jdf/.jdfx)${fs7.existsSync(cfgPath) ? `
|
|
2953
|
+
config: ${CONFIG_NAME}` : ""}
|
|
2954
|
+
embeddings: ${opts.noEmbed ? "skipped (--no-embed)" : `${provider}${opts.model ? " / " + opts.model : ""}`}
|
|
2955
|
+
transcribe: ${transcribe}${opts.dryRun ? "\n DRY RUN \u2014 nothing written" : ""}
|
|
2956
|
+
`);
|
|
2957
|
+
if (!files.length) {
|
|
2958
|
+
console.log("Nothing to do.");
|
|
2959
|
+
return;
|
|
2960
|
+
}
|
|
2961
|
+
const manifest = { dir, created: (/* @__PURE__ */ new Date()).toISOString(), provider: opts.noEmbed ? null : provider, model: opts.model ?? null, strategy: opts.strategy ?? "section", transcribe, files: [], totals: { files: files.length, chunks: 0, videoChunks: 0, videos: 0, transcribed: 0, untranscribed: 0 } };
|
|
2962
|
+
const indexLines = [];
|
|
2963
|
+
for (const file of files) {
|
|
2964
|
+
const rel = path7.relative(dir, file);
|
|
2965
|
+
const doc = await readDoc(file);
|
|
2966
|
+
const vids = videosIn(doc);
|
|
2967
|
+
let transcribedHere = 0;
|
|
2968
|
+
for (const v of vids) {
|
|
2969
|
+
if (v.hasTranscript) continue;
|
|
2970
|
+
if (transcribe === "none") {
|
|
2971
|
+
manifest.totals.untranscribed++;
|
|
2972
|
+
continue;
|
|
2973
|
+
}
|
|
2974
|
+
if (opts.dryRun) {
|
|
2975
|
+
transcribedHere++;
|
|
2976
|
+
continue;
|
|
2977
|
+
}
|
|
2978
|
+
try {
|
|
2979
|
+
await transcribeFile(file, { provider: transcribe, element: v.id, model: opts.transcribeModel, language: opts.language, prompt: opts.prompt });
|
|
2980
|
+
transcribedHere++;
|
|
2981
|
+
} catch (e) {
|
|
2982
|
+
console.warn(` ! ${rel}: transcription failed for video ${v.id ?? "#?"}: ${e.message}`);
|
|
2983
|
+
manifest.totals.untranscribed++;
|
|
2984
|
+
}
|
|
2985
|
+
}
|
|
2986
|
+
manifest.totals.videos += vids.length;
|
|
2987
|
+
manifest.totals.transcribed += transcribedHere;
|
|
2988
|
+
if (opts.dryRun) {
|
|
2989
|
+
manifest.files.push({ file: rel, videos: vids.length, wouldTranscribe: transcribedHere });
|
|
2990
|
+
continue;
|
|
2991
|
+
}
|
|
2992
|
+
const chunkOpts = { strategy: opts.strategy, maxTokens: opts.maxTokens, transcriptWindowSec: opts.transcriptWindowSec };
|
|
2993
|
+
const chunkOut = path7.join(outDir, "chunks", rel.replace(/\.(jdf|jdfx)$/i, ".chunks.jsonl"));
|
|
2994
|
+
fs7.mkdirSync(path7.dirname(chunkOut), { recursive: true });
|
|
2995
|
+
let chunks;
|
|
2996
|
+
if (opts.noEmbed) {
|
|
2997
|
+
chunks = await chunkFile(file, { ...chunkOpts, format: "jsonl", output: chunkOut });
|
|
2998
|
+
} else {
|
|
2999
|
+
const side = await embedFile(file, { ...chunkOpts, provider, model: opts.model, incremental: true });
|
|
3000
|
+
chunks = await chunkFile(file, { ...chunkOpts, format: "jsonl", output: chunkOut });
|
|
3001
|
+
manifest.files.push({ file: rel, chunks: chunks.length, vectors: Object.keys(side.vectors).length, sidecar: path7.relative(dir, file.replace(/\.(jdf|jdfx)$/i, ".embeddings.json")), videos: vids.length, transcribed: transcribedHere });
|
|
3002
|
+
}
|
|
3003
|
+
if (opts.noEmbed) manifest.files.push({ file: rel, chunks: chunks.length, videos: vids.length, transcribed: transcribedHere });
|
|
3004
|
+
for (const c of chunks) {
|
|
3005
|
+
indexLines.push(JSON.stringify({ file: rel, ...c }));
|
|
3006
|
+
manifest.totals.chunks++;
|
|
3007
|
+
if (c.media) manifest.totals.videoChunks++;
|
|
3008
|
+
}
|
|
3009
|
+
}
|
|
3010
|
+
if (!opts.dryRun) {
|
|
3011
|
+
fs7.mkdirSync(outDir, { recursive: true });
|
|
3012
|
+
fs7.writeFileSync(path7.join(outDir, "index.jsonl"), indexLines.join("\n") + (indexLines.length ? "\n" : ""));
|
|
3013
|
+
fs7.writeFileSync(path7.join(outDir, "manifest.json"), JSON.stringify(manifest, null, 2) + "\n");
|
|
3014
|
+
}
|
|
3015
|
+
const t = manifest.totals;
|
|
3016
|
+
console.log(`
|
|
3017
|
+
Done. ${t.files} files \u2192 ${t.chunks} chunks (${t.videoChunks} from video transcripts); videos ${t.videos}, transcribed now ${t.transcribed}, still without transcript ${t.untranscribed}${t.untranscribed && transcribe === "none" ? " (pass --transcribe whisper-cli|openai, or jdf transcribe --from subs.srt)" : ""}.`);
|
|
3018
|
+
if (!opts.dryRun) console.log(`Index: ${path7.join(outDir, "index.jsonl")}
|
|
3019
|
+
Report: ${path7.join(outDir, "manifest.json")}${opts.noEmbed ? "" : `
|
|
3020
|
+
Vectors: one <file>.embeddings.json next to each document (incremental \u2014 re-run any time)`}`);
|
|
3021
|
+
}
|
|
2387
3022
|
|
|
2388
3023
|
// src/index.ts
|
|
2389
3024
|
var HELP = `jdf \u2014 JSON Document Format CLI
|
|
@@ -2395,12 +3030,18 @@ The CLI exists for these workflows:
|
|
|
2395
3030
|
into a validated .jdf (or .jdfx) you can ship.
|
|
2396
3031
|
\u2022 JDF \u2192 chunks turn a document into retrieval-ready chunks (RAG).
|
|
2397
3032
|
\u2022 JDF \u2192 vectors embed those chunks, incrementally, for a vector store.
|
|
3033
|
+
\u2022 video \u2192 text attach a time-stamped transcript to a video element so
|
|
3034
|
+
RAG retrieves "video at 02:13", not just "a video".
|
|
3035
|
+
\u2022 folder \u2192 index one command over a directory of .jdf/.jdfx: transcribe,
|
|
3036
|
+
chunk, embed incrementally, write .jdf-rag/index.jsonl.
|
|
2398
3037
|
|
|
2399
3038
|
Usage:
|
|
2400
3039
|
jdf validate <file.jdf>
|
|
2401
3040
|
jdf convert <file.{pdf,json,md}> [-o output.{jdf,jdfx}] [--json] [--password PW] [--drop-invisible-text]
|
|
2402
3041
|
jdf chunk <file.{jdf,jdfx}> [--strategy section|element|fixed] [--format jsonl|json|inline] [--max-tokens N] [-o out]
|
|
2403
3042
|
jdf embed <file.{jdf,jdfx}> [--provider ollama|openai] [--model NAME] [--strategy \u2026] [--incremental] [-o out]
|
|
3043
|
+
jdf transcribe <file.{jdf,jdfx}> [--from subs.srt|.vtt|.json] [--provider whisper-cli|openai] [--element ID] [--chapters FILE] [-o out]
|
|
3044
|
+
jdf rag <dir> [--provider ollama|openai] [--model NAME] [--transcribe none|whisper-cli|openai] [--no-embed] [--dry-run] [--out DIR]
|
|
2404
3045
|
jdf --help
|
|
2405
3046
|
|
|
2406
3047
|
Commands:
|
|
@@ -2408,6 +3049,8 @@ Commands:
|
|
|
2408
3049
|
convert Convert a PDF, JSON, or Markdown file into JDF (alias: import)
|
|
2409
3050
|
chunk Split a JDF document into retrieval-ready chunks (offline, deterministic)
|
|
2410
3051
|
embed Compute embeddings for the chunks (local via Ollama by default)
|
|
3052
|
+
transcribe Store time-stamped text on a video element (import SRT/VTT/JSON, or run Whisper)
|
|
3053
|
+
rag Make a whole folder retrieval-ready (finds .jdf/.jdfx, transcribes, chunks, embeds, indexes)
|
|
2411
3054
|
|
|
2412
3055
|
Flags:
|
|
2413
3056
|
-o, --output <path> Explicit output path
|
|
@@ -2424,7 +3067,20 @@ Flags:
|
|
|
2424
3067
|
--provider <p> embed: ollama (default, local) | openai (remote API)
|
|
2425
3068
|
--model <name> embed: model id (default: nomic-embed-text / text-embedding-3-small)
|
|
2426
3069
|
--incremental embed: skip chunks whose content hash is unchanged
|
|
3070
|
+
--cache <path> embed: sidecar to reuse vectors from (default: the
|
|
3071
|
+
output path itself)
|
|
2427
3072
|
--no-auto-start embed(ollama): don't auto-launch Ollama via Docker
|
|
3073
|
+
--from <file> transcribe: import subtitles (.srt / .vtt / JSON segments) \u2014 offline, no model
|
|
3074
|
+
--element <id|n> transcribe: which video element (id, or 0-based index); default the only one
|
|
3075
|
+
--chapters <file> transcribe: JSON [{t,title}] or "mm:ss Title" lines \u2192 chapter breadcrumbs
|
|
3076
|
+
--language <tag> transcribe: BCP-47 language hint for Whisper
|
|
3077
|
+
--prompt <text> transcribe: Whisper vocabulary hint (names, acronyms) \u2014 not an instruction
|
|
3078
|
+
--window <sec> chunk/embed/rag: transcript window per video chunk (default 45)
|
|
3079
|
+
--transcribe <p> rag: none (default) | whisper-cli | openai \u2014 for videos that have no transcript yet
|
|
3080
|
+
--no-embed rag: chunk + index only
|
|
3081
|
+
--dry-run rag: list what would happen, write nothing
|
|
3082
|
+
--out <dir> rag: index folder (default <dir>/.jdf-rag)
|
|
3083
|
+
rag reads defaults from <dir>/jdf.rag.json (same keys as the flags; flags win)
|
|
2428
3084
|
|
|
2429
3085
|
Environment (embed):
|
|
2430
3086
|
ollama: OLLAMA_HOST (default http://localhost:11434)
|
|
@@ -2438,8 +3094,11 @@ Examples:
|
|
|
2438
3094
|
jdf chunk report.jdf --format inline # embed the chunk index into the .jdf
|
|
2439
3095
|
jdf embed report.jdf # local embeddings via Ollama (auto-setup)
|
|
2440
3096
|
jdf embed report.jdf --provider openai --incremental
|
|
3097
|
+
jdf transcribe talk.jdfx --from talk.srt --chapters chapters.txt # then: jdf chunk talk.jdfx
|
|
3098
|
+
jdf transcribe talk.jdfx --provider openai --language en --prompt "JDF, jdfx, Ollama"
|
|
3099
|
+
jdf rag ./knowledge-base --transcribe openai # whole folder \u2192 .jdf-rag/index.jsonl
|
|
2441
3100
|
`;
|
|
2442
|
-
var BOOLEAN_FLAGS = /* @__PURE__ */ new Set(["help", "h", "json", "verbose", "skip-validate", "incremental", "no-auto-start", "drop-invisible-text"]);
|
|
3101
|
+
var BOOLEAN_FLAGS = /* @__PURE__ */ new Set(["help", "h", "json", "verbose", "skip-validate", "incremental", "no-auto-start", "drop-invisible-text", "no-embed", "dry-run"]);
|
|
2443
3102
|
function parseArgs(argv) {
|
|
2444
3103
|
const positional = [];
|
|
2445
3104
|
const flags = {};
|
|
@@ -2543,14 +3202,55 @@ async function main() {
|
|
|
2543
3202
|
strategy: typeof flags.strategy === "string" ? flags.strategy : void 0,
|
|
2544
3203
|
format: typeof flags.format === "string" ? flags.format : void 0,
|
|
2545
3204
|
maxTokens: typeof flags["max-tokens"] === "string" ? parseInt(flags["max-tokens"], 10) : void 0,
|
|
3205
|
+
transcriptWindowSec: typeof flags.window === "string" ? parseInt(flags.window, 10) : void 0,
|
|
3206
|
+
output: typeof flags.output === "string" ? flags.output : void 0
|
|
3207
|
+
});
|
|
3208
|
+
process.exit(0);
|
|
3209
|
+
}
|
|
3210
|
+
case "transcribe": {
|
|
3211
|
+
const input = positional[0];
|
|
3212
|
+
if (!input) {
|
|
3213
|
+
console.error("Usage: jdf transcribe <file.{jdf,jdfx}> [--from subs.srt|.vtt|.json] [--provider whisper-cli|openai] [--model M] [--language tag] [--element id|n] [--chapters file] [-o out]");
|
|
3214
|
+
process.exit(1);
|
|
3215
|
+
}
|
|
3216
|
+
await transcribeFile(input, {
|
|
3217
|
+
from: typeof flags.from === "string" ? flags.from : void 0,
|
|
3218
|
+
provider: typeof flags.provider === "string" ? flags.provider : void 0,
|
|
3219
|
+
model: typeof flags.model === "string" ? flags.model : void 0,
|
|
3220
|
+
language: typeof flags.language === "string" ? flags.language : void 0,
|
|
3221
|
+
prompt: typeof flags.prompt === "string" ? flags.prompt : void 0,
|
|
3222
|
+
element: typeof flags.element === "string" ? flags.element : void 0,
|
|
3223
|
+
chapters: typeof flags.chapters === "string" ? flags.chapters : void 0,
|
|
2546
3224
|
output: typeof flags.output === "string" ? flags.output : void 0
|
|
2547
3225
|
});
|
|
2548
3226
|
process.exit(0);
|
|
2549
3227
|
}
|
|
3228
|
+
case "rag": {
|
|
3229
|
+
const input = positional[0];
|
|
3230
|
+
if (!input) {
|
|
3231
|
+
console.error("Usage: jdf rag <dir> [--provider ollama|openai] [--model NAME] [--strategy \u2026] [--window sec] [--transcribe none|whisper-cli|openai] [--language tag] [--prompt text] [--no-embed] [--dry-run] [--out DIR]");
|
|
3232
|
+
process.exit(1);
|
|
3233
|
+
}
|
|
3234
|
+
await ragFolder(input, {
|
|
3235
|
+
provider: typeof flags.provider === "string" ? flags.provider : void 0,
|
|
3236
|
+
model: typeof flags.model === "string" ? flags.model : void 0,
|
|
3237
|
+
strategy: typeof flags.strategy === "string" ? flags.strategy : void 0,
|
|
3238
|
+
maxTokens: typeof flags["max-tokens"] === "string" ? parseInt(flags["max-tokens"], 10) : void 0,
|
|
3239
|
+
transcriptWindowSec: typeof flags.window === "string" ? parseInt(flags.window, 10) : void 0,
|
|
3240
|
+
transcribe: typeof flags.transcribe === "string" ? flags.transcribe : void 0,
|
|
3241
|
+
transcribeModel: typeof flags["transcribe-model"] === "string" ? flags["transcribe-model"] : void 0,
|
|
3242
|
+
language: typeof flags.language === "string" ? flags.language : void 0,
|
|
3243
|
+
prompt: typeof flags.prompt === "string" ? flags.prompt : void 0,
|
|
3244
|
+
noEmbed: flags["no-embed"] === true,
|
|
3245
|
+
dryRun: flags["dry-run"] === true,
|
|
3246
|
+
out: typeof flags.out === "string" ? flags.out : void 0
|
|
3247
|
+
});
|
|
3248
|
+
process.exit(0);
|
|
3249
|
+
}
|
|
2550
3250
|
case "embed": {
|
|
2551
3251
|
const input = positional[0];
|
|
2552
3252
|
if (!input) {
|
|
2553
|
-
console.error("Usage: jdf embed <file.{jdf,jdfx}> [--provider ollama|openai] [--model NAME] [--strategy \u2026] [--incremental] [--no-auto-start] [-o out]");
|
|
3253
|
+
console.error("Usage: jdf embed <file.{jdf,jdfx}> [--provider ollama|openai] [--model NAME] [--strategy \u2026] [--incremental] [--cache prev.embeddings.json] [--no-auto-start] [-o out]");
|
|
2554
3254
|
process.exit(1);
|
|
2555
3255
|
}
|
|
2556
3256
|
await embedFile(input, {
|
|
@@ -2559,8 +3259,10 @@ async function main() {
|
|
|
2559
3259
|
strategy: typeof flags.strategy === "string" ? flags.strategy : void 0,
|
|
2560
3260
|
maxTokens: typeof flags["max-tokens"] === "string" ? parseInt(flags["max-tokens"], 10) : void 0,
|
|
2561
3261
|
incremental: flags.incremental === true,
|
|
3262
|
+
transcriptWindowSec: typeof flags.window === "string" ? parseInt(flags.window, 10) : void 0,
|
|
2562
3263
|
autoStart: flags["no-auto-start"] !== true,
|
|
2563
|
-
output: typeof flags.output === "string" ? flags.output : void 0
|
|
3264
|
+
output: typeof flags.output === "string" ? flags.output : void 0,
|
|
3265
|
+
cache: typeof flags.cache === "string" ? flags.cache : void 0
|
|
2564
3266
|
});
|
|
2565
3267
|
process.exit(0);
|
|
2566
3268
|
}
|