@uurtech/jdf-cli 0.1.25 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.js +1299 -238
- package/dist/jdf-schema.json +33 -2
- package/package.json +1 -1
package/dist/index.js
CHANGED
|
@@ -1,13 +1,20 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
|
-
import
|
|
3
|
-
import
|
|
2
|
+
import fs7 from 'fs';
|
|
3
|
+
import path7 from 'path';
|
|
4
4
|
import { fileURLToPath } from 'url';
|
|
5
5
|
import Ajv from 'ajv';
|
|
6
6
|
import addFormats from 'ajv-formats';
|
|
7
7
|
import JSZip from 'jszip';
|
|
8
8
|
import crypto, { createHash } from 'crypto';
|
|
9
9
|
import { readFile } from 'fs/promises';
|
|
10
|
-
import { execFileSync } from 'child_process';
|
|
10
|
+
import { execFileSync, spawnSync } from 'child_process';
|
|
11
|
+
|
|
12
|
+
var __require = /* @__PURE__ */ ((x) => typeof require !== "undefined" ? require : typeof Proxy !== "undefined" ? new Proxy(x, {
|
|
13
|
+
get: (a, b) => (typeof require !== "undefined" ? require : a)[b]
|
|
14
|
+
}) : x)(function(x) {
|
|
15
|
+
if (typeof require !== "undefined") return require.apply(this, arguments);
|
|
16
|
+
throw Error('Dynamic require of "' + x + '" is not supported');
|
|
17
|
+
});
|
|
11
18
|
|
|
12
19
|
// ../../packages/jdf-core/src/manifest.ts
|
|
13
20
|
var JDFX_MANIFEST_VERSION = "1.0.0";
|
|
@@ -16,17 +23,17 @@ var JDFX_MANIFEST_PATH = "manifest.json";
|
|
|
16
23
|
var JDFX_ASSET_DIR = "assets";
|
|
17
24
|
|
|
18
25
|
// src/commands/validate.ts
|
|
19
|
-
var __dirname$1 =
|
|
26
|
+
var __dirname$1 = path7.dirname(fileURLToPath(import.meta.url));
|
|
20
27
|
function resolveSchemaPath() {
|
|
21
|
-
const bundled =
|
|
22
|
-
if (
|
|
23
|
-
const dev =
|
|
28
|
+
const bundled = path7.resolve(__dirname$1, "jdf-schema.json");
|
|
29
|
+
if (fs7.existsSync(bundled)) return bundled;
|
|
30
|
+
const dev = path7.resolve(__dirname$1, "../../../../spec/jdf-schema.json");
|
|
24
31
|
return dev;
|
|
25
32
|
}
|
|
26
33
|
var SCHEMA_PATH = resolveSchemaPath();
|
|
27
34
|
async function loadDocument(filePath) {
|
|
28
35
|
if (filePath.toLowerCase().endsWith(".jdfx")) {
|
|
29
|
-
const zip = await JSZip.loadAsync(
|
|
36
|
+
const zip = await JSZip.loadAsync(fs7.readFileSync(filePath));
|
|
30
37
|
const docFile = zip.file(JDFX_DOCUMENT_PATH);
|
|
31
38
|
if (!docFile) {
|
|
32
39
|
console.error(`\u2717 Bundle missing ${JDFX_DOCUMENT_PATH}`);
|
|
@@ -51,11 +58,11 @@ async function loadDocument(filePath) {
|
|
|
51
58
|
}
|
|
52
59
|
return { doc, bundle: { manifest, assetCount } };
|
|
53
60
|
}
|
|
54
|
-
return { doc: JSON.parse(
|
|
61
|
+
return { doc: JSON.parse(fs7.readFileSync(filePath, "utf-8")) };
|
|
55
62
|
}
|
|
56
63
|
async function validate(file) {
|
|
57
|
-
const filePath =
|
|
58
|
-
if (!
|
|
64
|
+
const filePath = path7.resolve(file);
|
|
65
|
+
if (!fs7.existsSync(filePath)) {
|
|
59
66
|
console.error(`File not found: ${filePath}`);
|
|
60
67
|
return false;
|
|
61
68
|
}
|
|
@@ -68,11 +75,11 @@ async function validate(file) {
|
|
|
68
75
|
}
|
|
69
76
|
if (!loaded) return false;
|
|
70
77
|
const { doc, bundle } = loaded;
|
|
71
|
-
if (!
|
|
78
|
+
if (!fs7.existsSync(SCHEMA_PATH)) {
|
|
72
79
|
console.error(`Schema not found at ${SCHEMA_PATH}`);
|
|
73
80
|
return false;
|
|
74
81
|
}
|
|
75
|
-
const schema = JSON.parse(
|
|
82
|
+
const schema = JSON.parse(fs7.readFileSync(SCHEMA_PATH, "utf-8"));
|
|
76
83
|
const ajv = new Ajv({ allErrors: true, strict: false });
|
|
77
84
|
addFormats(ajv);
|
|
78
85
|
const validateFn = ajv.compile(schema);
|
|
@@ -81,7 +88,7 @@ async function validate(file) {
|
|
|
81
88
|
const d = doc;
|
|
82
89
|
const pageCount = Array.isArray(d.pages) ? d.pages.length : 0;
|
|
83
90
|
const elCount = Array.isArray(d.pages) ? d.pages.reduce((acc, p) => acc + (Array.isArray(p?.elements) ? p.elements.length : 0), 0) : 0;
|
|
84
|
-
console.log(`\u2713 Valid: ${
|
|
91
|
+
console.log(`\u2713 Valid: ${path7.basename(filePath)}`);
|
|
85
92
|
console.log(` Format: ${d.$jdf}${bundle ? " (jdfx bundle)" : ""}`);
|
|
86
93
|
console.log(` Title: ${d.meta?.title}`);
|
|
87
94
|
console.log(` Pages: ${pageCount}`);
|
|
@@ -94,7 +101,7 @@ async function validate(file) {
|
|
|
94
101
|
}
|
|
95
102
|
return true;
|
|
96
103
|
}
|
|
97
|
-
console.error(`\u2717 Invalid: ${
|
|
104
|
+
console.error(`\u2717 Invalid: ${path7.basename(filePath)}`);
|
|
98
105
|
for (const err of validateFn.errors || []) {
|
|
99
106
|
const loc = err.instancePath || "(root)";
|
|
100
107
|
console.error(` ${loc} \u2014 ${err.message}`);
|
|
@@ -109,15 +116,15 @@ function hashBytes(bytes) {
|
|
|
109
116
|
return createHash("sha1").update(bytes).digest("hex").slice(0, 16);
|
|
110
117
|
}
|
|
111
118
|
function rewriteResourceRefs(doc, oldKey, newKey) {
|
|
112
|
-
function
|
|
119
|
+
function walk2(els) {
|
|
113
120
|
if (!els) return;
|
|
114
121
|
for (const el of els) {
|
|
115
122
|
if (el?.resource === oldKey) el.resource = newKey;
|
|
116
|
-
if (el?.elements)
|
|
117
|
-
if (el?.children)
|
|
123
|
+
if (el?.elements) walk2(el.elements);
|
|
124
|
+
if (el?.children) walk2(el.children);
|
|
118
125
|
}
|
|
119
126
|
}
|
|
120
|
-
for (const page of doc.pages || [])
|
|
127
|
+
for (const page of doc.pages || []) walk2(page.elements);
|
|
121
128
|
}
|
|
122
129
|
function extractAssets(doc) {
|
|
123
130
|
const assets = [];
|
|
@@ -149,10 +156,10 @@ function extractAssets(doc) {
|
|
|
149
156
|
assets.push({ id, bytes, mimeType, ext });
|
|
150
157
|
return id;
|
|
151
158
|
}
|
|
152
|
-
function
|
|
159
|
+
function walk2(els) {
|
|
153
160
|
if (!els) return;
|
|
154
161
|
for (const el of els) {
|
|
155
|
-
if (el?.type === "image" && typeof el.src === "string" && el.src.startsWith("data:")) {
|
|
162
|
+
if ((el?.type === "image" || el?.type === "video") && typeof el.src === "string" && el.src.startsWith("data:")) {
|
|
156
163
|
const m = el.src.match(/^data:([^;]+);base64,(.*)$/);
|
|
157
164
|
if (m) {
|
|
158
165
|
const mimeType = m[1];
|
|
@@ -162,19 +169,21 @@ function extractAssets(doc) {
|
|
|
162
169
|
el.resource = id;
|
|
163
170
|
}
|
|
164
171
|
}
|
|
165
|
-
if (el?.elements)
|
|
166
|
-
if (el?.children)
|
|
172
|
+
if (el?.elements) walk2(el.elements);
|
|
173
|
+
if (el?.children) walk2(el.children);
|
|
167
174
|
}
|
|
168
175
|
}
|
|
169
|
-
for (const page of cloned.pages || [])
|
|
170
|
-
|
|
171
|
-
|
|
176
|
+
for (const page of cloned.pages || []) walk2(page.elements);
|
|
177
|
+
for (const bucket of ["images", "videos"]) {
|
|
178
|
+
const store = cloned.resources?.[bucket];
|
|
179
|
+
if (!store) continue;
|
|
180
|
+
for (const [key, res] of Object.entries(store)) {
|
|
172
181
|
if (!res || typeof res !== "object" || !("data" in res) || !res.data) continue;
|
|
173
182
|
const data = String(res.data);
|
|
174
183
|
const m = data.match(/^data:([^;]+);base64,(.*)$/);
|
|
175
184
|
const b64 = m ? m[2] : data;
|
|
176
|
-
const mimeType = m ? m[1] : res.mimeType || "image/png";
|
|
177
|
-
const ext = mimeType.split("/")[1]?.replace("jpeg", "jpg") || "bin";
|
|
185
|
+
const mimeType = m ? m[1] : res.mimeType || (bucket === "videos" ? "video/mp4" : "image/png");
|
|
186
|
+
const ext = mimeType.split("/")[1]?.replace("jpeg", "jpg").replace("quicktime", "mov") || "bin";
|
|
178
187
|
const bytes = decodeBase64(b64);
|
|
179
188
|
const h = hashBytes(bytes);
|
|
180
189
|
let canonicalId = hashToId.get(h);
|
|
@@ -188,10 +197,10 @@ function extractAssets(doc) {
|
|
|
188
197
|
delete updated.data;
|
|
189
198
|
updated.src = "embedded";
|
|
190
199
|
if (canonicalId !== key) {
|
|
191
|
-
delete
|
|
200
|
+
delete store[key];
|
|
192
201
|
rewriteResourceRefs(cloned, key, canonicalId);
|
|
193
202
|
} else {
|
|
194
|
-
|
|
203
|
+
store[key] = updated;
|
|
195
204
|
}
|
|
196
205
|
}
|
|
197
206
|
}
|
|
@@ -221,21 +230,22 @@ async function packJdfx(doc) {
|
|
|
221
230
|
return { bytes, manifest };
|
|
222
231
|
}
|
|
223
232
|
function shouldUseJdfx(doc) {
|
|
224
|
-
const
|
|
225
|
-
|
|
226
|
-
|
|
233
|
+
for (const store of [doc.resources?.images ?? {}, doc.resources?.videos ?? {}]) {
|
|
234
|
+
for (const v of Object.values(store)) {
|
|
235
|
+
if (v && typeof v === "object" && "data" in v && v.data) return true;
|
|
236
|
+
}
|
|
227
237
|
}
|
|
228
|
-
function
|
|
238
|
+
function walk2(els) {
|
|
229
239
|
if (!els) return false;
|
|
230
240
|
for (const el of els) {
|
|
231
|
-
if (el?.type === "image" && typeof el.src === "string" && el.src.startsWith("data:")) return true;
|
|
232
|
-
if (el?.elements &&
|
|
233
|
-
if (el?.children &&
|
|
241
|
+
if ((el?.type === "image" || el?.type === "video") && typeof el.src === "string" && el.src.startsWith("data:")) return true;
|
|
242
|
+
if (el?.elements && walk2(el.elements)) return true;
|
|
243
|
+
if (el?.children && walk2(el.children)) return true;
|
|
234
244
|
}
|
|
235
245
|
return false;
|
|
236
246
|
}
|
|
237
247
|
for (const page of doc.pages || []) {
|
|
238
|
-
if (
|
|
248
|
+
if (walk2(page.elements)) return true;
|
|
239
249
|
}
|
|
240
250
|
return false;
|
|
241
251
|
}
|
|
@@ -294,13 +304,13 @@ function stripInline(text) {
|
|
|
294
304
|
return parseInline(text).map((r) => r.text).join("");
|
|
295
305
|
}
|
|
296
306
|
async function importMarkdown(inputPath, outputPath) {
|
|
297
|
-
const input =
|
|
307
|
+
const input = path7.resolve(inputPath);
|
|
298
308
|
console.log(`Importing: ${input}`);
|
|
299
|
-
const content =
|
|
300
|
-
const doc = convertMarkdownToJdf(content,
|
|
309
|
+
const content = fs7.readFileSync(input, "utf-8");
|
|
310
|
+
const doc = convertMarkdownToJdf(content, path7.basename(input, path7.extname(input)), path7.dirname(input));
|
|
301
311
|
let output;
|
|
302
312
|
if (outputPath) {
|
|
303
|
-
output =
|
|
313
|
+
output = path7.resolve(outputPath);
|
|
304
314
|
} else {
|
|
305
315
|
const stem = input.replace(/\.(md|markdown)$/i, "");
|
|
306
316
|
output = stem + (shouldUseJdfx(doc) ? ".jdfx" : ".jdf");
|
|
@@ -308,11 +318,11 @@ async function importMarkdown(inputPath, outputPath) {
|
|
|
308
318
|
console.log(`Output: ${output}`);
|
|
309
319
|
if (output.toLowerCase().endsWith(".jdfx")) {
|
|
310
320
|
const { bytes, manifest } = await packJdfx(doc);
|
|
311
|
-
|
|
321
|
+
fs7.writeFileSync(output, bytes);
|
|
312
322
|
console.log(`
|
|
313
323
|
Done! Created ${doc.pages.length} page(s), ${manifest.assets.length} asset(s) bundled`);
|
|
314
324
|
} else {
|
|
315
|
-
|
|
325
|
+
fs7.writeFileSync(output, JSON.stringify(doc, null, 2));
|
|
316
326
|
console.log(`
|
|
317
327
|
Done! Created ${doc.pages.length} page(s)`);
|
|
318
328
|
}
|
|
@@ -329,10 +339,10 @@ var MIME_BY_EXT2 = {
|
|
|
329
339
|
};
|
|
330
340
|
function resolveImageSrc(src, baseDir) {
|
|
331
341
|
if (/^(https?:|data:|file:)/i.test(src)) return src;
|
|
332
|
-
const abs =
|
|
342
|
+
const abs = path7.isAbsolute(src) ? src : path7.resolve(baseDir, src);
|
|
333
343
|
try {
|
|
334
|
-
const bytes =
|
|
335
|
-
const ext =
|
|
344
|
+
const bytes = fs7.readFileSync(abs);
|
|
345
|
+
const ext = path7.extname(abs).slice(1).toLowerCase();
|
|
336
346
|
const mime = MIME_BY_EXT2[ext] || "application/octet-stream";
|
|
337
347
|
return `data:${mime};base64,${bytes.toString("base64")}`;
|
|
338
348
|
} catch {
|
|
@@ -613,8 +623,232 @@ function convertMarkdownToJdf(md, title, baseDir = process.cwd()) {
|
|
|
613
623
|
};
|
|
614
624
|
}
|
|
615
625
|
|
|
616
|
-
// ../../packages/jdf-pdf-import/src/
|
|
626
|
+
// ../../packages/jdf-pdf-import/src/tables.ts
|
|
617
627
|
var PT_TO_MM = 0.352778;
|
|
628
|
+
function calibrateGlyphWidth(runs) {
|
|
629
|
+
const ks = [];
|
|
630
|
+
for (const r of runs) {
|
|
631
|
+
const t = r.text;
|
|
632
|
+
if (/\s$/.test(t) || t.trim().length < 3 || r.width <= 0) continue;
|
|
633
|
+
ks.push(r.width / (t.length * r.fontSize * PT_TO_MM));
|
|
634
|
+
}
|
|
635
|
+
if (ks.length < 3) return 0.55;
|
|
636
|
+
ks.sort((a, b) => a - b);
|
|
637
|
+
return Math.min(0.7, Math.max(0.4, ks[Math.floor(ks.length / 2)]));
|
|
638
|
+
}
|
|
639
|
+
function hasStretchedSpaces(runs, k) {
|
|
640
|
+
let n = 0, wide = 0;
|
|
641
|
+
for (const r of runs) {
|
|
642
|
+
if (!/\s$/.test(r.text) || r.text.trim().length === 0) continue;
|
|
643
|
+
const em = r.fontSize * PT_TO_MM;
|
|
644
|
+
const est = r.text.trim().length * em * k + em * 0.25;
|
|
645
|
+
n++;
|
|
646
|
+
if (r.width > est * 1.4) wide++;
|
|
647
|
+
}
|
|
648
|
+
return n >= 4 && wide / n >= 0.3;
|
|
649
|
+
}
|
|
650
|
+
function textExtent(r, k, stretched) {
|
|
651
|
+
const em = r.fontSize * PT_TO_MM;
|
|
652
|
+
if (!/\s$/.test(r.text)) return Math.max(em * 0.5, r.width);
|
|
653
|
+
const chars = Math.max(1, r.text.trim().length);
|
|
654
|
+
const est = chars * em * k + em * 0.25;
|
|
655
|
+
if (stretched) return Math.max(em * 0.5, Math.min(r.width, est));
|
|
656
|
+
return Math.max(em * 0.5, r.width > est * 1.4 ? est : r.width);
|
|
657
|
+
}
|
|
658
|
+
function groupRows(runs, skip) {
|
|
659
|
+
const k = calibrateGlyphWidth(runs);
|
|
660
|
+
const stretched = hasStretchedSpaces(runs, k);
|
|
661
|
+
const idx = runs.map((_, i) => i).filter((i) => !skip(runs[i]) && runs[i].text.trim().length > 0);
|
|
662
|
+
idx.sort((a, b) => runs[a].y - runs[b].y || runs[a].x - runs[b].x);
|
|
663
|
+
const rows = [];
|
|
664
|
+
for (const i of idx) {
|
|
665
|
+
const r = runs[i];
|
|
666
|
+
const tol = Math.max(0.8, r.fontSize * PT_TO_MM * 0.35);
|
|
667
|
+
const last = rows[rows.length - 1];
|
|
668
|
+
const cell = { run: r, idx: i, x0: r.x, x1: r.x + textExtent(r, k, stretched) };
|
|
669
|
+
if (last && Math.abs(last.y - r.y) <= tol) {
|
|
670
|
+
last.cells.push(cell);
|
|
671
|
+
last.h = Math.max(last.h, r.height);
|
|
672
|
+
} else {
|
|
673
|
+
rows.push({ y: r.y, h: r.height, cells: [cell] });
|
|
674
|
+
}
|
|
675
|
+
}
|
|
676
|
+
for (const row of rows) {
|
|
677
|
+
row.cells.sort((a, b) => a.x0 - b.x0);
|
|
678
|
+
const merged = [];
|
|
679
|
+
for (const c of row.cells) {
|
|
680
|
+
const last = merged[merged.length - 1];
|
|
681
|
+
const em = c.run.fontSize * PT_TO_MM;
|
|
682
|
+
if (last && c.x0 - last.x1 <= em * 1) {
|
|
683
|
+
last.x1 = Math.max(last.x1, c.x1);
|
|
684
|
+
last.run = { ...last.run, text: `${last.run.text.replace(/\s+$/, "")} ${c.run.text.replace(/^\s+/, "")}`, width: last.x1 - last.x0 };
|
|
685
|
+
last.extra = [...last.extra ?? [], c.idx];
|
|
686
|
+
} else merged.push({ ...c });
|
|
687
|
+
}
|
|
688
|
+
row.cells = merged;
|
|
689
|
+
}
|
|
690
|
+
return rows;
|
|
691
|
+
}
|
|
692
|
+
function columnBands(rows) {
|
|
693
|
+
const cells = rows.flatMap((r) => r.cells);
|
|
694
|
+
const sorted = cells.slice().sort((a, b) => a.x0 - b.x0);
|
|
695
|
+
const bands = [];
|
|
696
|
+
for (const c of sorted) {
|
|
697
|
+
const last = bands[bands.length - 1];
|
|
698
|
+
if (last && c.x0 <= last.x1 - 0.2) {
|
|
699
|
+
last.x1 = Math.max(last.x1, c.x1);
|
|
700
|
+
last.members.push(c);
|
|
701
|
+
} else bands.push({ x0: c.x0, x1: c.x1, members: [c] });
|
|
702
|
+
}
|
|
703
|
+
for (const b of bands) {
|
|
704
|
+
const rowsSeen = /* @__PURE__ */ new Set();
|
|
705
|
+
for (const m of b.members) {
|
|
706
|
+
const row = rows.find((r) => r.cells.includes(m));
|
|
707
|
+
if (rowsSeen.has(row)) return null;
|
|
708
|
+
rowsSeen.add(row);
|
|
709
|
+
}
|
|
710
|
+
}
|
|
711
|
+
return bands.map(({ x0, x1 }) => ({ x0, x1 }));
|
|
712
|
+
}
|
|
713
|
+
var numeric = (s) => /^[\s$€£¥+\-−–]*[\d.,]+\s*(%|ms|s|k|m|b|M|K|B|x|×)?\s*(\/\w+)?$/i.test(s.trim()) || /^[+\-−]?\d/.test(s.trim()) && /\d$/.test(s.trim().replace(/[%)]$/, ""));
|
|
714
|
+
function detectTables(runs, shapes, pageWidthMm) {
|
|
715
|
+
const out = [];
|
|
716
|
+
const used = /* @__PURE__ */ new Set();
|
|
717
|
+
const rows = groupRows(runs, () => false);
|
|
718
|
+
const pageW = pageWidthMm;
|
|
719
|
+
let i = 0;
|
|
720
|
+
while (i < rows.length) {
|
|
721
|
+
if (rows[i].cells.length < 2) {
|
|
722
|
+
i++;
|
|
723
|
+
continue;
|
|
724
|
+
}
|
|
725
|
+
let j = i;
|
|
726
|
+
let cur = columnBands([rows[i]]);
|
|
727
|
+
let best = null;
|
|
728
|
+
while (j + 1 < rows.length && cur) {
|
|
729
|
+
const next = rows[j + 1];
|
|
730
|
+
const gap = next.y - (rows[j].y + rows[j].h);
|
|
731
|
+
const rowH = Math.max(rows[j].h, next.h);
|
|
732
|
+
if (gap > rowH * 2.2) break;
|
|
733
|
+
if (next.cells.length === 1) {
|
|
734
|
+
const c = next.cells[0];
|
|
735
|
+
const inBand = cur.findIndex((b) => c.x0 < b.x1 - 0.2 && c.x1 > b.x0 + 0.2);
|
|
736
|
+
const spansSeveral = cur.filter((b) => c.x0 < b.x1 - 0.2 && c.x1 > b.x0 + 0.2).length > 1;
|
|
737
|
+
if (inBand <= 0 || spansSeveral) break;
|
|
738
|
+
j++;
|
|
739
|
+
continue;
|
|
740
|
+
}
|
|
741
|
+
const nb = columnBands(rows.slice(i, j + 2));
|
|
742
|
+
if (!nb || nb.length < 2) break;
|
|
743
|
+
if (nb.length > cur.length && j - i >= 2) break;
|
|
744
|
+
cur = nb;
|
|
745
|
+
j++;
|
|
746
|
+
if (cur.length >= 2) best = { j, bands: cur };
|
|
747
|
+
}
|
|
748
|
+
const multiRows = best ? rows.slice(i, best.j + 1).filter((r) => r.cells.length >= 2).length : 0;
|
|
749
|
+
const blockRows = best ? rows.slice(i, best.j + 1) : [];
|
|
750
|
+
const bbox = blockRows.length ? {
|
|
751
|
+
x0: Math.min(...best.bands.map((b) => b.x0)) - 5,
|
|
752
|
+
x1: Math.max(...best.bands.map((b) => b.x1)) + 5,
|
|
753
|
+
// Cell padding puts backgrounds/borders well above the first baseline and below the last.
|
|
754
|
+
y0: blockRows[0].y - blockRows[0].h * 2.5,
|
|
755
|
+
y1: blockRows[blockRows.length - 1].y + blockRows[blockRows.length - 1].h * 3
|
|
756
|
+
} : null;
|
|
757
|
+
const gridShapes = bbox ? shapes.map((s, k) => ({ s, k })).filter(({ s }) => s.x >= bbox.x0 - 1 && s.x + s.width <= bbox.x1 + 1 && s.y >= bbox.y0 - 1 && s.y + s.height <= bbox.y1 + 1 && (s.kind === "line" || s.kind === "rect")) : [];
|
|
758
|
+
const hasLattice = gridShapes.length >= 3;
|
|
759
|
+
if (!best || multiRows < (hasLattice ? 2 : 3) || best.bands.length < 2) {
|
|
760
|
+
i++;
|
|
761
|
+
continue;
|
|
762
|
+
}
|
|
763
|
+
const bands = best.bands;
|
|
764
|
+
const cellText = (row, b) => row.cells.filter((c) => c.x0 < bands[b].x1 - 0.2 && c.x1 > bands[b].x0 + 0.2).map((c) => c.run.text.trim()).join(" ").trim();
|
|
765
|
+
const grid = [];
|
|
766
|
+
const lineIdx = [];
|
|
767
|
+
for (const row of blockRows) {
|
|
768
|
+
if (row.cells.length === 1 && grid.length) {
|
|
769
|
+
const c = row.cells[0];
|
|
770
|
+
const b = bands.findIndex((bb) => c.x0 < bb.x1 - 0.2 && c.x1 > bb.x0 + 0.2);
|
|
771
|
+
if (b > 0) {
|
|
772
|
+
grid[grid.length - 1][b] = (grid[grid.length - 1][b] + " " + c.run.text.trim()).trim();
|
|
773
|
+
lineIdx.push(c.idx, ...c.extra ?? []);
|
|
774
|
+
continue;
|
|
775
|
+
}
|
|
776
|
+
}
|
|
777
|
+
grid.push(bands.map((_, b) => cellText(row, b)));
|
|
778
|
+
for (const c of row.cells) {
|
|
779
|
+
lineIdx.push(c.idx);
|
|
780
|
+
for (const k of c.extra ?? []) lineIdx.push(k);
|
|
781
|
+
}
|
|
782
|
+
}
|
|
783
|
+
if (lineIdx.some((k) => used.has(k))) {
|
|
784
|
+
i = best.j + 1;
|
|
785
|
+
continue;
|
|
786
|
+
}
|
|
787
|
+
const first = blockRows[0];
|
|
788
|
+
const tableW = bbox.x1 - bbox.x0;
|
|
789
|
+
const rowFill = (row) => {
|
|
790
|
+
const cy = row.y + row.h * 0.5;
|
|
791
|
+
const rects = gridShapes.filter(({ s }) => s.kind === "rect" && s.fill && s.fill.toLowerCase() !== "#ffffff" && s.height >= row.h * 0.6 && s.height < row.h * 4.5 && s.y <= cy && s.y + s.height >= cy);
|
|
792
|
+
const covered = rects.reduce((a, { s }) => a + s.width, 0);
|
|
793
|
+
if (!rects.length || covered < tableW * 0.5) return null;
|
|
794
|
+
const counts = /* @__PURE__ */ new Map();
|
|
795
|
+
for (const { s } of rects) counts.set(s.fill, (counts.get(s.fill) ?? 0) + s.width);
|
|
796
|
+
const fill = [...counts.entries()].sort((a, b) => b[1] - a[1])[0][0];
|
|
797
|
+
return { fill, rects };
|
|
798
|
+
};
|
|
799
|
+
const headerBg = rowFill(first);
|
|
800
|
+
const firstBold = first.cells.every((c) => c.run.bold);
|
|
801
|
+
const bodyFills = blockRows.slice(1).map(rowFill);
|
|
802
|
+
const headerDistinct = !!headerBg && !bodyFills.every((f) => f?.fill === headerBg.fill);
|
|
803
|
+
const isHeader = headerDistinct || firstBold && !blockRows.slice(1).every((r) => r.cells.every((c) => c.run.bold));
|
|
804
|
+
const columns = bands.map((b, k) => {
|
|
805
|
+
const vals = grid.slice(isHeader ? 1 : 0).map((r) => r[k]).filter(Boolean);
|
|
806
|
+
const numericShare = vals.length ? vals.filter(numeric).length / vals.length : 0;
|
|
807
|
+
const col = { width: Math.round((b.x1 - b.x0) * 10) / 10 };
|
|
808
|
+
if (numericShare >= 0.7) col.align = "right";
|
|
809
|
+
return col;
|
|
810
|
+
});
|
|
811
|
+
const x0 = Math.max(0, bands[0].x0 - 2.5);
|
|
812
|
+
const x1 = Math.min(pageW, bands[bands.length - 1].x1 + 2.5);
|
|
813
|
+
for (let k = 0; k < bands.length; k++) {
|
|
814
|
+
const left = k === 0 ? x0 : (bands[k - 1].x1 + bands[k].x0) / 2;
|
|
815
|
+
const right = k === bands.length - 1 ? x1 : (bands[k].x1 + bands[k + 1].x0) / 2;
|
|
816
|
+
columns[k].width = Math.round((right - left) * 10) / 10;
|
|
817
|
+
}
|
|
818
|
+
const rowFills = (isHeader ? bodyFills : [headerBg, ...bodyFills]).map((f) => f?.fill ?? null);
|
|
819
|
+
const odd = rowFills.filter((_, k) => k % 2 === 1), even = rowFills.filter((_, k) => k % 2 === 0);
|
|
820
|
+
const altColor = odd.length && odd[0] && odd.every((f) => f === odd[0]) && even.every((f) => f !== odd[0]) ? odd[0] : void 0;
|
|
821
|
+
const borderShape = gridShapes.find(({ s }) => s.kind === "line" || s.kind === "rect" && (s.height < 0.6 || s.width < 0.6) && (s.fill || s.stroke));
|
|
822
|
+
const borderColor = borderShape ? borderShape.s.stroke || borderShape.s.fill : void 0;
|
|
823
|
+
const fontSize = Math.round(first.cells[0].run.fontSize * 10) / 10;
|
|
824
|
+
const y0 = headerBg ? Math.min(...headerBg.rects.map(({ s }) => s.y)) : first.y - first.h * 0.5;
|
|
825
|
+
const element = {
|
|
826
|
+
type: "table",
|
|
827
|
+
position: { x: Math.round(x0 * 100) / 100, y: Math.round(Math.max(0, y0) * 100) / 100 },
|
|
828
|
+
width: Math.round((x1 - x0) * 100) / 100,
|
|
829
|
+
columns,
|
|
830
|
+
rows: isHeader ? grid.slice(1) : grid,
|
|
831
|
+
style: { fontSize }
|
|
832
|
+
};
|
|
833
|
+
if (isHeader) {
|
|
834
|
+
element.headers = grid[0];
|
|
835
|
+
const hs = { fontWeight: "bold" };
|
|
836
|
+
if (headerBg) hs.backgroundColor = headerBg.fill;
|
|
837
|
+
const hc = first.cells[0].run.color;
|
|
838
|
+
if (hc && hc !== "#000000") hs.color = hc;
|
|
839
|
+
element.headerStyle = hs;
|
|
840
|
+
}
|
|
841
|
+
if (altColor) element.alternatingRowColor = altColor;
|
|
842
|
+
element.borders = borderColor ? { outer: true, inner: true, color: borderColor, width: 1 } : false;
|
|
843
|
+
for (const k of lineIdx) used.add(k);
|
|
844
|
+
out.push({ element, lineIdx, shapeIdx: gridShapes.map(({ k }) => k) });
|
|
845
|
+
i = best.j + 1;
|
|
846
|
+
}
|
|
847
|
+
return out;
|
|
848
|
+
}
|
|
849
|
+
|
|
850
|
+
// ../../packages/jdf-pdf-import/src/core.ts
|
|
851
|
+
var PT_TO_MM2 = 0.352778;
|
|
618
852
|
function classifyFont(name) {
|
|
619
853
|
const n = (name || "").toLowerCase();
|
|
620
854
|
const bold = /bold|black|heavy|semibold|demibold|extrabold/.test(n);
|
|
@@ -640,6 +874,38 @@ function rgbToHex(r, g, b) {
|
|
|
640
874
|
const h = (n) => clampByte(n).toString(16).padStart(2, "0");
|
|
641
875
|
return `#${h(r)}${h(g)}${h(b)}`;
|
|
642
876
|
}
|
|
877
|
+
function averageStops(stops) {
|
|
878
|
+
if (!Array.isArray(stops) || stops.length === 0) return null;
|
|
879
|
+
let r = 0, g = 0, b = 0, n = 0;
|
|
880
|
+
for (const st of stops) {
|
|
881
|
+
const css = Array.isArray(st) ? st[1] : null;
|
|
882
|
+
const m = typeof css === "string" && css.match(/^#([0-9a-f]{2})([0-9a-f]{2})([0-9a-f]{2})$/i);
|
|
883
|
+
if (!m) continue;
|
|
884
|
+
r += parseInt(m[1], 16);
|
|
885
|
+
g += parseInt(m[2], 16);
|
|
886
|
+
b += parseInt(m[3], 16);
|
|
887
|
+
n++;
|
|
888
|
+
}
|
|
889
|
+
return n ? rgbToHex(r / n, g / n, b / n) : null;
|
|
890
|
+
}
|
|
891
|
+
function patternToColor(page, arg) {
|
|
892
|
+
if (!Array.isArray(arg)) return null;
|
|
893
|
+
if (arg[0] === "TilingPattern") {
|
|
894
|
+
const c = arg[1];
|
|
895
|
+
return Array.isArray(c) && c.length >= 3 ? rgbToHex(c[0], c[1], c[2]) : null;
|
|
896
|
+
}
|
|
897
|
+
if (arg[0] === "Shading") {
|
|
898
|
+
const id = arg[1];
|
|
899
|
+
try {
|
|
900
|
+
const store = typeof id === "string" && id.startsWith("g_") ? page.commonObjs : page.objs;
|
|
901
|
+
if (typeof store?.has === "function" && !store.has(id)) return null;
|
|
902
|
+
const ir = store.get(id);
|
|
903
|
+
if (Array.isArray(ir) && ir[0] === "RadialAxial") return averageStops(ir[3]);
|
|
904
|
+
} catch {
|
|
905
|
+
}
|
|
906
|
+
}
|
|
907
|
+
return null;
|
|
908
|
+
}
|
|
643
909
|
function multiplyCtm(a, b) {
|
|
644
910
|
return [
|
|
645
911
|
a[0] * b[0] + a[1] * b[2],
|
|
@@ -658,7 +924,7 @@ async function walkOps(page, OPS, viewport) {
|
|
|
658
924
|
const [vx, vy] = viewport.convertToViewportPoint(x, y);
|
|
659
925
|
return { x: vx, y: vy };
|
|
660
926
|
};
|
|
661
|
-
const opList = await page.getOperatorList();
|
|
927
|
+
const opList = await page.getOperatorList({ annotationMode: annotationModeDisable });
|
|
662
928
|
const fnArr = opList.fnArray;
|
|
663
929
|
const argsArr = opList.argsArray;
|
|
664
930
|
const gs = {
|
|
@@ -668,20 +934,60 @@ async function walkOps(page, OPS, viewport) {
|
|
|
668
934
|
lineWidth: 1,
|
|
669
935
|
fillAlpha: 1,
|
|
670
936
|
strokeAlpha: 1,
|
|
671
|
-
textRenderingMode: 0
|
|
937
|
+
textRenderingMode: 0,
|
|
938
|
+
// Text state (PDF 9.3) — needed to know where each showText lands.
|
|
939
|
+
fontSize: 0,
|
|
940
|
+
charSpacing: 0,
|
|
941
|
+
wordSpacing: 0,
|
|
942
|
+
hscale: 1,
|
|
943
|
+
leading: 0,
|
|
944
|
+
rise: 0
|
|
672
945
|
};
|
|
946
|
+
const snapshot = () => ({ ...gs, ctm: [...gs.ctm] });
|
|
673
947
|
const stack = [];
|
|
674
|
-
const
|
|
675
|
-
const textOpacities = [];
|
|
676
|
-
const textRenderingModes = [];
|
|
948
|
+
const textOps = [];
|
|
677
949
|
const shapes = [];
|
|
678
950
|
const imagePositions = [];
|
|
679
|
-
let
|
|
951
|
+
let tm = [1, 0, 0, 1, 0, 0];
|
|
952
|
+
let tlm = [1, 0, 0, 1, 0, 0];
|
|
953
|
+
function pushImage(name, ctm, inline, maskFill) {
|
|
954
|
+
const corners = [tx(ctm, 0, 0), tx(ctm, 1, 0), tx(ctm, 1, 1), tx(ctm, 0, 1)];
|
|
955
|
+
const vpCorners = corners.map((p) => toViewport(p.x, p.y));
|
|
956
|
+
const xs = vpCorners.map((p) => p.x), ys = vpCorners.map((p) => p.y);
|
|
957
|
+
const minX = Math.min(...xs), maxX = Math.max(...xs);
|
|
958
|
+
const minY = Math.min(...ys), maxY = Math.max(...ys);
|
|
959
|
+
imagePositions.push({
|
|
960
|
+
name,
|
|
961
|
+
x: minX * PT_TO_MM2,
|
|
962
|
+
y: minY * PT_TO_MM2,
|
|
963
|
+
w: (maxX - minX) * PT_TO_MM2,
|
|
964
|
+
h: (maxY - minY) * PT_TO_MM2,
|
|
965
|
+
inline,
|
|
966
|
+
maskFill
|
|
967
|
+
});
|
|
968
|
+
}
|
|
680
969
|
let pathSegments = [];
|
|
681
970
|
let pathRect = null;
|
|
971
|
+
let pathRects = [];
|
|
682
972
|
let pathStart = null;
|
|
683
973
|
let pathLast = null;
|
|
684
974
|
function flushPath(isFill, isStroke) {
|
|
975
|
+
for (const r of pathRects) {
|
|
976
|
+
const tl = toViewport(r.x, r.y + r.h);
|
|
977
|
+
const br = toViewport(r.x + r.w, r.y);
|
|
978
|
+
shapes.push({
|
|
979
|
+
kind: "rect",
|
|
980
|
+
x: Math.min(tl.x, br.x) * PT_TO_MM2,
|
|
981
|
+
y: Math.min(tl.y, br.y) * PT_TO_MM2,
|
|
982
|
+
width: Math.abs(br.x - tl.x) * PT_TO_MM2,
|
|
983
|
+
height: Math.abs(br.y - tl.y) * PT_TO_MM2,
|
|
984
|
+
fill: isFill ? gs.fill : void 0,
|
|
985
|
+
stroke: isStroke ? gs.stroke : void 0,
|
|
986
|
+
strokeWidth: isStroke ? gs.lineWidth * PT_TO_MM2 : void 0,
|
|
987
|
+
opacity: isFill ? gs.fillAlpha : gs.strokeAlpha
|
|
988
|
+
});
|
|
989
|
+
}
|
|
990
|
+
pathRects = [];
|
|
685
991
|
if (pathRect) {
|
|
686
992
|
const tl = toViewport(pathRect.x, pathRect.y + pathRect.h);
|
|
687
993
|
const br = toViewport(pathRect.x + pathRect.w, pathRect.y);
|
|
@@ -691,13 +997,13 @@ async function walkOps(page, OPS, viewport) {
|
|
|
691
997
|
const h = Math.abs(br.y - tl.y);
|
|
692
998
|
shapes.push({
|
|
693
999
|
kind: "rect",
|
|
694
|
-
x: x *
|
|
695
|
-
y: y *
|
|
696
|
-
width: w *
|
|
697
|
-
height: h *
|
|
1000
|
+
x: x * PT_TO_MM2,
|
|
1001
|
+
y: y * PT_TO_MM2,
|
|
1002
|
+
width: w * PT_TO_MM2,
|
|
1003
|
+
height: h * PT_TO_MM2,
|
|
698
1004
|
fill: isFill ? gs.fill : void 0,
|
|
699
1005
|
stroke: isStroke ? gs.stroke : void 0,
|
|
700
|
-
strokeWidth: isStroke ? gs.lineWidth *
|
|
1006
|
+
strokeWidth: isStroke ? gs.lineWidth * PT_TO_MM2 : void 0,
|
|
701
1007
|
opacity: isFill ? gs.fillAlpha : gs.strokeAlpha
|
|
702
1008
|
});
|
|
703
1009
|
} else if (pathSegments.length === 2 && pathSegments[0].type === "M" && pathSegments[1].type === "L") {
|
|
@@ -709,35 +1015,35 @@ async function walkOps(page, OPS, viewport) {
|
|
|
709
1015
|
const minY = Math.min(va.y, vb.y);
|
|
710
1016
|
const maxX = Math.max(va.x, vb.x);
|
|
711
1017
|
const maxY = Math.max(va.y, vb.y);
|
|
712
|
-
const x1Local = (va.x - minX) *
|
|
713
|
-
const y1Local = (va.y - minY) *
|
|
714
|
-
const x2Local = (vb.x - minX) *
|
|
715
|
-
const y2Local = (vb.y - minY) *
|
|
716
|
-
const wLocal = Math.max(0.05, (maxX - minX) *
|
|
717
|
-
const hLocal = Math.max(0.05, (maxY - minY) *
|
|
1018
|
+
const x1Local = (va.x - minX) * PT_TO_MM2;
|
|
1019
|
+
const y1Local = (va.y - minY) * PT_TO_MM2;
|
|
1020
|
+
const x2Local = (vb.x - minX) * PT_TO_MM2;
|
|
1021
|
+
const y2Local = (vb.y - minY) * PT_TO_MM2;
|
|
1022
|
+
const wLocal = Math.max(0.05, (maxX - minX) * PT_TO_MM2);
|
|
1023
|
+
const hLocal = Math.max(0.05, (maxY - minY) * PT_TO_MM2);
|
|
718
1024
|
const dx = Math.abs(va.x - vb.x);
|
|
719
1025
|
const dy = Math.abs(va.y - vb.y);
|
|
720
1026
|
const axisAligned = dx < 0.5 || dy < 0.5;
|
|
721
1027
|
if (axisAligned) {
|
|
722
1028
|
shapes.push({
|
|
723
1029
|
kind: "line",
|
|
724
|
-
x: minX *
|
|
725
|
-
y: minY *
|
|
1030
|
+
x: minX * PT_TO_MM2,
|
|
1031
|
+
y: minY * PT_TO_MM2,
|
|
726
1032
|
width: wLocal,
|
|
727
1033
|
height: hLocal,
|
|
728
1034
|
stroke: isStroke ? gs.stroke : void 0,
|
|
729
|
-
strokeWidth: isStroke ? gs.lineWidth *
|
|
1035
|
+
strokeWidth: isStroke ? gs.lineWidth * PT_TO_MM2 : void 0,
|
|
730
1036
|
opacity: gs.strokeAlpha
|
|
731
1037
|
});
|
|
732
1038
|
} else {
|
|
733
1039
|
shapes.push({
|
|
734
1040
|
kind: "path",
|
|
735
|
-
x: minX *
|
|
736
|
-
y: minY *
|
|
1041
|
+
x: minX * PT_TO_MM2,
|
|
1042
|
+
y: minY * PT_TO_MM2,
|
|
737
1043
|
width: wLocal,
|
|
738
1044
|
height: hLocal,
|
|
739
1045
|
stroke: isStroke ? gs.stroke : void 0,
|
|
740
|
-
strokeWidth: isStroke ? gs.lineWidth *
|
|
1046
|
+
strokeWidth: isStroke ? gs.lineWidth * PT_TO_MM2 : void 0,
|
|
741
1047
|
opacity: gs.strokeAlpha,
|
|
742
1048
|
path: `M ${x1Local.toFixed(2)} ${y1Local.toFixed(2)} L ${x2Local.toFixed(2)} ${y2Local.toFixed(2)}`
|
|
743
1049
|
});
|
|
@@ -770,20 +1076,20 @@ async function walkOps(page, OPS, viewport) {
|
|
|
770
1076
|
if (seg.type === "Z") return "Z";
|
|
771
1077
|
const p = [];
|
|
772
1078
|
for (let i = 0; i < seg.pts.length; i += 2) {
|
|
773
|
-
p.push(((seg.pts[i] - minX) *
|
|
774
|
-
p.push(((seg.pts[i + 1] - minY) *
|
|
1079
|
+
p.push(((seg.pts[i] - minX) * PT_TO_MM2).toFixed(2));
|
|
1080
|
+
p.push(((seg.pts[i + 1] - minY) * PT_TO_MM2).toFixed(2));
|
|
775
1081
|
}
|
|
776
1082
|
return `${seg.type} ${p.join(" ")}`;
|
|
777
1083
|
}).join(" ");
|
|
778
1084
|
shapes.push({
|
|
779
1085
|
kind: "path",
|
|
780
|
-
x: minX *
|
|
781
|
-
y: minY *
|
|
782
|
-
width: bw *
|
|
783
|
-
height: bh *
|
|
1086
|
+
x: minX * PT_TO_MM2,
|
|
1087
|
+
y: minY * PT_TO_MM2,
|
|
1088
|
+
width: bw * PT_TO_MM2,
|
|
1089
|
+
height: bh * PT_TO_MM2,
|
|
784
1090
|
fill: isFill ? gs.fill : void 0,
|
|
785
1091
|
stroke: isStroke ? gs.stroke : void 0,
|
|
786
|
-
strokeWidth: isStroke ? gs.lineWidth *
|
|
1092
|
+
strokeWidth: isStroke ? gs.lineWidth * PT_TO_MM2 : void 0,
|
|
787
1093
|
opacity: isFill ? gs.fillAlpha : gs.strokeAlpha,
|
|
788
1094
|
path: d
|
|
789
1095
|
});
|
|
@@ -798,10 +1104,51 @@ async function walkOps(page, OPS, viewport) {
|
|
|
798
1104
|
const fn = fnArr[i];
|
|
799
1105
|
const args = argsArr[i] || [];
|
|
800
1106
|
if (fn === OPS.save) {
|
|
801
|
-
stack.push(
|
|
1107
|
+
stack.push(snapshot());
|
|
802
1108
|
} else if (fn === OPS.restore) {
|
|
803
1109
|
const s = stack.pop();
|
|
804
1110
|
if (s) Object.assign(gs, s);
|
|
1111
|
+
} else if (fn === OPS.paintFormXObjectBegin) {
|
|
1112
|
+
stack.push(snapshot());
|
|
1113
|
+
const matrix = args[0];
|
|
1114
|
+
if (Array.isArray(matrix) && matrix.length === 6) gs.ctm = multiplyCtm(matrix, gs.ctm);
|
|
1115
|
+
} else if (fn === OPS.paintFormXObjectEnd) {
|
|
1116
|
+
const s = stack.pop();
|
|
1117
|
+
if (s) Object.assign(gs, s);
|
|
1118
|
+
} else if (fn === OPS.beginText) {
|
|
1119
|
+
tm = [1, 0, 0, 1, 0, 0];
|
|
1120
|
+
tlm = [1, 0, 0, 1, 0, 0];
|
|
1121
|
+
} else if (fn === OPS.setTextMatrix) {
|
|
1122
|
+
tm = [...args];
|
|
1123
|
+
tlm = [...tm];
|
|
1124
|
+
} else if (fn === OPS.moveText) {
|
|
1125
|
+
tlm = multiplyCtm([1, 0, 0, 1, args[0], args[1]], tlm);
|
|
1126
|
+
tm = [...tlm];
|
|
1127
|
+
} else if (fn === OPS.setLeadingMoveText) {
|
|
1128
|
+
gs.leading = -args[1];
|
|
1129
|
+
tlm = multiplyCtm([1, 0, 0, 1, args[0], args[1]], tlm);
|
|
1130
|
+
tm = [...tlm];
|
|
1131
|
+
} else if (fn === OPS.nextLine) {
|
|
1132
|
+
tlm = multiplyCtm([1, 0, 0, 1, 0, -gs.leading], tlm);
|
|
1133
|
+
tm = [...tlm];
|
|
1134
|
+
} else if (fn === OPS.setLeading) {
|
|
1135
|
+
gs.leading = args[0];
|
|
1136
|
+
} else if (fn === OPS.setFont) {
|
|
1137
|
+
gs.fontSize = typeof args[1] === "number" ? args[1] : gs.fontSize;
|
|
1138
|
+
} else if (fn === OPS.setCharSpacing) {
|
|
1139
|
+
gs.charSpacing = args[0];
|
|
1140
|
+
} else if (fn === OPS.setWordSpacing) {
|
|
1141
|
+
gs.wordSpacing = args[0];
|
|
1142
|
+
} else if (fn === OPS.setHScale) {
|
|
1143
|
+
gs.hscale = (args[0] ?? 100) / 100;
|
|
1144
|
+
} else if (fn === OPS.setTextRise) {
|
|
1145
|
+
gs.rise = args[0];
|
|
1146
|
+
} else if (fn === OPS.setFillColorN) {
|
|
1147
|
+
const c = patternToColor(page, args[0]);
|
|
1148
|
+
if (c) gs.fill = c;
|
|
1149
|
+
} else if (fn === OPS.setStrokeColorN) {
|
|
1150
|
+
const c = patternToColor(page, args[0]);
|
|
1151
|
+
if (c) gs.stroke = c;
|
|
805
1152
|
} else if (fn === OPS.transform) {
|
|
806
1153
|
gs.ctm = multiplyCtm(args, gs.ctm);
|
|
807
1154
|
} else if (fn === OPS.setFillRGBColor) {
|
|
@@ -835,11 +1182,23 @@ async function walkOps(page, OPS, viewport) {
|
|
|
835
1182
|
else if (key === "CA") gs.strokeAlpha = val;
|
|
836
1183
|
}
|
|
837
1184
|
}
|
|
838
|
-
} else if (fn === OPS.showText
|
|
839
|
-
|
|
840
|
-
|
|
841
|
-
|
|
842
|
-
|
|
1185
|
+
} else if (fn === OPS.showText) {
|
|
1186
|
+
const trm = multiplyCtm(tm, gs.ctm);
|
|
1187
|
+
const origin = tx(trm, 0, gs.rise);
|
|
1188
|
+
const vp = toViewport(origin.x, origin.y);
|
|
1189
|
+
const scale = Math.hypot(trm[2], trm[3]) || 1;
|
|
1190
|
+
textOps.push({ x: vp.x, y: vp.y, fontSize: gs.fontSize * scale, fill: gs.fill, alpha: gs.fillAlpha, mode: gs.textRenderingMode });
|
|
1191
|
+
let advance = 0;
|
|
1192
|
+
const glyphs = Array.isArray(args[0]) ? args[0] : [];
|
|
1193
|
+
for (const g of glyphs) {
|
|
1194
|
+
if (typeof g === "number") {
|
|
1195
|
+
advance += -g / 1e3 * gs.fontSize * gs.hscale;
|
|
1196
|
+
} else if (g && typeof g === "object") {
|
|
1197
|
+
const w = typeof g.width === "number" ? g.width : 0;
|
|
1198
|
+
advance += (w / 1e3 * gs.fontSize + gs.charSpacing + (g.isSpace ? gs.wordSpacing : 0)) * gs.hscale;
|
|
1199
|
+
}
|
|
1200
|
+
}
|
|
1201
|
+
tm = multiplyCtm([1, 0, 0, 1, advance, 0], tm);
|
|
843
1202
|
} else if (fn === OPS.rectangle) {
|
|
844
1203
|
const [x, y, w, h] = args;
|
|
845
1204
|
const p1 = tx(gs.ctm, x, y);
|
|
@@ -892,6 +1251,12 @@ async function walkOps(page, OPS, viewport) {
|
|
|
892
1251
|
} else if (op === OPS.closePath) {
|
|
893
1252
|
pathSegments.push({ type: "Z", pts: [] });
|
|
894
1253
|
if (pathStart) pathLast = { ...pathStart };
|
|
1254
|
+
} else if (op === OPS.rectangle) {
|
|
1255
|
+
const x = pathArgs[ai], y = pathArgs[ai + 1], w = pathArgs[ai + 2], h = pathArgs[ai + 3];
|
|
1256
|
+
ai += 4;
|
|
1257
|
+
const p1 = tx(gs.ctm, x, y);
|
|
1258
|
+
const p3 = tx(gs.ctm, x + w, y + h);
|
|
1259
|
+
pathRects.push({ x: Math.min(p1.x, p3.x), y: Math.min(p1.y, p3.y), w: Math.abs(p3.x - p1.x), h: Math.abs(p3.y - p1.y) });
|
|
895
1260
|
}
|
|
896
1261
|
}
|
|
897
1262
|
} else if (fn === OPS.fill || fn === OPS.stroke || fn === OPS.fillStroke || fn === OPS.eoFill || fn === OPS.eoFillStroke || fn === OPS.closeFillStroke || fn === OPS.closeStroke || fn === OPS.closeEOFillStroke) {
|
|
@@ -900,59 +1265,128 @@ async function walkOps(page, OPS, viewport) {
|
|
|
900
1265
|
flushPath(isFill, isStroke);
|
|
901
1266
|
} else if (fn === OPS.endPath || fn === OPS.clip || fn === OPS.eoClip) {
|
|
902
1267
|
pathSegments = [];
|
|
1268
|
+
pathRects = [];
|
|
903
1269
|
pathRect = null;
|
|
904
1270
|
pathStart = null;
|
|
905
1271
|
pathLast = null;
|
|
906
|
-
} else if (fn === OPS.paintImageXObject
|
|
907
|
-
|
|
908
|
-
|
|
909
|
-
const
|
|
910
|
-
|
|
911
|
-
|
|
912
|
-
|
|
913
|
-
|
|
914
|
-
|
|
915
|
-
|
|
916
|
-
|
|
917
|
-
|
|
918
|
-
|
|
919
|
-
|
|
920
|
-
});
|
|
1272
|
+
} else if (fn === OPS.paintImageXObject) {
|
|
1273
|
+
pushImage(String(args[0]), gs.ctm);
|
|
1274
|
+
} else if (fn === OPS.paintImageXObjectRepeat) {
|
|
1275
|
+
const [name, scaleX, scaleY, positions] = args;
|
|
1276
|
+
if (Array.isArray(positions)) {
|
|
1277
|
+
for (let k = 0; k + 1 < positions.length; k += 2) {
|
|
1278
|
+
pushImage(String(name), multiplyCtm([scaleX, 0, 0, scaleY, positions[k], positions[k + 1]], gs.ctm));
|
|
1279
|
+
}
|
|
1280
|
+
}
|
|
1281
|
+
} else if (fn === OPS.paintInlineImageXObject) {
|
|
1282
|
+
const img = args[0];
|
|
1283
|
+
if (img && typeof img === "object") pushImage(`inline-${imagePositions.length}`, gs.ctm, img);
|
|
1284
|
+
} else if (fn === OPS.paintImageMaskXObject) {
|
|
1285
|
+
const img = args[0];
|
|
1286
|
+
if (img && typeof img === "object") pushImage(`mask-${imagePositions.length}`, gs.ctm, img, gs.fill);
|
|
1287
|
+
} else if (fn === OPS.paintImageMaskXObjectRepeat) {
|
|
1288
|
+
const [img, scaleX, skewX, skewY, scaleY, positions] = args;
|
|
1289
|
+
if (img && typeof img === "object" && Array.isArray(positions)) {
|
|
1290
|
+
for (let k = 0; k + 1 < positions.length; k += 2) {
|
|
1291
|
+
pushImage(`mask-${imagePositions.length}`, multiplyCtm([scaleX, skewX, skewY, scaleY, positions[k], positions[k + 1]], gs.ctm), img, gs.fill);
|
|
1292
|
+
}
|
|
1293
|
+
}
|
|
1294
|
+
} else if (fn === OPS.paintImageMaskXObjectGroup) {
|
|
1295
|
+
const group = args[0];
|
|
1296
|
+
if (Array.isArray(group)) {
|
|
1297
|
+
for (const img of group) {
|
|
1298
|
+
if (img && typeof img === "object" && Array.isArray(img.transform)) {
|
|
1299
|
+
pushImage(`mask-${imagePositions.length}`, multiplyCtm(img.transform, gs.ctm), img, gs.fill);
|
|
1300
|
+
}
|
|
1301
|
+
}
|
|
1302
|
+
}
|
|
921
1303
|
}
|
|
922
1304
|
}
|
|
923
|
-
return {
|
|
1305
|
+
return { textOps, shapes, imagePositions };
|
|
924
1306
|
}
|
|
925
1307
|
async function extractImages(page, positions, runtime, dataUrlCache) {
|
|
1308
|
+
const resolveObj = (id) => new Promise((resolve) => {
|
|
1309
|
+
let settled2 = false;
|
|
1310
|
+
const timer = setTimeout(() => done(null), 250);
|
|
1311
|
+
const done = (v) => {
|
|
1312
|
+
if (settled2) return;
|
|
1313
|
+
settled2 = true;
|
|
1314
|
+
clearTimeout(timer);
|
|
1315
|
+
resolve(v);
|
|
1316
|
+
};
|
|
1317
|
+
try {
|
|
1318
|
+
const store = id.startsWith("g_") ? page.commonObjs : page.objs;
|
|
1319
|
+
store.get(id, (img) => done(img));
|
|
1320
|
+
} catch {
|
|
1321
|
+
try {
|
|
1322
|
+
page.objs.get(id, (img) => done(img));
|
|
1323
|
+
} catch {
|
|
1324
|
+
done(null);
|
|
1325
|
+
}
|
|
1326
|
+
}
|
|
1327
|
+
});
|
|
1328
|
+
const maskToRgba = (data, width, height, fill) => {
|
|
1329
|
+
const m = fill.match(/^#([0-9a-f]{2})([0-9a-f]{2})([0-9a-f]{2})$/i);
|
|
1330
|
+
const r = m ? parseInt(m[1], 16) : 0, g = m ? parseInt(m[2], 16) : 0, b = m ? parseInt(m[3], 16) : 0;
|
|
1331
|
+
const out = new Uint8ClampedArray(width * height * 4);
|
|
1332
|
+
const rowBytes = width + 7 >> 3;
|
|
1333
|
+
for (let y = 0; y < height; y++) {
|
|
1334
|
+
for (let x = 0; x < width; x++) {
|
|
1335
|
+
const byte = data[y * rowBytes + (x >> 3)] ?? 255;
|
|
1336
|
+
const bit = byte >> 7 - (x & 7) & 1;
|
|
1337
|
+
const o = (y * width + x) * 4;
|
|
1338
|
+
out[o] = r;
|
|
1339
|
+
out[o + 1] = g;
|
|
1340
|
+
out[o + 2] = b;
|
|
1341
|
+
out[o + 3] = bit ? 0 : 255;
|
|
1342
|
+
}
|
|
1343
|
+
}
|
|
1344
|
+
return out;
|
|
1345
|
+
};
|
|
1346
|
+
const encode = async (imgObj, maskFill) => {
|
|
1347
|
+
if (!imgObj || !imgObj.width || !imgObj.height) return null;
|
|
1348
|
+
let data = imgObj.data;
|
|
1349
|
+
if (typeof data === "string") data = (await resolveObj(data))?.data ?? null;
|
|
1350
|
+
if (maskFill) {
|
|
1351
|
+
if (!data) return null;
|
|
1352
|
+
return runtime.encodePng(imgObj.width, imgObj.height, 3, maskToRgba(data, imgObj.width, imgObj.height, maskFill));
|
|
1353
|
+
}
|
|
1354
|
+
if (data) {
|
|
1355
|
+
let kind = imgObj.kind || 0;
|
|
1356
|
+
if (!kind) {
|
|
1357
|
+
const px = imgObj.width * imgObj.height;
|
|
1358
|
+
if (data.length === px * 4) kind = 3;
|
|
1359
|
+
else if (data.length === px * 3) kind = 2;
|
|
1360
|
+
else if (data.length === (imgObj.width + 7 >> 3) * imgObj.height) kind = 1;
|
|
1361
|
+
}
|
|
1362
|
+
return runtime.encodePng(imgObj.width, imgObj.height, kind, data);
|
|
1363
|
+
}
|
|
1364
|
+
if (imgObj.bitmap) {
|
|
1365
|
+
try {
|
|
1366
|
+
const { canvas, context } = runtime.createCanvas(imgObj.width, imgObj.height);
|
|
1367
|
+
context.drawImage(imgObj.bitmap, 0, 0);
|
|
1368
|
+
if (typeof canvas.toDataURL === "function") return canvas.toDataURL("image/png");
|
|
1369
|
+
if (typeof canvas.toBuffer === "function") return `data:image/png;base64,${canvas.toBuffer("image/png").toString("base64")}`;
|
|
1370
|
+
} catch {
|
|
1371
|
+
}
|
|
1372
|
+
}
|
|
1373
|
+
return null;
|
|
1374
|
+
};
|
|
926
1375
|
const tasks = positions.map(async (pos) => {
|
|
1376
|
+
if (pos.inline) {
|
|
1377
|
+
const dataUrl2 = await encode(pos.inline, pos.maskFill);
|
|
1378
|
+
return dataUrl2 ? { pos, dataUrl: dataUrl2 } : null;
|
|
1379
|
+
}
|
|
927
1380
|
if (dataUrlCache.has(pos.name)) {
|
|
928
1381
|
return { pos, dataUrl: dataUrlCache.get(pos.name) };
|
|
929
1382
|
}
|
|
930
1383
|
let imgObj = null;
|
|
931
1384
|
try {
|
|
932
|
-
imgObj = await
|
|
933
|
-
let settled2 = false;
|
|
934
|
-
const timer = setTimeout(() => done(null), 250);
|
|
935
|
-
const done = (v) => {
|
|
936
|
-
if (settled2) return;
|
|
937
|
-
settled2 = true;
|
|
938
|
-
clearTimeout(timer);
|
|
939
|
-
resolve(v);
|
|
940
|
-
};
|
|
941
|
-
try {
|
|
942
|
-
page.objs.get(pos.name, (img) => done(img));
|
|
943
|
-
} catch {
|
|
944
|
-
try {
|
|
945
|
-
page.commonObjs.get(pos.name, (img) => done(img));
|
|
946
|
-
} catch {
|
|
947
|
-
done(null);
|
|
948
|
-
}
|
|
949
|
-
}
|
|
950
|
-
});
|
|
1385
|
+
imgObj = await resolveObj(pos.name);
|
|
951
1386
|
} catch {
|
|
952
1387
|
imgObj = null;
|
|
953
1388
|
}
|
|
954
|
-
|
|
955
|
-
const dataUrl = runtime.encodePng(imgObj.width, imgObj.height, imgObj.kind || 0, imgObj.data);
|
|
1389
|
+
const dataUrl = await encode(imgObj);
|
|
956
1390
|
if (!dataUrl) return null;
|
|
957
1391
|
dataUrlCache.set(pos.name, dataUrl);
|
|
958
1392
|
return { pos, dataUrl };
|
|
@@ -960,7 +1394,20 @@ async function extractImages(page, positions, runtime, dataUrlCache) {
|
|
|
960
1394
|
const settled = await Promise.all(tasks);
|
|
961
1395
|
return settled.filter((x) => x !== null);
|
|
962
1396
|
}
|
|
963
|
-
async function
|
|
1397
|
+
async function resolveDestPage(doc, dest) {
|
|
1398
|
+
try {
|
|
1399
|
+
let d = dest;
|
|
1400
|
+
if (typeof d === "string") d = await doc.getDestination(d);
|
|
1401
|
+
if (Array.isArray(d) && d[0] != null) {
|
|
1402
|
+
if (typeof d[0] === "number") return d[0];
|
|
1403
|
+
const idx = await doc.getPageIndex(d[0]);
|
|
1404
|
+
if (typeof idx === "number") return idx;
|
|
1405
|
+
}
|
|
1406
|
+
} catch {
|
|
1407
|
+
}
|
|
1408
|
+
return void 0;
|
|
1409
|
+
}
|
|
1410
|
+
async function extractLinks(doc, page, viewport) {
|
|
964
1411
|
const out = [];
|
|
965
1412
|
let annots = [];
|
|
966
1413
|
try {
|
|
@@ -983,13 +1430,15 @@ async function extractLinks(page, viewport) {
|
|
|
983
1430
|
const xMax = Math.max(c1.x, c2.x);
|
|
984
1431
|
const yMax = Math.max(c1.y, c2.y);
|
|
985
1432
|
const rectMm = {
|
|
986
|
-
x: xMin *
|
|
987
|
-
y: yMin *
|
|
988
|
-
w: (xMax - xMin) *
|
|
989
|
-
h: (yMax - yMin) *
|
|
1433
|
+
x: xMin * PT_TO_MM2,
|
|
1434
|
+
y: yMin * PT_TO_MM2,
|
|
1435
|
+
w: (xMax - xMin) * PT_TO_MM2,
|
|
1436
|
+
h: (yMax - yMin) * PT_TO_MM2
|
|
990
1437
|
};
|
|
991
1438
|
const url = a.url || a.unsafeUrl;
|
|
992
|
-
|
|
1439
|
+
const destPage = url ? void 0 : await resolveDestPage(doc, a.dest);
|
|
1440
|
+
if (!url && destPage == null) continue;
|
|
1441
|
+
out.push({ rectMm, url, destPage });
|
|
993
1442
|
}
|
|
994
1443
|
return out;
|
|
995
1444
|
}
|
|
@@ -1022,10 +1471,10 @@ async function extractFormWidgets(page, viewport) {
|
|
|
1022
1471
|
})).filter((o) => o.value !== "") : [];
|
|
1023
1472
|
out.push({
|
|
1024
1473
|
rectMm: {
|
|
1025
|
-
x: xMin *
|
|
1026
|
-
y: yMin *
|
|
1027
|
-
w: (xMax - xMin) *
|
|
1028
|
-
h: (yMax - yMin) *
|
|
1474
|
+
x: xMin * PT_TO_MM2,
|
|
1475
|
+
y: yMin * PT_TO_MM2,
|
|
1476
|
+
w: (xMax - xMin) * PT_TO_MM2,
|
|
1477
|
+
h: (yMax - yMin) * PT_TO_MM2
|
|
1029
1478
|
},
|
|
1030
1479
|
fieldType: a.fieldType || "",
|
|
1031
1480
|
fieldName: a.fieldName || `field-${out.length + 1}`,
|
|
@@ -1045,26 +1494,62 @@ async function extractFormWidgets(page, viewport) {
|
|
|
1045
1494
|
async function flattenOutline(doc, outline) {
|
|
1046
1495
|
if (!outline) return [];
|
|
1047
1496
|
const out = [];
|
|
1048
|
-
async function
|
|
1497
|
+
async function walk2(items, depth) {
|
|
1049
1498
|
for (const item of items) {
|
|
1050
|
-
|
|
1051
|
-
|
|
1052
|
-
|
|
1053
|
-
dest = await doc.getDestination(dest);
|
|
1054
|
-
}
|
|
1055
|
-
if (Array.isArray(dest) && dest[0]) {
|
|
1056
|
-
const ref = dest[0];
|
|
1057
|
-
const idx = await doc.getPageIndex(ref);
|
|
1058
|
-
if (typeof idx === "number") out.push({ title: item.title, pageIndex: idx });
|
|
1059
|
-
}
|
|
1060
|
-
} catch {
|
|
1499
|
+
const idx = await resolveDestPage(doc, item.dest);
|
|
1500
|
+
if (idx != null && typeof item.title === "string" && item.title.trim()) {
|
|
1501
|
+
out.push({ title: item.title.trim(), pageIndex: idx, depth });
|
|
1061
1502
|
}
|
|
1062
|
-
if (item.items?.length) await
|
|
1503
|
+
if (item.items?.length) await walk2(item.items, depth + 1);
|
|
1063
1504
|
}
|
|
1064
1505
|
}
|
|
1065
|
-
await
|
|
1506
|
+
await walk2(outline, 1);
|
|
1066
1507
|
return out;
|
|
1067
1508
|
}
|
|
1509
|
+
var normTitle = (s) => s.toLowerCase().replace(/[\s\u00a0]+/g, " ").replace(/[^\p{L}\p{N} ]/gu, "").trim();
|
|
1510
|
+
function applyOutline(pages, outline) {
|
|
1511
|
+
for (const entry of outline) {
|
|
1512
|
+
const page = pages[entry.pageIndex];
|
|
1513
|
+
if (!page) continue;
|
|
1514
|
+
const want = normTitle(entry.title);
|
|
1515
|
+
if (!want) continue;
|
|
1516
|
+
let best = null;
|
|
1517
|
+
let bestScore = 0;
|
|
1518
|
+
for (const el of page.elements) {
|
|
1519
|
+
if (el.type !== "text") continue;
|
|
1520
|
+
const have = normTitle(el.content || "");
|
|
1521
|
+
if (!have) continue;
|
|
1522
|
+
let score = 0;
|
|
1523
|
+
if (have === want) score = 3;
|
|
1524
|
+
else if (have.startsWith(want) || want.startsWith(have)) score = 2;
|
|
1525
|
+
else if (have.length >= 6 && want.includes(have)) score = 1;
|
|
1526
|
+
if (score > bestScore || score === bestScore && best && score > 0 && (el.position?.y ?? 0) < (best.position?.y ?? 0)) {
|
|
1527
|
+
best = el;
|
|
1528
|
+
bestScore = score;
|
|
1529
|
+
}
|
|
1530
|
+
}
|
|
1531
|
+
if (best && bestScore > 0) {
|
|
1532
|
+
const level = Math.min(6, Math.max(1, entry.depth));
|
|
1533
|
+
if (!best.heading) best.heading = level;
|
|
1534
|
+
best.tocEntry = entry.title;
|
|
1535
|
+
best.tocLevel = level;
|
|
1536
|
+
}
|
|
1537
|
+
}
|
|
1538
|
+
}
|
|
1539
|
+
function pdfDateToIso(v) {
|
|
1540
|
+
if (typeof v !== "string") return void 0;
|
|
1541
|
+
const m = v.match(/^D:(\d{4})(\d{2})?(\d{2})?(\d{2})?(\d{2})?(\d{2})?([Zz+-])?(\d{2})?'?(\d{2})?/);
|
|
1542
|
+
if (!m) {
|
|
1543
|
+
const t2 = Date.parse(v);
|
|
1544
|
+
return Number.isFinite(t2) ? new Date(t2).toISOString() : void 0;
|
|
1545
|
+
}
|
|
1546
|
+
const [, Y, Mo = "01", D = "01", h = "00", mi = "00", s = "00", sign, oh = "00", om = "00"] = m;
|
|
1547
|
+
const tz = !sign || sign === "Z" || sign === "z" ? "Z" : `${sign}${oh}:${om}`;
|
|
1548
|
+
const iso = `${Y}-${Mo}-${D}T${h}:${mi}:${s}${tz}`;
|
|
1549
|
+
const t = Date.parse(iso);
|
|
1550
|
+
return Number.isFinite(t) ? new Date(t).toISOString() : void 0;
|
|
1551
|
+
}
|
|
1552
|
+
var annotationModeDisable = 0;
|
|
1068
1553
|
async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
1069
1554
|
const pdfjs = options.pdfjs || runtime.pdfjs;
|
|
1070
1555
|
if (!pdfjs) {
|
|
@@ -1085,7 +1570,7 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
|
1085
1570
|
} else {
|
|
1086
1571
|
data = source;
|
|
1087
1572
|
}
|
|
1088
|
-
const
|
|
1573
|
+
const loadingTask = pdfjs.getDocument({
|
|
1089
1574
|
data,
|
|
1090
1575
|
// The runtime adapter declares whether it supports a real Web Worker.
|
|
1091
1576
|
// We don't sniff `typeof Worker` here because Node 22+ exposes a global
|
|
@@ -1094,15 +1579,55 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
|
1094
1579
|
// newer Node. Browser entry leaves this unset (= false = real worker
|
|
1095
1580
|
// via GlobalWorkerOptions.workerSrc); node entry sets `true`.
|
|
1096
1581
|
disableWorker: runtime.disableWorker === true,
|
|
1097
|
-
isEvalSupported: false
|
|
1098
|
-
|
|
1582
|
+
isEvalSupported: false,
|
|
1583
|
+
// Keep going past malformed content streams instead of failing the page.
|
|
1584
|
+
stopAtErrors: false,
|
|
1585
|
+
// Force the classic pixel-array image path on every host so the reader
|
|
1586
|
+
// (WKWebView has OffscreenCanvas) and the CLI produce identical PNGs.
|
|
1587
|
+
isOffscreenCanvasSupported: false,
|
|
1588
|
+
password: options.password,
|
|
1589
|
+
// Standard-14 font metrics + CJK CMaps: without these PDF.js falls back
|
|
1590
|
+
// to guesses for non-embedded fonts and logs a warning per page.
|
|
1591
|
+
...runtime.standardFontDataUrl ? { standardFontDataUrl: runtime.standardFontDataUrl } : {},
|
|
1592
|
+
...runtime.cMapUrl ? { cMapUrl: runtime.cMapUrl, cMapPacked: true } : {}
|
|
1593
|
+
});
|
|
1594
|
+
let passwordTried = typeof options.password === "string";
|
|
1595
|
+
loadingTask.onPassword = (updatePassword, reason) => {
|
|
1596
|
+
const retry = reason === 2 || passwordTried;
|
|
1597
|
+
if (!options.onPassword) {
|
|
1598
|
+
updatePassword(new Error(retry ? "[@jdf/pdf-import] Wrong password for encrypted PDF" : "[@jdf/pdf-import] PDF is password-protected \u2014 pass `password`"));
|
|
1599
|
+
return;
|
|
1600
|
+
}
|
|
1601
|
+
passwordTried = true;
|
|
1602
|
+
options.onPassword(retry).then((pw) => {
|
|
1603
|
+
if (pw == null) updatePassword(new Error("[@jdf/pdf-import] Password entry cancelled"));
|
|
1604
|
+
else updatePassword(pw);
|
|
1605
|
+
}).catch((e) => updatePassword(e instanceof Error ? e : new Error(String(e))));
|
|
1606
|
+
};
|
|
1607
|
+
let doc;
|
|
1608
|
+
try {
|
|
1609
|
+
doc = await loadingTask.promise;
|
|
1610
|
+
} catch (e) {
|
|
1611
|
+
if (e?.name === "PasswordException") {
|
|
1612
|
+
const msg = /no password/i.test(e.message || "") ? "PDF is password-protected \u2014 pass a password (CLI: --password)" : e.message || "PDF is password-protected";
|
|
1613
|
+
throw new Error(`[@jdf/pdf-import] ${msg}`);
|
|
1614
|
+
}
|
|
1615
|
+
throw e;
|
|
1616
|
+
}
|
|
1099
1617
|
const pages = [];
|
|
1100
1618
|
const imageResources = {};
|
|
1101
1619
|
let imgCounter = 0;
|
|
1102
1620
|
const dataUrlCache = /* @__PURE__ */ new Map();
|
|
1103
1621
|
const resourceKeyByName = /* @__PURE__ */ new Map();
|
|
1104
|
-
const outline = await doc.getOutline().catch(() => null);
|
|
1105
|
-
|
|
1622
|
+
const outline = await flattenOutline(doc, await doc.getOutline().catch(() => null));
|
|
1623
|
+
let pdfInfo = null;
|
|
1624
|
+
let pdfMetadata = null;
|
|
1625
|
+
try {
|
|
1626
|
+
const md = await doc.getMetadata();
|
|
1627
|
+
pdfInfo = md?.info ?? null;
|
|
1628
|
+
pdfMetadata = md?.metadata ?? null;
|
|
1629
|
+
} catch {
|
|
1630
|
+
}
|
|
1106
1631
|
for (let pi = 1; pi <= doc.numPages; pi++) {
|
|
1107
1632
|
let findLinkForRun2 = function(r) {
|
|
1108
1633
|
const cx = r.x + r.width / 2;
|
|
@@ -1124,7 +1649,7 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
|
1124
1649
|
} catch {
|
|
1125
1650
|
}
|
|
1126
1651
|
const ops = await walkOps(page, OPS, viewport);
|
|
1127
|
-
const links = await extractLinks(page, viewport);
|
|
1652
|
+
const links = await extractLinks(doc, page, viewport);
|
|
1128
1653
|
const formWidgets = await extractFormWidgets(page, viewport);
|
|
1129
1654
|
const textContent = await page.getTextContent({ disableCombineTextItems: false });
|
|
1130
1655
|
const items = textContent.items;
|
|
@@ -1167,9 +1692,53 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
|
1167
1692
|
const n = typeof v === "number" ? v : Number(v);
|
|
1168
1693
|
return Number.isFinite(n) ? n : fallback;
|
|
1169
1694
|
};
|
|
1170
|
-
|
|
1695
|
+
const opBins = /* @__PURE__ */ new Map();
|
|
1696
|
+
for (const op of ops.textOps) {
|
|
1697
|
+
const key = `${Math.round(op.x)},${Math.round(op.y)}`;
|
|
1698
|
+
const arr = opBins.get(key);
|
|
1699
|
+
if (arr) arr.push(op);
|
|
1700
|
+
else opBins.set(key, [op]);
|
|
1701
|
+
}
|
|
1702
|
+
const findOp = (x, y, fontSize) => {
|
|
1703
|
+
let best = null;
|
|
1704
|
+
let bestD = Infinity;
|
|
1705
|
+
const bx = Math.round(x), by = Math.round(y);
|
|
1706
|
+
for (let dy = -1; dy <= 1; dy++) {
|
|
1707
|
+
for (let dx = -1; dx <= 1; dx++) {
|
|
1708
|
+
const arr = opBins.get(`${bx + dx},${by + dy}`);
|
|
1709
|
+
if (!arr) continue;
|
|
1710
|
+
for (const op of arr) {
|
|
1711
|
+
const d = Math.hypot(op.x - x, op.y - y);
|
|
1712
|
+
if (d < bestD) {
|
|
1713
|
+
bestD = d;
|
|
1714
|
+
best = op;
|
|
1715
|
+
}
|
|
1716
|
+
}
|
|
1717
|
+
}
|
|
1718
|
+
}
|
|
1719
|
+
if (best) return best;
|
|
1720
|
+
const tol = Math.max(2, fontSize * 0.6);
|
|
1721
|
+
for (const op of ops.textOps) {
|
|
1722
|
+
if (Math.abs(op.y - y) > tol) continue;
|
|
1723
|
+
const d = Math.abs(op.x - x) + Math.abs(op.y - y) * 4;
|
|
1724
|
+
if (d < bestD) {
|
|
1725
|
+
bestD = d;
|
|
1726
|
+
best = op;
|
|
1727
|
+
}
|
|
1728
|
+
}
|
|
1729
|
+
if (best) return best;
|
|
1730
|
+
for (const op of ops.textOps) {
|
|
1731
|
+
const d = Math.hypot(op.x - x, op.y - y);
|
|
1732
|
+
if (d < bestD) {
|
|
1733
|
+
bestD = d;
|
|
1734
|
+
best = op;
|
|
1735
|
+
}
|
|
1736
|
+
}
|
|
1737
|
+
return bestD < 40 ? best : null;
|
|
1738
|
+
};
|
|
1739
|
+
const keepInvisible = options.invisibleText !== "drop";
|
|
1740
|
+
items.forEach((it) => {
|
|
1171
1741
|
if (!it.str || !it.str.length) return;
|
|
1172
|
-
if ((ops.textRenderingModes[idx] ?? 0) === 3) return;
|
|
1173
1742
|
const tr = it.transform;
|
|
1174
1743
|
const fontSize = safeNum(Math.hypot(safeNum(tr?.[2], 0), safeNum(tr?.[3], 0)), 0) || safeNum(it.height, 0) || 10;
|
|
1175
1744
|
const baseX = safeNum(tr?.[4], 0);
|
|
@@ -1177,24 +1746,35 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
|
1177
1746
|
const conv = viewport.convertToViewportPoint(baseX, baseY);
|
|
1178
1747
|
const vx = safeNum(conv?.[0], 0);
|
|
1179
1748
|
const vy = safeNum(conv?.[1], 0);
|
|
1749
|
+
const op = findOp(vx, vy, fontSize);
|
|
1750
|
+
const mode = op?.mode ?? 0;
|
|
1751
|
+
if (mode === 7) return;
|
|
1752
|
+
const invisible = mode === 3;
|
|
1753
|
+
if (invisible && !keepInvisible) return;
|
|
1180
1754
|
const ascent = it.height ? safeNum(it.height, fontSize) * 0.78 : fontSize * 0.78;
|
|
1181
1755
|
const yTop = vy - ascent;
|
|
1182
1756
|
const w = safeNum(it.width, 0);
|
|
1183
1757
|
runs.push({
|
|
1184
1758
|
text: it.str,
|
|
1185
|
-
x: safeNum(vx *
|
|
1186
|
-
y: safeNum(yTop *
|
|
1759
|
+
x: safeNum(vx * PT_TO_MM2, 0),
|
|
1760
|
+
y: safeNum(yTop * PT_TO_MM2, 0),
|
|
1187
1761
|
fontSize: safeNum(fontSize, 10),
|
|
1188
1762
|
fontName: it.fontName,
|
|
1189
|
-
width: safeNum(w *
|
|
1190
|
-
height: safeNum((it.height || fontSize) *
|
|
1191
|
-
color:
|
|
1192
|
-
opacity: safeNum(
|
|
1763
|
+
width: safeNum(w * PT_TO_MM2, 0),
|
|
1764
|
+
height: safeNum((it.height || fontSize) * PT_TO_MM2, fontSize * PT_TO_MM2),
|
|
1765
|
+
color: op?.fill || "#000000",
|
|
1766
|
+
opacity: invisible ? 0 : safeNum(op?.alpha, 1)
|
|
1193
1767
|
});
|
|
1194
1768
|
});
|
|
1195
1769
|
runs.sort((a, b) => a.y - b.y || a.x - b.x);
|
|
1196
1770
|
const lines = [];
|
|
1197
1771
|
const Y_TOL = 0.6;
|
|
1772
|
+
const kGlyph = calibrateGlyphWidth(runs);
|
|
1773
|
+
const stretchedSpaces = hasStretchedSpaces(runs, kGlyph);
|
|
1774
|
+
const fontKey = (name) => {
|
|
1775
|
+
const c = fontMap.get(name) || classifyFont(name || "");
|
|
1776
|
+
return `${c.family}|${c.weight || ""}|${c.style || ""}`;
|
|
1777
|
+
};
|
|
1198
1778
|
for (const r of runs) {
|
|
1199
1779
|
if (!r.text.length) continue;
|
|
1200
1780
|
const last = lines[lines.length - 1];
|
|
@@ -1203,24 +1783,48 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
|
1203
1783
|
continue;
|
|
1204
1784
|
}
|
|
1205
1785
|
const sameLine = Math.abs(last.y - r.y) <= Y_TOL;
|
|
1206
|
-
const sameStyle = Math.abs(last.fontSize - r.fontSize) < 0.4 && last.fontName === r.fontName && last.color === r.color && Math.abs(last.opacity - r.opacity) < 0.05;
|
|
1207
|
-
const
|
|
1208
|
-
const
|
|
1209
|
-
|
|
1786
|
+
const sameStyle = Math.abs(last.fontSize - r.fontSize) < 0.4 && (last.fontName === r.fontName || fontKey(last.fontName) === fontKey(r.fontName)) && last.color === r.color && Math.abs(last.opacity - r.opacity) < 0.05;
|
|
1787
|
+
const emMm = r.fontSize * PT_TO_MM2;
|
|
1788
|
+
const extent = (t) => {
|
|
1789
|
+
if (!/\s$/.test(t.text)) return t.width;
|
|
1790
|
+
const em = t.fontSize * PT_TO_MM2;
|
|
1791
|
+
const est = Math.max(1, t.text.trim().length) * em * kGlyph + em * 0.25;
|
|
1792
|
+
if (stretchedSpaces) return Math.min(t.width, est);
|
|
1793
|
+
return t.width > est * 1.4 ? est : t.width;
|
|
1794
|
+
};
|
|
1795
|
+
const gapMm = r.x - (last.x + extent(last));
|
|
1796
|
+
const mergeOk = sameLine && sameStyle && gapMm >= -emMm * 0.5 && gapMm <= emMm * 0.45;
|
|
1210
1797
|
if (mergeOk) {
|
|
1211
1798
|
const lastEndsSpace = /\s$/.test(last.text);
|
|
1212
1799
|
const currStartsSpace = /^\s/.test(r.text);
|
|
1213
1800
|
const sep = gapMm > emMm * 0.08 && !lastEndsSpace && !currStartsSpace ? " " : "";
|
|
1214
1801
|
last.text = last.text + sep + r.text;
|
|
1215
1802
|
const newExtent = r.x - last.x + r.width;
|
|
1216
|
-
last.width = Math.max(last
|
|
1803
|
+
last.width = Math.max(extent(last), newExtent);
|
|
1217
1804
|
} else {
|
|
1218
1805
|
lines.push({ ...r });
|
|
1219
1806
|
}
|
|
1220
1807
|
}
|
|
1221
1808
|
const elements = [];
|
|
1222
|
-
|
|
1223
|
-
|
|
1809
|
+
const tRuns = lines.map((l) => {
|
|
1810
|
+
const cls = fontMap.get(l.fontName) || classifyFont(l.fontName || "");
|
|
1811
|
+
return { text: l.text, x: l.x, y: l.y, width: l.width, height: l.height, fontSize: l.fontSize, fontName: l.fontName, color: l.color, bold: cls.weight === "bold" };
|
|
1812
|
+
});
|
|
1813
|
+
const detected = options.detectTables === false ? [] : detectTables(tRuns, ops.shapes, pageW * PT_TO_MM2);
|
|
1814
|
+
const consumedLines = /* @__PURE__ */ new Set();
|
|
1815
|
+
const consumedShapes = /* @__PURE__ */ new Set();
|
|
1816
|
+
const tableAtLine = /* @__PURE__ */ new Map();
|
|
1817
|
+
for (const t of detected) {
|
|
1818
|
+
for (const k of t.lineIdx) consumedLines.add(k);
|
|
1819
|
+
for (const k of t.shapeIdx) consumedShapes.add(k);
|
|
1820
|
+
tableAtLine.set(Math.min(...t.lineIdx), t.element);
|
|
1821
|
+
}
|
|
1822
|
+
const pageWmm = pageW * PT_TO_MM2, pageHmm = pageH * PT_TO_MM2;
|
|
1823
|
+
ops.shapes.forEach((sh, shapeIdx) => {
|
|
1824
|
+
if (consumedShapes.has(shapeIdx)) return;
|
|
1825
|
+
if (sh.width < 0.3 && sh.height < 0.3) return;
|
|
1826
|
+
if (sh.x + sh.width <= 0 || sh.y + sh.height <= 0 || sh.x >= pageWmm || sh.y >= pageHmm) return;
|
|
1827
|
+
if (sh.kind === "rect" && sh.fill && !sh.stroke && sh.width * sh.height >= pageWmm * pageHmm * 0.9) return;
|
|
1224
1828
|
const shapeType = sh.kind;
|
|
1225
1829
|
const shape = {
|
|
1226
1830
|
type: "shape",
|
|
@@ -1236,7 +1840,7 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
|
1236
1840
|
shape.style = { opacity: Math.round(sh.opacity * 100) / 100 };
|
|
1237
1841
|
}
|
|
1238
1842
|
elements.push(shape);
|
|
1239
|
-
}
|
|
1843
|
+
});
|
|
1240
1844
|
const imgs = await extractImages(page, ops.imagePositions, runtime, dataUrlCache);
|
|
1241
1845
|
for (const { pos, dataUrl } of imgs) {
|
|
1242
1846
|
let resourceKey = resourceKeyByName.get(pos.name);
|
|
@@ -1259,7 +1863,16 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
|
1259
1863
|
fit: "fill"
|
|
1260
1864
|
});
|
|
1261
1865
|
}
|
|
1866
|
+
const sizeChars = /* @__PURE__ */ new Map();
|
|
1262
1867
|
for (const l of lines) {
|
|
1868
|
+
const k = Math.round(l.fontSize * 2) / 2;
|
|
1869
|
+
sizeChars.set(k, (sizeChars.get(k) ?? 0) + l.text.length);
|
|
1870
|
+
}
|
|
1871
|
+
const bodyFontSize = [...sizeChars.entries()].sort((a, b) => b[1] - a[1])[0]?.[0] ?? 0;
|
|
1872
|
+
lines.forEach((l, lineIdx) => {
|
|
1873
|
+
const tableEl = tableAtLine.get(lineIdx);
|
|
1874
|
+
if (tableEl) elements.push(tableEl);
|
|
1875
|
+
if (consumedLines.has(lineIdx)) return;
|
|
1263
1876
|
const cls = fontMap.get(l.fontName) || classifyFont(l.fontName || "");
|
|
1264
1877
|
const style = {
|
|
1265
1878
|
fontSize: Math.round(l.fontSize * 10) / 10,
|
|
@@ -1270,8 +1883,7 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
|
1270
1883
|
if (l.color !== "#000000") style.color = l.color;
|
|
1271
1884
|
if (l.opacity < 0.999) style.opacity = Math.round(l.opacity * 100) / 100;
|
|
1272
1885
|
const link = findLinkForRun2(l);
|
|
1273
|
-
const
|
|
1274
|
-
const measured = Math.max(l.width + l.fontSize * PT_TO_MM * 0.4, l.fontSize * PT_TO_MM);
|
|
1886
|
+
const measured = Math.max(l.width + l.fontSize * PT_TO_MM2 * 0.4, l.fontSize * PT_TO_MM2);
|
|
1275
1887
|
const remaining = Math.max(measured, pageWmm - l.x);
|
|
1276
1888
|
const elWidth = Math.min(measured, remaining);
|
|
1277
1889
|
const text = {
|
|
@@ -1281,18 +1893,26 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
|
1281
1893
|
width: Math.max(2, Math.round(elWidth * 100) / 100),
|
|
1282
1894
|
style
|
|
1283
1895
|
};
|
|
1284
|
-
if (cls.weight === "bold") {
|
|
1285
|
-
|
|
1286
|
-
|
|
1287
|
-
else if (l.fontSize >=
|
|
1896
|
+
if (cls.weight === "bold" && l.text.trim().length <= 120 && !consumedLines.has(lineIdx)) {
|
|
1897
|
+
const ratio = bodyFontSize > 0 ? l.fontSize / bodyFontSize : 1;
|
|
1898
|
+
if (l.fontSize >= 22 || ratio >= 1.8) text.heading = 1;
|
|
1899
|
+
else if (l.fontSize >= 17 || ratio >= 1.35) text.heading = 2;
|
|
1900
|
+
else if (l.fontSize >= 16 || ratio >= 1.2) text.heading = 3;
|
|
1288
1901
|
}
|
|
1289
1902
|
if (text.heading) text.tocEntry = text.content;
|
|
1290
1903
|
if (link) {
|
|
1291
1904
|
if (link.url) text.link = link.url;
|
|
1292
1905
|
else if (link.destPage != null) text.link = { type: "internal", target: `#page-${link.destPage + 1}` };
|
|
1293
1906
|
}
|
|
1907
|
+
const prev = elements[elements.length - 1];
|
|
1908
|
+
if (text.heading && prev && prev.type === "text" && prev.heading === text.heading && !link && !prev.link && Math.abs(prev.style?.fontSize - style.fontSize) < 0.5 && Math.abs(prev.position.x - text.position.x) < 1 && text.position.y - prev.position.y < l.fontSize * PT_TO_MM2 * 2.2 && text.position.y > prev.position.y) {
|
|
1909
|
+
prev.content = `${prev.content} ${text.content}`.replace(/\s+/g, " ");
|
|
1910
|
+
prev.tocEntry = prev.content;
|
|
1911
|
+
prev.width = Math.max(prev.width ?? 0, text.width ?? 0);
|
|
1912
|
+
return;
|
|
1913
|
+
}
|
|
1294
1914
|
elements.push(text);
|
|
1295
|
-
}
|
|
1915
|
+
});
|
|
1296
1916
|
for (const w of formWidgets) {
|
|
1297
1917
|
if (w.pushButton) continue;
|
|
1298
1918
|
const baseEl = {
|
|
@@ -1327,19 +1947,41 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
|
1327
1947
|
}
|
|
1328
1948
|
pages.push({
|
|
1329
1949
|
id: `page-${pi}`,
|
|
1330
|
-
pageSize: { width: Math.round(pageW *
|
|
1950
|
+
pageSize: { width: Math.round(pageW * PT_TO_MM2 * 100) / 100, height: Math.round(pageH * PT_TO_MM2 * 100) / 100 },
|
|
1331
1951
|
margins: { top: 0, right: 0, bottom: 0, left: 0 },
|
|
1332
1952
|
elements
|
|
1333
1953
|
});
|
|
1334
1954
|
}
|
|
1955
|
+
applyOutline(pages, outline);
|
|
1956
|
+
const meta = {
|
|
1957
|
+
title,
|
|
1958
|
+
pageSize: pages[0]?.pageSize || "A4",
|
|
1959
|
+
unit: "mm",
|
|
1960
|
+
margins: { top: 0, right: 0, bottom: 0, left: 0 }
|
|
1961
|
+
};
|
|
1962
|
+
if (pdfInfo) {
|
|
1963
|
+
if (typeof pdfInfo.Author === "string" && pdfInfo.Author.trim()) meta.author = pdfInfo.Author.trim();
|
|
1964
|
+
const created = pdfDateToIso(pdfInfo.CreationDate);
|
|
1965
|
+
const modified = pdfDateToIso(pdfInfo.ModDate);
|
|
1966
|
+
if (created) meta.created = created;
|
|
1967
|
+
if (modified) meta.modified = modified;
|
|
1968
|
+
if (typeof pdfInfo.Keywords === "string") {
|
|
1969
|
+
const kws = pdfInfo.Keywords.split(/[,;]+/).map((k) => k.trim()).filter(Boolean);
|
|
1970
|
+
if (kws.length) meta.keywords = kws;
|
|
1971
|
+
}
|
|
1972
|
+
if (typeof pdfInfo.Language === "string" && pdfInfo.Language.trim()) meta.language = pdfInfo.Language.trim();
|
|
1973
|
+
}
|
|
1974
|
+
if (!meta.language && pdfMetadata && typeof pdfMetadata.get === "function") {
|
|
1975
|
+
try {
|
|
1976
|
+
const lang = pdfMetadata.get("dc:language");
|
|
1977
|
+
const first = Array.isArray(lang) ? lang[0] : lang;
|
|
1978
|
+
if (typeof first === "string" && first.trim()) meta.language = first.trim();
|
|
1979
|
+
} catch {
|
|
1980
|
+
}
|
|
1981
|
+
}
|
|
1335
1982
|
const result = {
|
|
1336
1983
|
$jdf: "1.0.0",
|
|
1337
|
-
meta
|
|
1338
|
-
title,
|
|
1339
|
-
pageSize: pages[0]?.pageSize || "A4",
|
|
1340
|
-
unit: "mm",
|
|
1341
|
-
margins: { top: 0, right: 0, bottom: 0, left: 0 }
|
|
1342
|
-
},
|
|
1984
|
+
meta,
|
|
1343
1985
|
pages
|
|
1344
1986
|
};
|
|
1345
1987
|
if (Object.keys(imageResources).length > 0) {
|
|
@@ -1351,27 +1993,31 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
|
|
|
1351
1993
|
// ../../packages/jdf-pdf-import/src/node.ts
|
|
1352
1994
|
var pdfjsModule = null;
|
|
1353
1995
|
var pdfjsLoadPromise = null;
|
|
1996
|
+
var pdfjsAssetDirs = null;
|
|
1354
1997
|
async function loadNodePdfJs() {
|
|
1355
1998
|
if (pdfjsModule) return pdfjsModule;
|
|
1356
1999
|
if (pdfjsLoadPromise) return pdfjsLoadPromise;
|
|
1357
2000
|
pdfjsLoadPromise = (async () => {
|
|
1358
2001
|
const { createRequire } = await import('module');
|
|
1359
2002
|
const require_ = createRequire(import.meta.url);
|
|
1360
|
-
const
|
|
1361
|
-
console.
|
|
1362
|
-
if (typeof args[0] === "string" && args[0].includes("legacy")) return;
|
|
1363
|
-
|
|
2003
|
+
const origLog = console.log;
|
|
2004
|
+
console.log = (...args) => {
|
|
2005
|
+
if (typeof args[0] === "string" && args[0].includes("`legacy` build")) return;
|
|
2006
|
+
origLog.apply(console, args);
|
|
1364
2007
|
};
|
|
1365
2008
|
let lib;
|
|
1366
2009
|
try {
|
|
1367
2010
|
lib = await import('pdfjs-dist/build/pdf.mjs');
|
|
1368
2011
|
} finally {
|
|
1369
|
-
console.
|
|
2012
|
+
console.log = origLog;
|
|
1370
2013
|
}
|
|
1371
2014
|
const workerPath = require_.resolve("pdfjs-dist/build/pdf.worker.mjs");
|
|
1372
2015
|
if (lib.GlobalWorkerOptions) {
|
|
1373
2016
|
lib.GlobalWorkerOptions.workerSrc = workerPath;
|
|
1374
2017
|
}
|
|
2018
|
+
const { dirname, join } = await import('path');
|
|
2019
|
+
const pkgDir = dirname(dirname(workerPath));
|
|
2020
|
+
pdfjsAssetDirs = { standardFontDataUrl: join(pkgDir, "standard_fonts") + "/", cMapUrl: join(pkgDir, "cmaps") + "/" };
|
|
1375
2021
|
pdfjsModule = lib;
|
|
1376
2022
|
return lib;
|
|
1377
2023
|
})();
|
|
@@ -1382,6 +2028,10 @@ async function loadCanvas() {
|
|
|
1382
2028
|
if (canvasModule) return canvasModule;
|
|
1383
2029
|
try {
|
|
1384
2030
|
canvasModule = await import('@napi-rs/canvas');
|
|
2031
|
+
const g = globalThis;
|
|
2032
|
+
if (typeof g.DOMMatrix === "undefined" && canvasModule.DOMMatrix) g.DOMMatrix = canvasModule.DOMMatrix;
|
|
2033
|
+
if (typeof g.Path2D === "undefined" && canvasModule.Path2D) g.Path2D = canvasModule.Path2D;
|
|
2034
|
+
if (typeof g.ImageData === "undefined" && canvasModule.ImageData) g.ImageData = canvasModule.ImageData;
|
|
1385
2035
|
return canvasModule;
|
|
1386
2036
|
} catch (err) {
|
|
1387
2037
|
throw new Error(
|
|
@@ -1433,6 +2083,7 @@ async function importPdfToJdf2(source, title, options = {}) {
|
|
|
1433
2083
|
const runtime = {
|
|
1434
2084
|
pdfjs,
|
|
1435
2085
|
disableWorker: true,
|
|
2086
|
+
...pdfjsAssetDirs ?? {},
|
|
1436
2087
|
createCanvas(width, height) {
|
|
1437
2088
|
const canvas = canvasMod.createCanvas(width, height);
|
|
1438
2089
|
const context = canvas.getContext("2d");
|
|
@@ -1449,19 +2100,22 @@ async function importPdfToJdf2(source, title, options = {}) {
|
|
|
1449
2100
|
|
|
1450
2101
|
// src/commands/import-pdf.ts
|
|
1451
2102
|
async function importPdf(inputPath, outputPath, options = {}) {
|
|
1452
|
-
const input =
|
|
1453
|
-
if (!
|
|
2103
|
+
const input = path7.resolve(inputPath);
|
|
2104
|
+
if (!fs7.existsSync(input)) {
|
|
1454
2105
|
console.error(`File not found: ${input}`);
|
|
1455
2106
|
process.exit(1);
|
|
1456
2107
|
}
|
|
1457
2108
|
console.log(`Importing: ${input}`);
|
|
1458
|
-
const title =
|
|
2109
|
+
const title = path7.basename(input, path7.extname(input));
|
|
1459
2110
|
const t0 = Date.now();
|
|
1460
|
-
const doc = await importPdfToJdf2(input, title
|
|
2111
|
+
const doc = await importPdfToJdf2(input, title, {
|
|
2112
|
+
password: options.password,
|
|
2113
|
+
invisibleText: options.dropInvisibleText ? "drop" : "keep"
|
|
2114
|
+
});
|
|
1461
2115
|
console.log(`Parsed in ${((Date.now() - t0) / 1e3).toFixed(1)}s \u2014 ${doc.pages.length} page(s)`);
|
|
1462
2116
|
let output;
|
|
1463
2117
|
if (outputPath) {
|
|
1464
|
-
output =
|
|
2118
|
+
output = path7.resolve(outputPath);
|
|
1465
2119
|
} else {
|
|
1466
2120
|
const stem = input.replace(/\.pdf$/i, "");
|
|
1467
2121
|
const wantJdfx = !options.forceJson && shouldUseJdfx(doc);
|
|
@@ -1470,11 +2124,11 @@ async function importPdf(inputPath, outputPath, options = {}) {
|
|
|
1470
2124
|
console.log(`Output: ${output}`);
|
|
1471
2125
|
if (output.toLowerCase().endsWith(".jdfx")) {
|
|
1472
2126
|
const { bytes, manifest } = await packJdfx(doc);
|
|
1473
|
-
|
|
2127
|
+
fs7.writeFileSync(output, bytes);
|
|
1474
2128
|
console.log(`
|
|
1475
2129
|
Done! Created ${doc.pages.length} page(s), ${manifest.assets.length} asset(s) bundled`);
|
|
1476
2130
|
} else {
|
|
1477
|
-
|
|
2131
|
+
fs7.writeFileSync(output, JSON.stringify(doc, null, 2));
|
|
1478
2132
|
console.log(`
|
|
1479
2133
|
Done! Created ${doc.pages.length} page(s)`);
|
|
1480
2134
|
}
|
|
@@ -1487,23 +2141,23 @@ var ImportJsonError = class extends Error {
|
|
|
1487
2141
|
}
|
|
1488
2142
|
};
|
|
1489
2143
|
async function importJson(inputPath, outputPath, options = {}) {
|
|
1490
|
-
const input =
|
|
1491
|
-
if (!
|
|
2144
|
+
const input = path7.resolve(inputPath);
|
|
2145
|
+
if (!fs7.existsSync(input)) {
|
|
1492
2146
|
throw new ImportJsonError(`File not found: ${input}`);
|
|
1493
2147
|
}
|
|
1494
2148
|
console.log(`Importing: ${input}`);
|
|
1495
|
-
const raw =
|
|
2149
|
+
const raw = fs7.readFileSync(input, "utf-8");
|
|
1496
2150
|
let parsed;
|
|
1497
2151
|
try {
|
|
1498
2152
|
parsed = JSON.parse(raw);
|
|
1499
2153
|
} catch (e) {
|
|
1500
2154
|
throw new ImportJsonError(`Not valid JSON: ${e.message}`);
|
|
1501
2155
|
}
|
|
1502
|
-
const title =
|
|
2156
|
+
const title = path7.basename(input, path7.extname(input));
|
|
1503
2157
|
const doc = normaliseToJdf(parsed, title);
|
|
1504
2158
|
let output;
|
|
1505
2159
|
if (outputPath) {
|
|
1506
|
-
output =
|
|
2160
|
+
output = path7.resolve(outputPath);
|
|
1507
2161
|
} else {
|
|
1508
2162
|
const stem = input.replace(/\.json$/i, "");
|
|
1509
2163
|
const wantJdfx = !options.forceJson && shouldUseJdfx(doc);
|
|
@@ -1512,11 +2166,11 @@ async function importJson(inputPath, outputPath, options = {}) {
|
|
|
1512
2166
|
console.log(`Output: ${output}`);
|
|
1513
2167
|
if (output.toLowerCase().endsWith(".jdfx")) {
|
|
1514
2168
|
const { bytes, manifest } = await packJdfx(doc);
|
|
1515
|
-
|
|
2169
|
+
fs7.writeFileSync(output, bytes);
|
|
1516
2170
|
console.log(`
|
|
1517
2171
|
Done! Created ${doc.pages.length} page(s), ${manifest.assets.length} asset(s) bundled`);
|
|
1518
2172
|
} else {
|
|
1519
|
-
|
|
2173
|
+
fs7.writeFileSync(output, JSON.stringify(doc, null, 2));
|
|
1520
2174
|
console.log(`
|
|
1521
2175
|
Done! Created ${doc.pages.length} page(s)`);
|
|
1522
2176
|
}
|
|
@@ -1592,6 +2246,47 @@ function wrapElements(elements, title, meta) {
|
|
|
1592
2246
|
]
|
|
1593
2247
|
};
|
|
1594
2248
|
}
|
|
2249
|
+
var DEFAULT_TRANSCRIPT_WINDOW = 45;
|
|
2250
|
+
var fmtTime = (sec) => {
|
|
2251
|
+
const s = Math.max(0, Math.round(sec));
|
|
2252
|
+
const h = Math.floor(s / 3600), m = Math.floor(s % 3600 / 60), r = s % 60;
|
|
2253
|
+
return h ? `${h}:${String(m).padStart(2, "0")}:${String(r).padStart(2, "0")}` : `${String(m).padStart(2, "0")}:${String(r).padStart(2, "0")}`;
|
|
2254
|
+
};
|
|
2255
|
+
function transcriptChunks(el, elementId2, page, crumb, windowSec, maxTokens) {
|
|
2256
|
+
const segs = Array.isArray(el?.transcript?.segments) ? el.transcript.segments : [];
|
|
2257
|
+
if (!segs.length) return [];
|
|
2258
|
+
const chapters = Array.isArray(el?.chapters) ? [...el.chapters].sort((a, b) => a.t - b.t) : [];
|
|
2259
|
+
const chapterAt = (t) => {
|
|
2260
|
+
let cur = null;
|
|
2261
|
+
for (const c of chapters) {
|
|
2262
|
+
if (c.t <= t + 1e-6) cur = c;
|
|
2263
|
+
else break;
|
|
2264
|
+
}
|
|
2265
|
+
return cur;
|
|
2266
|
+
};
|
|
2267
|
+
const out = [];
|
|
2268
|
+
let win = [];
|
|
2269
|
+
const flush = () => {
|
|
2270
|
+
if (!win.length) return;
|
|
2271
|
+
const t0 = win[0].t0, t1 = win[win.length - 1].t1;
|
|
2272
|
+
const body = win.map((sg) => sg.speaker ? `${sg.speaker}: ${sg.text}` : sg.text).join(" ").replace(/\s+/g, " ").trim();
|
|
2273
|
+
const text = `[${fmtTime(t0)}\u2013${fmtTime(t1)}] ${body}`;
|
|
2274
|
+
const chapter = chapterAt(t0);
|
|
2275
|
+
const path9 = [...crumb, ...el.title ? [String(el.title)] : [], ...chapter ? [String(chapter.title)] : []];
|
|
2276
|
+
out.push({ id: `${elementId2}@${Math.round(t0)}`, text, path: path9, page, types: ["video"], tokens: estimateTokens(text), hash: hashText(text), media: { element: elementId2, t0, t1 } });
|
|
2277
|
+
win = [];
|
|
2278
|
+
};
|
|
2279
|
+
for (const sg of segs) {
|
|
2280
|
+
if (typeof sg?.text !== "string" || !sg.text.trim()) continue;
|
|
2281
|
+
const startsNewChapter = win.length && chapterAt(sg.t0) !== chapterAt(win[0].t0);
|
|
2282
|
+
const spansWindow = win.length && sg.t1 - win[0].t0 > windowSec;
|
|
2283
|
+
const overBudget = win.length && estimateTokens(win.map((w) => w.text).join(" ") + sg.text) > maxTokens;
|
|
2284
|
+
if (startsNewChapter || spansWindow || overBudget) flush();
|
|
2285
|
+
win.push(sg);
|
|
2286
|
+
}
|
|
2287
|
+
flush();
|
|
2288
|
+
return out;
|
|
2289
|
+
}
|
|
1595
2290
|
var DEFAULT_MAX_TOKENS = 512;
|
|
1596
2291
|
function estimateTokens(text) {
|
|
1597
2292
|
return Math.ceil(text.length / 4);
|
|
@@ -1614,11 +2309,11 @@ function serializeElement(el) {
|
|
|
1614
2309
|
case "richtext":
|
|
1615
2310
|
return (e.runs || []).map((r) => r.text ?? "").join("").trim();
|
|
1616
2311
|
case "list": {
|
|
1617
|
-
const
|
|
2312
|
+
const walk2 = (items, depth = 0) => (items || []).flatMap((it) => {
|
|
1618
2313
|
const line = " ".repeat(depth) + "- " + String(it.content ?? "").trim();
|
|
1619
|
-
return it.children?.length ? [line, ...
|
|
2314
|
+
return it.children?.length ? [line, ...walk2(it.children, depth + 1)] : [line];
|
|
1620
2315
|
});
|
|
1621
|
-
return
|
|
2316
|
+
return walk2(e.items).join("\n");
|
|
1622
2317
|
}
|
|
1623
2318
|
case "table": {
|
|
1624
2319
|
const headers = e.headers ?? e.columns?.map((c) => c.header || "").filter((h) => h) ?? [];
|
|
@@ -1649,6 +2344,8 @@ function serializeElement(el) {
|
|
|
1649
2344
|
return `${e.checked ? "[x]" : "[ ]"} ${e.label ?? ""}`.trim();
|
|
1650
2345
|
case "image":
|
|
1651
2346
|
return e.alt ? `[image: ${e.alt}]` : "";
|
|
2347
|
+
case "video":
|
|
2348
|
+
return e.title ? `[video: ${e.title}]` : "";
|
|
1652
2349
|
case "toc":
|
|
1653
2350
|
case "shape":
|
|
1654
2351
|
case "signature":
|
|
@@ -1688,8 +2385,15 @@ function makeChunk(group, breadcrumb) {
|
|
|
1688
2385
|
function chunkDocument(doc, options = {}) {
|
|
1689
2386
|
const strategy = options.strategy ?? "section";
|
|
1690
2387
|
const maxTokens = options.maxTokens ?? DEFAULT_MAX_TOKENS;
|
|
2388
|
+
const windowSec = options.transcriptWindowSec ?? DEFAULT_TRANSCRIPT_WINDOW;
|
|
1691
2389
|
const flat = flatten(doc);
|
|
1692
2390
|
const chunks = [];
|
|
2391
|
+
const withTranscripts = (group, crumb2, c) => {
|
|
2392
|
+
if (c) chunks.push(c);
|
|
2393
|
+
for (const f of group) {
|
|
2394
|
+
if (f.el.type === "video") chunks.push(...transcriptChunks(f.el, f.id, f.page, crumb2, windowSec, maxTokens));
|
|
2395
|
+
}
|
|
2396
|
+
};
|
|
1693
2397
|
if (strategy === "element") {
|
|
1694
2398
|
const crumb2 = [];
|
|
1695
2399
|
for (const f of flat) {
|
|
@@ -1698,8 +2402,7 @@ function chunkDocument(doc, options = {}) {
|
|
|
1698
2402
|
crumb2.length = Math.max(0, lvl - 1);
|
|
1699
2403
|
crumb2[lvl - 1] = serializeElement(f.el);
|
|
1700
2404
|
}
|
|
1701
|
-
|
|
1702
|
-
if (c) chunks.push(c);
|
|
2405
|
+
withTranscripts([f], crumb2, makeChunk([f], crumb2));
|
|
1703
2406
|
}
|
|
1704
2407
|
return chunks;
|
|
1705
2408
|
}
|
|
@@ -1708,8 +2411,7 @@ function chunkDocument(doc, options = {}) {
|
|
|
1708
2411
|
let buf2 = [];
|
|
1709
2412
|
let bufTokens = 0;
|
|
1710
2413
|
const flush = () => {
|
|
1711
|
-
|
|
1712
|
-
if (c) chunks.push(c);
|
|
2414
|
+
withTranscripts(buf2, crumb2, makeChunk(buf2, crumb2));
|
|
1713
2415
|
buf2 = [];
|
|
1714
2416
|
bufTokens = 0;
|
|
1715
2417
|
};
|
|
@@ -1736,22 +2438,21 @@ function chunkDocument(doc, options = {}) {
|
|
|
1736
2438
|
for (const f of buf) {
|
|
1737
2439
|
const t = estimateTokens(serializeElement(f.el));
|
|
1738
2440
|
if (subTokens + t > maxTokens && sub.length > 0) {
|
|
1739
|
-
|
|
1740
|
-
if (c2) chunks.push(c2);
|
|
2441
|
+
withTranscripts(sub, crumb, makeChunk(sub, crumb));
|
|
1741
2442
|
sub = [];
|
|
1742
2443
|
subTokens = 0;
|
|
1743
2444
|
}
|
|
1744
2445
|
sub.push(f);
|
|
1745
2446
|
subTokens += t;
|
|
1746
2447
|
}
|
|
1747
|
-
|
|
1748
|
-
if (c) chunks.push(c);
|
|
2448
|
+
withTranscripts(sub, crumb, makeChunk(sub, crumb));
|
|
1749
2449
|
buf = [];
|
|
1750
2450
|
};
|
|
1751
2451
|
for (const f of flat) {
|
|
1752
2452
|
const lvl = headingLevel(f.el);
|
|
1753
2453
|
if (lvl != null) {
|
|
1754
|
-
|
|
2454
|
+
const onlyHeadings = buf.length > 0 && buf.every((b) => headingLevel(b.el) != null);
|
|
2455
|
+
if (!onlyHeadings) flushSection();
|
|
1755
2456
|
crumb.length = Math.max(0, lvl - 1);
|
|
1756
2457
|
crumb[lvl - 1] = serializeElement(f.el);
|
|
1757
2458
|
}
|
|
@@ -1762,34 +2463,34 @@ function chunkDocument(doc, options = {}) {
|
|
|
1762
2463
|
}
|
|
1763
2464
|
async function loadJdf(filePath) {
|
|
1764
2465
|
if (filePath.toLowerCase().endsWith(".jdfx")) {
|
|
1765
|
-
const zip = await JSZip.loadAsync(
|
|
2466
|
+
const zip = await JSZip.loadAsync(fs7.readFileSync(filePath));
|
|
1766
2467
|
const docFile = zip.file(JDFX_DOCUMENT_PATH);
|
|
1767
2468
|
if (!docFile) throw new Error(`Bundle missing ${JDFX_DOCUMENT_PATH}`);
|
|
1768
2469
|
return JSON.parse(await docFile.async("string"));
|
|
1769
2470
|
}
|
|
1770
|
-
return JSON.parse(
|
|
2471
|
+
return JSON.parse(fs7.readFileSync(filePath, "utf-8"));
|
|
1771
2472
|
}
|
|
1772
2473
|
async function chunkFile(inputPath, opts = {}) {
|
|
1773
|
-
const input =
|
|
1774
|
-
if (!
|
|
2474
|
+
const input = path7.resolve(inputPath);
|
|
2475
|
+
if (!fs7.existsSync(input)) throw new Error(`File not found: ${input}`);
|
|
1775
2476
|
const doc = await loadJdf(input);
|
|
1776
2477
|
const strategy = opts.strategy ?? "section";
|
|
1777
|
-
const chunks = chunkDocument(doc, { strategy, maxTokens: opts.maxTokens });
|
|
2478
|
+
const chunks = chunkDocument(doc, { strategy, maxTokens: opts.maxTokens, transcriptWindowSec: opts.transcriptWindowSec });
|
|
1778
2479
|
const format = opts.format ?? "jsonl";
|
|
1779
2480
|
console.log(`Chunking: ${input}`);
|
|
1780
2481
|
console.log(`Strategy: ${strategy}${opts.maxTokens ? ` (max ${opts.maxTokens} tokens)` : ""}`);
|
|
1781
2482
|
if (format === "inline") {
|
|
1782
|
-
const out = opts.output ?
|
|
2483
|
+
const out = opts.output ? path7.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".jdf");
|
|
1783
2484
|
const withIndex = { ...doc, index: { chunker: `jdf-${strategy}-v1`, chunks } };
|
|
1784
|
-
|
|
2485
|
+
fs7.writeFileSync(out, JSON.stringify(withIndex, null, 2));
|
|
1785
2486
|
console.log(`Output: ${out} (${chunks.length} chunks in "index" block)`);
|
|
1786
2487
|
} else if (format === "json") {
|
|
1787
|
-
const out = opts.output ?
|
|
1788
|
-
|
|
2488
|
+
const out = opts.output ? path7.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".chunks.json");
|
|
2489
|
+
fs7.writeFileSync(out, JSON.stringify(chunks, null, 2));
|
|
1789
2490
|
console.log(`Output: ${out} (${chunks.length} chunks)`);
|
|
1790
2491
|
} else {
|
|
1791
|
-
const out = opts.output ?
|
|
1792
|
-
|
|
2492
|
+
const out = opts.output ? path7.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".chunks.jsonl");
|
|
2493
|
+
fs7.writeFileSync(out, chunks.map((c) => JSON.stringify(c)).join("\n") + "\n");
|
|
1793
2494
|
console.log(`Output: ${out} (${chunks.length} chunks)`);
|
|
1794
2495
|
}
|
|
1795
2496
|
const totalTokens = chunks.reduce((a, c) => a + c.tokens, 0);
|
|
@@ -1961,17 +2662,17 @@ async function embedOpenAI(model, inputs) {
|
|
|
1961
2662
|
}
|
|
1962
2663
|
async function loadJdf2(filePath) {
|
|
1963
2664
|
if (filePath.toLowerCase().endsWith(".jdfx")) {
|
|
1964
|
-
const zip = await JSZip.loadAsync(
|
|
2665
|
+
const zip = await JSZip.loadAsync(fs7.readFileSync(filePath));
|
|
1965
2666
|
const docFile = zip.file(JDFX_DOCUMENT_PATH);
|
|
1966
2667
|
if (!docFile) throw new Error(`Bundle missing ${JDFX_DOCUMENT_PATH}`);
|
|
1967
2668
|
return JSON.parse(await docFile.async("string"));
|
|
1968
2669
|
}
|
|
1969
|
-
return JSON.parse(
|
|
2670
|
+
return JSON.parse(fs7.readFileSync(filePath, "utf-8"));
|
|
1970
2671
|
}
|
|
1971
2672
|
function loadCache(cachePath) {
|
|
1972
2673
|
try {
|
|
1973
|
-
if (!
|
|
1974
|
-
return JSON.parse(
|
|
2674
|
+
if (!fs7.existsSync(cachePath)) return null;
|
|
2675
|
+
return JSON.parse(fs7.readFileSync(cachePath, "utf-8"));
|
|
1975
2676
|
} catch {
|
|
1976
2677
|
return null;
|
|
1977
2678
|
}
|
|
@@ -1982,15 +2683,15 @@ function batched(items, size) {
|
|
|
1982
2683
|
return out;
|
|
1983
2684
|
}
|
|
1984
2685
|
async function embedFile(inputPath, opts = {}) {
|
|
1985
|
-
const input =
|
|
1986
|
-
if (!
|
|
2686
|
+
const input = path7.resolve(inputPath);
|
|
2687
|
+
if (!fs7.existsSync(input)) throw new Error(`File not found: ${input}`);
|
|
1987
2688
|
const provider = opts.provider ?? "ollama";
|
|
1988
2689
|
const model = opts.model ?? DEFAULT_MODEL[provider];
|
|
1989
2690
|
const strategy = opts.strategy ?? "section";
|
|
1990
2691
|
const doc = await loadJdf2(input);
|
|
1991
|
-
const chunks = chunkDocument(doc, { strategy, maxTokens: opts.maxTokens });
|
|
1992
|
-
const output = opts.output ?
|
|
1993
|
-
const cachePath = opts.cache ?
|
|
2692
|
+
const chunks = chunkDocument(doc, { strategy, maxTokens: opts.maxTokens, transcriptWindowSec: opts.transcriptWindowSec });
|
|
2693
|
+
const output = opts.output ? path7.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".embeddings.json");
|
|
2694
|
+
const cachePath = opts.cache ? path7.resolve(opts.cache) : output;
|
|
1994
2695
|
console.log(`Embedding: ${input}`);
|
|
1995
2696
|
console.log(`Provider: ${provider} / ${model}${provider === "ollama" ? " (local \u2014 no data leaves this machine)" : " (remote API)"}`);
|
|
1996
2697
|
console.log(`Strategy: ${strategy} \u2192 ${chunks.length} chunks`);
|
|
@@ -2029,11 +2730,295 @@ async function embedFile(inputPath, opts = {}) {
|
|
|
2029
2730
|
chunker: `jdf-${strategy}-v1`,
|
|
2030
2731
|
vectors
|
|
2031
2732
|
};
|
|
2032
|
-
|
|
2733
|
+
fs7.writeFileSync(output, JSON.stringify(sidecar));
|
|
2033
2734
|
console.log(`
|
|
2034
2735
|
Done! ${Object.keys(vectors).length} vectors (${dims}-dim) \u2192 ${output}`);
|
|
2035
2736
|
return sidecar;
|
|
2036
2737
|
}
|
|
2738
|
+
var toSec = (ts) => {
|
|
2739
|
+
const m = ts.trim().replace(",", ".").match(/^(?:(\d+):)?(\d{1,2}):(\d{2}(?:\.\d+)?)$/);
|
|
2740
|
+
if (!m) throw new Error(`bad timestamp "${ts}"`);
|
|
2741
|
+
return (m[1] ? Number(m[1]) * 3600 : 0) + Number(m[2]) * 60 + Number(m[3]);
|
|
2742
|
+
};
|
|
2743
|
+
function parseSubtitles(text, filename = "") {
|
|
2744
|
+
const trimmed = text.replace(/^/, "").trim();
|
|
2745
|
+
if (trimmed.startsWith("{") || trimmed.startsWith("[")) {
|
|
2746
|
+
const j = JSON.parse(trimmed);
|
|
2747
|
+
const arr = Array.isArray(j) ? j : Array.isArray(j.segments) ? j.segments : [];
|
|
2748
|
+
return arr.map((sg) => ({
|
|
2749
|
+
t0: Number(sg.t0 ?? sg.start ?? sg.from ?? 0),
|
|
2750
|
+
t1: Number(sg.t1 ?? sg.end ?? sg.to ?? 0),
|
|
2751
|
+
text: String(sg.text ?? "").trim(),
|
|
2752
|
+
...sg.speaker ? { speaker: String(sg.speaker) } : {}
|
|
2753
|
+
})).filter((sg) => sg.text);
|
|
2754
|
+
}
|
|
2755
|
+
const segs = [];
|
|
2756
|
+
for (const block of trimmed.split(/\r?\n\r?\n+/)) {
|
|
2757
|
+
const lines = block.split(/\r?\n/).filter((l) => l.trim() !== "" && l.trim() !== "WEBVTT");
|
|
2758
|
+
const ti = lines.findIndex((l) => l.includes("-->"));
|
|
2759
|
+
if (ti < 0) continue;
|
|
2760
|
+
const [a, b] = lines[ti].split("-->").map((x) => x.trim().split(/\s+/)[0]);
|
|
2761
|
+
const body = lines.slice(ti + 1).join(" ").replace(/<[^>]+>/g, "").replace(/\s+/g, " ").trim();
|
|
2762
|
+
if (!body) continue;
|
|
2763
|
+
segs.push({ t0: toSec(a), t1: toSec(b), text: body });
|
|
2764
|
+
}
|
|
2765
|
+
if (!segs.length) throw new Error(`no cues found in ${filename || "subtitle input"} (expected SRT, WebVTT or JSON segments)`);
|
|
2766
|
+
return segs;
|
|
2767
|
+
}
|
|
2768
|
+
function parseChapters(text) {
|
|
2769
|
+
const t = text.trim();
|
|
2770
|
+
if (t.startsWith("[")) return JSON.parse(t).map((c) => ({ t: Number(c.t ?? c.start ?? 0), title: String(c.title ?? "") }));
|
|
2771
|
+
return t.split(/\r?\n/).map((l) => l.trim()).filter(Boolean).map((l) => {
|
|
2772
|
+
const m = l.match(/^(\S+)\s+(.+)$/);
|
|
2773
|
+
if (!m) throw new Error(`bad chapter line "${l}" (expected "mm:ss Title")`);
|
|
2774
|
+
return { t: toSec(m[1]), title: m[2].trim() };
|
|
2775
|
+
});
|
|
2776
|
+
}
|
|
2777
|
+
async function loadDoc(file) {
|
|
2778
|
+
if (file.toLowerCase().endsWith(".jdfx")) {
|
|
2779
|
+
const zip = await JSZip.loadAsync(fs7.readFileSync(file));
|
|
2780
|
+
const f = zip.file(JDFX_DOCUMENT_PATH);
|
|
2781
|
+
if (!f) throw new Error(`Bundle missing ${JDFX_DOCUMENT_PATH}`);
|
|
2782
|
+
const doc = JSON.parse(await f.async("string"));
|
|
2783
|
+
const manifest = zip.file("manifest.json") ? JSON.parse(await zip.file("manifest.json").async("string")) : { assets: [] };
|
|
2784
|
+
for (const a of manifest.assets ?? []) {
|
|
2785
|
+
const af = zip.file(a.path);
|
|
2786
|
+
if (!af) continue;
|
|
2787
|
+
const data = (await af.async("nodebuffer")).toString("base64");
|
|
2788
|
+
const res = { src: "embedded", mimeType: a.mimeType, data };
|
|
2789
|
+
doc.resources ??= {};
|
|
2790
|
+
if (/^video\//i.test(a.mimeType || "")) (doc.resources.videos ??= {})[a.id] = res;
|
|
2791
|
+
else (doc.resources.images ??= {})[a.id] = res;
|
|
2792
|
+
}
|
|
2793
|
+
return { doc, bundle: true, zip };
|
|
2794
|
+
}
|
|
2795
|
+
return { doc: JSON.parse(fs7.readFileSync(file, "utf-8")), bundle: false };
|
|
2796
|
+
}
|
|
2797
|
+
function findVideos(doc) {
|
|
2798
|
+
const out = [];
|
|
2799
|
+
const walk2 = (els, page) => {
|
|
2800
|
+
for (const el of els ?? []) {
|
|
2801
|
+
if (el?.type === "video") out.push({ el, page, index: out.length });
|
|
2802
|
+
if (el?.elements) walk2(el.elements, page);
|
|
2803
|
+
}
|
|
2804
|
+
};
|
|
2805
|
+
doc.pages.forEach((p, i) => walk2(p.elements, i + 1));
|
|
2806
|
+
return out;
|
|
2807
|
+
}
|
|
2808
|
+
async function clipToTempFile(doc, el, docDir) {
|
|
2809
|
+
const tmp = path7.join(fs7.mkdtempSync(path7.join(__require("os").tmpdir(), "jdf-transcribe-")), "clip.mp4");
|
|
2810
|
+
const res = el.resource ? doc.resources?.videos?.[el.resource] ?? doc.resources?.images?.[el.resource] : void 0;
|
|
2811
|
+
if (res?.data) {
|
|
2812
|
+
fs7.writeFileSync(tmp, Buffer.from(res.data.replace(/^data:[^,]*,/, ""), "base64"));
|
|
2813
|
+
return tmp;
|
|
2814
|
+
}
|
|
2815
|
+
if (res?.path) return path7.resolve(docDir, res.path);
|
|
2816
|
+
const src = el.src;
|
|
2817
|
+
if (!src) return null;
|
|
2818
|
+
if (src.startsWith("data:")) {
|
|
2819
|
+
fs7.writeFileSync(tmp, Buffer.from(src.replace(/^data:[^,]*,/, ""), "base64"));
|
|
2820
|
+
return tmp;
|
|
2821
|
+
}
|
|
2822
|
+
if (/^https?:\/\//i.test(src)) {
|
|
2823
|
+
const r = await fetch(src);
|
|
2824
|
+
if (!r.ok) throw new Error(`download failed ${r.status}: ${src}`);
|
|
2825
|
+
fs7.writeFileSync(tmp, Buffer.from(await r.arrayBuffer()));
|
|
2826
|
+
return tmp;
|
|
2827
|
+
}
|
|
2828
|
+
const local = path7.resolve(docDir, src);
|
|
2829
|
+
return fs7.existsSync(local) ? local : null;
|
|
2830
|
+
}
|
|
2831
|
+
function whisperCli(clip, model, language, prompt2) {
|
|
2832
|
+
const ffmpeg = spawnSync("ffmpeg", ["-version"]);
|
|
2833
|
+
if (ffmpeg.error) throw new Error("ffmpeg not found \u2014 needed to extract audio for whisper-cli (brew install ffmpeg)");
|
|
2834
|
+
const wav = clip.replace(/\.[^.]+$/, "") + ".16k.wav";
|
|
2835
|
+
const ex = spawnSync("ffmpeg", ["-y", "-i", clip, "-vn", "-ac", "1", "-ar", "16000", "-f", "wav", wav], { encoding: "utf-8" });
|
|
2836
|
+
if (ex.status !== 0) throw new Error(`ffmpeg failed: ${ex.stderr.slice(-400)}`);
|
|
2837
|
+
const args = ["-f", wav, "-oj", "-of", wav.replace(/\.wav$/, "")];
|
|
2838
|
+
if (model) args.push("-m", model);
|
|
2839
|
+
if (language) args.push("-l", language);
|
|
2840
|
+
if (prompt2) args.push("--prompt", prompt2);
|
|
2841
|
+
const run = spawnSync("whisper-cli", args, { encoding: "utf-8" });
|
|
2842
|
+
if (run.error) throw new Error("whisper-cli not found \u2014 install whisper.cpp (brew install whisper-cpp) or use --from / --provider openai");
|
|
2843
|
+
if (run.status !== 0) throw new Error(`whisper-cli failed: ${run.stderr.slice(-400)}`);
|
|
2844
|
+
const j = JSON.parse(fs7.readFileSync(wav.replace(/\.wav$/, "") + ".json", "utf-8"));
|
|
2845
|
+
const segs = j.transcription ?? j.segments ?? [];
|
|
2846
|
+
const ms = (x) => typeof x === "number" ? x / 1e3 : toSec(String(x).replace(",", "."));
|
|
2847
|
+
return segs.map((sg) => ({ t0: ms(sg.offsets?.from ?? sg.start), t1: ms(sg.offsets?.to ?? sg.end), text: String(sg.text ?? "").trim() })).filter((sg) => sg.text);
|
|
2848
|
+
}
|
|
2849
|
+
async function openaiTranscribe(clip, model, language, prompt2) {
|
|
2850
|
+
const key = process.env.OPENAI_API_KEY;
|
|
2851
|
+
if (!key) throw new Error("OPENAI_API_KEY is not set");
|
|
2852
|
+
const base = process.env.OPENAI_BASE_URL || "https://api.openai.com/v1";
|
|
2853
|
+
const form = new FormData();
|
|
2854
|
+
form.append("file", new Blob([fs7.readFileSync(clip)]), path7.basename(clip));
|
|
2855
|
+
form.append("model", model || "whisper-1");
|
|
2856
|
+
form.append("response_format", "verbose_json");
|
|
2857
|
+
form.append("timestamp_granularities[]", "segment");
|
|
2858
|
+
if (language) form.append("language", language);
|
|
2859
|
+
if (prompt2) form.append("prompt", prompt2);
|
|
2860
|
+
const r = await fetch(`${base}/audio/transcriptions`, { method: "POST", headers: { Authorization: `Bearer ${key}` }, body: form });
|
|
2861
|
+
if (!r.ok) throw new Error(`OpenAI transcription failed ${r.status}: ${(await r.text()).slice(0, 300)}`);
|
|
2862
|
+
const j = await r.json();
|
|
2863
|
+
return (j.segments ?? []).map((sg) => ({ t0: Number(sg.start), t1: Number(sg.end), text: String(sg.text).trim() })).filter((sg) => sg.text);
|
|
2864
|
+
}
|
|
2865
|
+
async function transcribeFile(inputPath, opts = {}) {
|
|
2866
|
+
const input = path7.resolve(inputPath);
|
|
2867
|
+
if (!fs7.existsSync(input)) throw new Error(`File not found: ${input}`);
|
|
2868
|
+
const { doc, bundle } = await loadDoc(input);
|
|
2869
|
+
const videos = findVideos(doc);
|
|
2870
|
+
if (!videos.length) throw new Error("document has no video element");
|
|
2871
|
+
let target = videos[0];
|
|
2872
|
+
if (opts.element != null) {
|
|
2873
|
+
const byId = videos.find((v) => v.el.id === opts.element);
|
|
2874
|
+
const byIdx = /^\d+$/.test(opts.element) ? videos[Number(opts.element)] : void 0;
|
|
2875
|
+
target = byId ?? byIdx ?? (() => {
|
|
2876
|
+
throw new Error(`no video element "${opts.element}" (have: ${videos.map((v) => v.el.id ?? `#${v.index}`).join(", ")})`);
|
|
2877
|
+
})();
|
|
2878
|
+
} else if (videos.length > 1) {
|
|
2879
|
+
throw new Error(`document has ${videos.length} videos \u2014 pick one with --element <id|index>`);
|
|
2880
|
+
}
|
|
2881
|
+
let segments;
|
|
2882
|
+
let source;
|
|
2883
|
+
if (opts.from) {
|
|
2884
|
+
segments = parseSubtitles(fs7.readFileSync(path7.resolve(opts.from), "utf-8"), opts.from);
|
|
2885
|
+
source = `${path7.extname(opts.from).slice(1).toLowerCase() || "file"}-import`;
|
|
2886
|
+
} else {
|
|
2887
|
+
const provider = opts.provider ?? "whisper-cli";
|
|
2888
|
+
const clip = await clipToTempFile(doc, target.el, path7.dirname(input));
|
|
2889
|
+
if (!clip) throw new Error("could not locate the clip bytes (no bundled asset, data URL, local path or http URL) \u2014 use --from to import subtitles instead");
|
|
2890
|
+
segments = provider === "openai" ? await openaiTranscribe(clip, opts.model, opts.language, opts.prompt) : whisperCli(clip, opts.model, opts.language, opts.prompt);
|
|
2891
|
+
source = provider === "openai" ? `openai:${opts.model || "whisper-1"}` : `whisper-cli${opts.model ? ":" + path7.basename(opts.model) : ""}`;
|
|
2892
|
+
}
|
|
2893
|
+
segments.sort((a, b) => a.t0 - b.t0);
|
|
2894
|
+
const transcript = { ...opts.language ? { language: opts.language } : {}, source, created: (/* @__PURE__ */ new Date()).toISOString(), segments };
|
|
2895
|
+
target.el.transcript = transcript;
|
|
2896
|
+
if (opts.chapters) target.el.chapters = parseChapters(fs7.readFileSync(path7.resolve(opts.chapters), "utf-8"));
|
|
2897
|
+
if (!target.el.id) target.el.id = `video-${target.index + 1}`;
|
|
2898
|
+
const output = opts.output ? path7.resolve(opts.output) : input;
|
|
2899
|
+
if (output.toLowerCase().endsWith(".jdfx") || bundle && !opts.output) {
|
|
2900
|
+
const { bytes } = await packJdfx(doc);
|
|
2901
|
+
fs7.writeFileSync(output, bytes);
|
|
2902
|
+
} else {
|
|
2903
|
+
fs7.writeFileSync(output, JSON.stringify(doc, null, 2));
|
|
2904
|
+
}
|
|
2905
|
+
const dur = segments.length ? segments[segments.length - 1].t1 : 0;
|
|
2906
|
+
console.log(`Transcribed: ${path7.basename(input)} \u2192 element "${target.el.id}" (${segments.length} segments, ${Math.round(dur)} s, source ${source})`);
|
|
2907
|
+
if (target.el.chapters) console.log(`Chapters: ${target.el.chapters.length}`);
|
|
2908
|
+
console.log(`Output: ${output}
|
|
2909
|
+
Next: jdf chunk ${path7.basename(output)} # transcript \u2192 time-windowed chunks with media.t0/t1`);
|
|
2910
|
+
return transcript;
|
|
2911
|
+
}
|
|
2912
|
+
var CONFIG_NAME = "jdf.rag.json";
|
|
2913
|
+
var OUT_DIR = ".jdf-rag";
|
|
2914
|
+
function walk(dir, acc = []) {
|
|
2915
|
+
for (const ent of fs7.readdirSync(dir, { withFileTypes: true })) {
|
|
2916
|
+
if (ent.name === "node_modules" || ent.name === OUT_DIR || ent.name.startsWith(".")) continue;
|
|
2917
|
+
const p = path7.join(dir, ent.name);
|
|
2918
|
+
if (ent.isDirectory()) walk(p, acc);
|
|
2919
|
+
else if (/\.(jdf|jdfx)$/i.test(ent.name)) acc.push(p);
|
|
2920
|
+
}
|
|
2921
|
+
return acc.sort();
|
|
2922
|
+
}
|
|
2923
|
+
async function readDoc(file) {
|
|
2924
|
+
if (file.toLowerCase().endsWith(".jdfx")) {
|
|
2925
|
+
const zip = await JSZip.loadAsync(fs7.readFileSync(file));
|
|
2926
|
+
return JSON.parse(await zip.file(JDFX_DOCUMENT_PATH).async("string"));
|
|
2927
|
+
}
|
|
2928
|
+
return JSON.parse(fs7.readFileSync(file, "utf-8"));
|
|
2929
|
+
}
|
|
2930
|
+
function videosIn(doc) {
|
|
2931
|
+
const out = [];
|
|
2932
|
+
const w = (els) => {
|
|
2933
|
+
for (const el of els ?? []) {
|
|
2934
|
+
if (el?.type === "video") out.push({ id: el.id, hasTranscript: !!el.transcript?.segments?.length });
|
|
2935
|
+
if (el?.elements) w(el.elements);
|
|
2936
|
+
}
|
|
2937
|
+
};
|
|
2938
|
+
for (const p of doc.pages ?? []) w(p.elements);
|
|
2939
|
+
return out;
|
|
2940
|
+
}
|
|
2941
|
+
async function ragFolder(dirPath, cli = {}) {
|
|
2942
|
+
const dir = path7.resolve(dirPath);
|
|
2943
|
+
if (!fs7.existsSync(dir) || !fs7.statSync(dir).isDirectory()) throw new Error(`Not a directory: ${dir}`);
|
|
2944
|
+
const cfgPath = path7.join(dir, CONFIG_NAME);
|
|
2945
|
+
const cfg = fs7.existsSync(cfgPath) ? JSON.parse(fs7.readFileSync(cfgPath, "utf-8")) : {};
|
|
2946
|
+
const opts = { ...cfg, ...Object.fromEntries(Object.entries(cli).filter(([, v]) => v !== void 0)) };
|
|
2947
|
+
const provider = opts.provider ?? "ollama";
|
|
2948
|
+
const transcribe = opts.transcribe ?? "none";
|
|
2949
|
+
const outDir = path7.resolve(opts.out ?? path7.join(dir, OUT_DIR));
|
|
2950
|
+
const files = walk(dir);
|
|
2951
|
+
console.log(`jdf rag: ${dir}
|
|
2952
|
+
files: ${files.length} (.jdf/.jdfx)${fs7.existsSync(cfgPath) ? `
|
|
2953
|
+
config: ${CONFIG_NAME}` : ""}
|
|
2954
|
+
embeddings: ${opts.noEmbed ? "skipped (--no-embed)" : `${provider}${opts.model ? " / " + opts.model : ""}`}
|
|
2955
|
+
transcribe: ${transcribe}${opts.dryRun ? "\n DRY RUN \u2014 nothing written" : ""}
|
|
2956
|
+
`);
|
|
2957
|
+
if (!files.length) {
|
|
2958
|
+
console.log("Nothing to do.");
|
|
2959
|
+
return;
|
|
2960
|
+
}
|
|
2961
|
+
const manifest = { dir, created: (/* @__PURE__ */ new Date()).toISOString(), provider: opts.noEmbed ? null : provider, model: opts.model ?? null, strategy: opts.strategy ?? "section", transcribe, files: [], totals: { files: files.length, chunks: 0, videoChunks: 0, videos: 0, transcribed: 0, untranscribed: 0 } };
|
|
2962
|
+
const indexLines = [];
|
|
2963
|
+
for (const file of files) {
|
|
2964
|
+
const rel = path7.relative(dir, file);
|
|
2965
|
+
const doc = await readDoc(file);
|
|
2966
|
+
const vids = videosIn(doc);
|
|
2967
|
+
let transcribedHere = 0;
|
|
2968
|
+
for (const v of vids) {
|
|
2969
|
+
if (v.hasTranscript) continue;
|
|
2970
|
+
if (transcribe === "none") {
|
|
2971
|
+
manifest.totals.untranscribed++;
|
|
2972
|
+
continue;
|
|
2973
|
+
}
|
|
2974
|
+
if (opts.dryRun) {
|
|
2975
|
+
transcribedHere++;
|
|
2976
|
+
continue;
|
|
2977
|
+
}
|
|
2978
|
+
try {
|
|
2979
|
+
await transcribeFile(file, { provider: transcribe, element: v.id, model: opts.transcribeModel, language: opts.language, prompt: opts.prompt });
|
|
2980
|
+
transcribedHere++;
|
|
2981
|
+
} catch (e) {
|
|
2982
|
+
console.warn(` ! ${rel}: transcription failed for video ${v.id ?? "#?"}: ${e.message}`);
|
|
2983
|
+
manifest.totals.untranscribed++;
|
|
2984
|
+
}
|
|
2985
|
+
}
|
|
2986
|
+
manifest.totals.videos += vids.length;
|
|
2987
|
+
manifest.totals.transcribed += transcribedHere;
|
|
2988
|
+
if (opts.dryRun) {
|
|
2989
|
+
manifest.files.push({ file: rel, videos: vids.length, wouldTranscribe: transcribedHere });
|
|
2990
|
+
continue;
|
|
2991
|
+
}
|
|
2992
|
+
const chunkOpts = { strategy: opts.strategy, maxTokens: opts.maxTokens, transcriptWindowSec: opts.transcriptWindowSec };
|
|
2993
|
+
const chunkOut = path7.join(outDir, "chunks", rel.replace(/\.(jdf|jdfx)$/i, ".chunks.jsonl"));
|
|
2994
|
+
fs7.mkdirSync(path7.dirname(chunkOut), { recursive: true });
|
|
2995
|
+
let chunks;
|
|
2996
|
+
if (opts.noEmbed) {
|
|
2997
|
+
chunks = await chunkFile(file, { ...chunkOpts, format: "jsonl", output: chunkOut });
|
|
2998
|
+
} else {
|
|
2999
|
+
const side = await embedFile(file, { ...chunkOpts, provider, model: opts.model, incremental: true });
|
|
3000
|
+
chunks = await chunkFile(file, { ...chunkOpts, format: "jsonl", output: chunkOut });
|
|
3001
|
+
manifest.files.push({ file: rel, chunks: chunks.length, vectors: Object.keys(side.vectors).length, sidecar: path7.relative(dir, file.replace(/\.(jdf|jdfx)$/i, ".embeddings.json")), videos: vids.length, transcribed: transcribedHere });
|
|
3002
|
+
}
|
|
3003
|
+
if (opts.noEmbed) manifest.files.push({ file: rel, chunks: chunks.length, videos: vids.length, transcribed: transcribedHere });
|
|
3004
|
+
for (const c of chunks) {
|
|
3005
|
+
indexLines.push(JSON.stringify({ file: rel, ...c }));
|
|
3006
|
+
manifest.totals.chunks++;
|
|
3007
|
+
if (c.media) manifest.totals.videoChunks++;
|
|
3008
|
+
}
|
|
3009
|
+
}
|
|
3010
|
+
if (!opts.dryRun) {
|
|
3011
|
+
fs7.mkdirSync(outDir, { recursive: true });
|
|
3012
|
+
fs7.writeFileSync(path7.join(outDir, "index.jsonl"), indexLines.join("\n") + (indexLines.length ? "\n" : ""));
|
|
3013
|
+
fs7.writeFileSync(path7.join(outDir, "manifest.json"), JSON.stringify(manifest, null, 2) + "\n");
|
|
3014
|
+
}
|
|
3015
|
+
const t = manifest.totals;
|
|
3016
|
+
console.log(`
|
|
3017
|
+
Done. ${t.files} files \u2192 ${t.chunks} chunks (${t.videoChunks} from video transcripts); videos ${t.videos}, transcribed now ${t.transcribed}, still without transcript ${t.untranscribed}${t.untranscribed && transcribe === "none" ? " (pass --transcribe whisper-cli|openai, or jdf transcribe --from subs.srt)" : ""}.`);
|
|
3018
|
+
if (!opts.dryRun) console.log(`Index: ${path7.join(outDir, "index.jsonl")}
|
|
3019
|
+
Report: ${path7.join(outDir, "manifest.json")}${opts.noEmbed ? "" : `
|
|
3020
|
+
Vectors: one <file>.embeddings.json next to each document (incremental \u2014 re-run any time)`}`);
|
|
3021
|
+
}
|
|
2037
3022
|
|
|
2038
3023
|
// src/index.ts
|
|
2039
3024
|
var HELP = `jdf \u2014 JSON Document Format CLI
|
|
@@ -2045,12 +3030,18 @@ The CLI exists for these workflows:
|
|
|
2045
3030
|
into a validated .jdf (or .jdfx) you can ship.
|
|
2046
3031
|
\u2022 JDF \u2192 chunks turn a document into retrieval-ready chunks (RAG).
|
|
2047
3032
|
\u2022 JDF \u2192 vectors embed those chunks, incrementally, for a vector store.
|
|
3033
|
+
\u2022 video \u2192 text attach a time-stamped transcript to a video element so
|
|
3034
|
+
RAG retrieves "video at 02:13", not just "a video".
|
|
3035
|
+
\u2022 folder \u2192 index one command over a directory of .jdf/.jdfx: transcribe,
|
|
3036
|
+
chunk, embed incrementally, write .jdf-rag/index.jsonl.
|
|
2048
3037
|
|
|
2049
3038
|
Usage:
|
|
2050
3039
|
jdf validate <file.jdf>
|
|
2051
|
-
jdf convert <file.{pdf,json,md}> [-o output.{jdf,jdfx}] [--json]
|
|
3040
|
+
jdf convert <file.{pdf,json,md}> [-o output.{jdf,jdfx}] [--json] [--password PW] [--drop-invisible-text]
|
|
2052
3041
|
jdf chunk <file.{jdf,jdfx}> [--strategy section|element|fixed] [--format jsonl|json|inline] [--max-tokens N] [-o out]
|
|
2053
3042
|
jdf embed <file.{jdf,jdfx}> [--provider ollama|openai] [--model NAME] [--strategy \u2026] [--incremental] [-o out]
|
|
3043
|
+
jdf transcribe <file.{jdf,jdfx}> [--from subs.srt|.vtt|.json] [--provider whisper-cli|openai] [--element ID] [--chapters FILE] [-o out]
|
|
3044
|
+
jdf rag <dir> [--provider ollama|openai] [--model NAME] [--transcribe none|whisper-cli|openai] [--no-embed] [--dry-run] [--out DIR]
|
|
2054
3045
|
jdf --help
|
|
2055
3046
|
|
|
2056
3047
|
Commands:
|
|
@@ -2058,18 +3049,38 @@ Commands:
|
|
|
2058
3049
|
convert Convert a PDF, JSON, or Markdown file into JDF (alias: import)
|
|
2059
3050
|
chunk Split a JDF document into retrieval-ready chunks (offline, deterministic)
|
|
2060
3051
|
embed Compute embeddings for the chunks (local via Ollama by default)
|
|
3052
|
+
transcribe Store time-stamped text on a video element (import SRT/VTT/JSON, or run Whisper)
|
|
3053
|
+
rag Make a whole folder retrieval-ready (finds .jdf/.jdfx, transcribes, chunks, embeds, indexes)
|
|
2061
3054
|
|
|
2062
3055
|
Flags:
|
|
2063
3056
|
-o, --output <path> Explicit output path
|
|
2064
3057
|
--json convert: force pure JSON .jdf output (inline base64
|
|
2065
3058
|
instead of a .jdfx zip bundle)
|
|
3059
|
+
--password <pw> convert(pdf): password for an encrypted PDF
|
|
3060
|
+
--drop-invisible-text
|
|
3061
|
+
convert(pdf): omit invisible (OCR-layer) text; by
|
|
3062
|
+
default it is kept with opacity 0 so RAG / search
|
|
3063
|
+
still see the words of a scanned PDF
|
|
2066
3064
|
--strategy <s> chunk/embed: section (default) | element | fixed
|
|
2067
3065
|
--format <f> chunk: jsonl (default) | json | inline
|
|
2068
3066
|
--max-tokens <n> chunk/embed: soft cap per chunk (default 512)
|
|
2069
3067
|
--provider <p> embed: ollama (default, local) | openai (remote API)
|
|
2070
3068
|
--model <name> embed: model id (default: nomic-embed-text / text-embedding-3-small)
|
|
2071
3069
|
--incremental embed: skip chunks whose content hash is unchanged
|
|
3070
|
+
--cache <path> embed: sidecar to reuse vectors from (default: the
|
|
3071
|
+
output path itself)
|
|
2072
3072
|
--no-auto-start embed(ollama): don't auto-launch Ollama via Docker
|
|
3073
|
+
--from <file> transcribe: import subtitles (.srt / .vtt / JSON segments) \u2014 offline, no model
|
|
3074
|
+
--element <id|n> transcribe: which video element (id, or 0-based index); default the only one
|
|
3075
|
+
--chapters <file> transcribe: JSON [{t,title}] or "mm:ss Title" lines \u2192 chapter breadcrumbs
|
|
3076
|
+
--language <tag> transcribe: BCP-47 language hint for Whisper
|
|
3077
|
+
--prompt <text> transcribe: Whisper vocabulary hint (names, acronyms) \u2014 not an instruction
|
|
3078
|
+
--window <sec> chunk/embed/rag: transcript window per video chunk (default 45)
|
|
3079
|
+
--transcribe <p> rag: none (default) | whisper-cli | openai \u2014 for videos that have no transcript yet
|
|
3080
|
+
--no-embed rag: chunk + index only
|
|
3081
|
+
--dry-run rag: list what would happen, write nothing
|
|
3082
|
+
--out <dir> rag: index folder (default <dir>/.jdf-rag)
|
|
3083
|
+
rag reads defaults from <dir>/jdf.rag.json (same keys as the flags; flags win)
|
|
2073
3084
|
|
|
2074
3085
|
Environment (embed):
|
|
2075
3086
|
ollama: OLLAMA_HOST (default http://localhost:11434)
|
|
@@ -2083,8 +3094,11 @@ Examples:
|
|
|
2083
3094
|
jdf chunk report.jdf --format inline # embed the chunk index into the .jdf
|
|
2084
3095
|
jdf embed report.jdf # local embeddings via Ollama (auto-setup)
|
|
2085
3096
|
jdf embed report.jdf --provider openai --incremental
|
|
3097
|
+
jdf transcribe talk.jdfx --from talk.srt --chapters chapters.txt # then: jdf chunk talk.jdfx
|
|
3098
|
+
jdf transcribe talk.jdfx --provider openai --language en --prompt "JDF, jdfx, Ollama"
|
|
3099
|
+
jdf rag ./knowledge-base --transcribe openai # whole folder \u2192 .jdf-rag/index.jsonl
|
|
2086
3100
|
`;
|
|
2087
|
-
var BOOLEAN_FLAGS = /* @__PURE__ */ new Set(["help", "h", "json", "verbose", "skip-validate", "incremental", "no-auto-start"]);
|
|
3101
|
+
var BOOLEAN_FLAGS = /* @__PURE__ */ new Set(["help", "h", "json", "verbose", "skip-validate", "incremental", "no-auto-start", "drop-invisible-text", "no-embed", "dry-run"]);
|
|
2088
3102
|
function parseArgs(argv) {
|
|
2089
3103
|
const positional = [];
|
|
2090
3104
|
const flags = {};
|
|
@@ -2164,7 +3178,11 @@ async function main() {
|
|
|
2164
3178
|
await importMarkdown(input, output);
|
|
2165
3179
|
process.exit(0);
|
|
2166
3180
|
} else if (lower.endsWith(".pdf")) {
|
|
2167
|
-
await importPdf(input, output, {
|
|
3181
|
+
await importPdf(input, output, {
|
|
3182
|
+
forceJson,
|
|
3183
|
+
password: typeof flags.password === "string" ? flags.password : void 0,
|
|
3184
|
+
dropInvisibleText: flags["drop-invisible-text"] === true
|
|
3185
|
+
});
|
|
2168
3186
|
process.exit(0);
|
|
2169
3187
|
} else if (lower.endsWith(".json")) {
|
|
2170
3188
|
await importJson(input, output, { forceJson });
|
|
@@ -2184,14 +3202,55 @@ async function main() {
|
|
|
2184
3202
|
strategy: typeof flags.strategy === "string" ? flags.strategy : void 0,
|
|
2185
3203
|
format: typeof flags.format === "string" ? flags.format : void 0,
|
|
2186
3204
|
maxTokens: typeof flags["max-tokens"] === "string" ? parseInt(flags["max-tokens"], 10) : void 0,
|
|
3205
|
+
transcriptWindowSec: typeof flags.window === "string" ? parseInt(flags.window, 10) : void 0,
|
|
2187
3206
|
output: typeof flags.output === "string" ? flags.output : void 0
|
|
2188
3207
|
});
|
|
2189
3208
|
process.exit(0);
|
|
2190
3209
|
}
|
|
3210
|
+
case "transcribe": {
|
|
3211
|
+
const input = positional[0];
|
|
3212
|
+
if (!input) {
|
|
3213
|
+
console.error("Usage: jdf transcribe <file.{jdf,jdfx}> [--from subs.srt|.vtt|.json] [--provider whisper-cli|openai] [--model M] [--language tag] [--element id|n] [--chapters file] [-o out]");
|
|
3214
|
+
process.exit(1);
|
|
3215
|
+
}
|
|
3216
|
+
await transcribeFile(input, {
|
|
3217
|
+
from: typeof flags.from === "string" ? flags.from : void 0,
|
|
3218
|
+
provider: typeof flags.provider === "string" ? flags.provider : void 0,
|
|
3219
|
+
model: typeof flags.model === "string" ? flags.model : void 0,
|
|
3220
|
+
language: typeof flags.language === "string" ? flags.language : void 0,
|
|
3221
|
+
prompt: typeof flags.prompt === "string" ? flags.prompt : void 0,
|
|
3222
|
+
element: typeof flags.element === "string" ? flags.element : void 0,
|
|
3223
|
+
chapters: typeof flags.chapters === "string" ? flags.chapters : void 0,
|
|
3224
|
+
output: typeof flags.output === "string" ? flags.output : void 0
|
|
3225
|
+
});
|
|
3226
|
+
process.exit(0);
|
|
3227
|
+
}
|
|
3228
|
+
case "rag": {
|
|
3229
|
+
const input = positional[0];
|
|
3230
|
+
if (!input) {
|
|
3231
|
+
console.error("Usage: jdf rag <dir> [--provider ollama|openai] [--model NAME] [--strategy \u2026] [--window sec] [--transcribe none|whisper-cli|openai] [--language tag] [--prompt text] [--no-embed] [--dry-run] [--out DIR]");
|
|
3232
|
+
process.exit(1);
|
|
3233
|
+
}
|
|
3234
|
+
await ragFolder(input, {
|
|
3235
|
+
provider: typeof flags.provider === "string" ? flags.provider : void 0,
|
|
3236
|
+
model: typeof flags.model === "string" ? flags.model : void 0,
|
|
3237
|
+
strategy: typeof flags.strategy === "string" ? flags.strategy : void 0,
|
|
3238
|
+
maxTokens: typeof flags["max-tokens"] === "string" ? parseInt(flags["max-tokens"], 10) : void 0,
|
|
3239
|
+
transcriptWindowSec: typeof flags.window === "string" ? parseInt(flags.window, 10) : void 0,
|
|
3240
|
+
transcribe: typeof flags.transcribe === "string" ? flags.transcribe : void 0,
|
|
3241
|
+
transcribeModel: typeof flags["transcribe-model"] === "string" ? flags["transcribe-model"] : void 0,
|
|
3242
|
+
language: typeof flags.language === "string" ? flags.language : void 0,
|
|
3243
|
+
prompt: typeof flags.prompt === "string" ? flags.prompt : void 0,
|
|
3244
|
+
noEmbed: flags["no-embed"] === true,
|
|
3245
|
+
dryRun: flags["dry-run"] === true,
|
|
3246
|
+
out: typeof flags.out === "string" ? flags.out : void 0
|
|
3247
|
+
});
|
|
3248
|
+
process.exit(0);
|
|
3249
|
+
}
|
|
2191
3250
|
case "embed": {
|
|
2192
3251
|
const input = positional[0];
|
|
2193
3252
|
if (!input) {
|
|
2194
|
-
console.error("Usage: jdf embed <file.{jdf,jdfx}> [--provider ollama|openai] [--model NAME] [--strategy \u2026] [--incremental] [--no-auto-start] [-o out]");
|
|
3253
|
+
console.error("Usage: jdf embed <file.{jdf,jdfx}> [--provider ollama|openai] [--model NAME] [--strategy \u2026] [--incremental] [--cache prev.embeddings.json] [--no-auto-start] [-o out]");
|
|
2195
3254
|
process.exit(1);
|
|
2196
3255
|
}
|
|
2197
3256
|
await embedFile(input, {
|
|
@@ -2200,8 +3259,10 @@ async function main() {
|
|
|
2200
3259
|
strategy: typeof flags.strategy === "string" ? flags.strategy : void 0,
|
|
2201
3260
|
maxTokens: typeof flags["max-tokens"] === "string" ? parseInt(flags["max-tokens"], 10) : void 0,
|
|
2202
3261
|
incremental: flags.incremental === true,
|
|
3262
|
+
transcriptWindowSec: typeof flags.window === "string" ? parseInt(flags.window, 10) : void 0,
|
|
2203
3263
|
autoStart: flags["no-auto-start"] !== true,
|
|
2204
|
-
output: typeof flags.output === "string" ? flags.output : void 0
|
|
3264
|
+
output: typeof flags.output === "string" ? flags.output : void 0,
|
|
3265
|
+
cache: typeof flags.cache === "string" ? flags.cache : void 0
|
|
2205
3266
|
});
|
|
2206
3267
|
process.exit(0);
|
|
2207
3268
|
}
|