@uurtech/jdf-cli 0.1.25 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -1,13 +1,20 @@
1
1
  #!/usr/bin/env node
2
- import fs from 'fs';
3
- import path2 from 'path';
2
+ import fs7 from 'fs';
3
+ import path7 from 'path';
4
4
  import { fileURLToPath } from 'url';
5
5
  import Ajv from 'ajv';
6
6
  import addFormats from 'ajv-formats';
7
7
  import JSZip from 'jszip';
8
8
  import crypto, { createHash } from 'crypto';
9
9
  import { readFile } from 'fs/promises';
10
- import { execFileSync } from 'child_process';
10
+ import { execFileSync, spawnSync } from 'child_process';
11
+
12
+ var __require = /* @__PURE__ */ ((x) => typeof require !== "undefined" ? require : typeof Proxy !== "undefined" ? new Proxy(x, {
13
+ get: (a, b) => (typeof require !== "undefined" ? require : a)[b]
14
+ }) : x)(function(x) {
15
+ if (typeof require !== "undefined") return require.apply(this, arguments);
16
+ throw Error('Dynamic require of "' + x + '" is not supported');
17
+ });
11
18
 
12
19
  // ../../packages/jdf-core/src/manifest.ts
13
20
  var JDFX_MANIFEST_VERSION = "1.0.0";
@@ -16,17 +23,17 @@ var JDFX_MANIFEST_PATH = "manifest.json";
16
23
  var JDFX_ASSET_DIR = "assets";
17
24
 
18
25
  // src/commands/validate.ts
19
- var __dirname$1 = path2.dirname(fileURLToPath(import.meta.url));
26
+ var __dirname$1 = path7.dirname(fileURLToPath(import.meta.url));
20
27
  function resolveSchemaPath() {
21
- const bundled = path2.resolve(__dirname$1, "jdf-schema.json");
22
- if (fs.existsSync(bundled)) return bundled;
23
- const dev = path2.resolve(__dirname$1, "../../../../spec/jdf-schema.json");
28
+ const bundled = path7.resolve(__dirname$1, "jdf-schema.json");
29
+ if (fs7.existsSync(bundled)) return bundled;
30
+ const dev = path7.resolve(__dirname$1, "../../../../spec/jdf-schema.json");
24
31
  return dev;
25
32
  }
26
33
  var SCHEMA_PATH = resolveSchemaPath();
27
34
  async function loadDocument(filePath) {
28
35
  if (filePath.toLowerCase().endsWith(".jdfx")) {
29
- const zip = await JSZip.loadAsync(fs.readFileSync(filePath));
36
+ const zip = await JSZip.loadAsync(fs7.readFileSync(filePath));
30
37
  const docFile = zip.file(JDFX_DOCUMENT_PATH);
31
38
  if (!docFile) {
32
39
  console.error(`\u2717 Bundle missing ${JDFX_DOCUMENT_PATH}`);
@@ -51,11 +58,11 @@ async function loadDocument(filePath) {
51
58
  }
52
59
  return { doc, bundle: { manifest, assetCount } };
53
60
  }
54
- return { doc: JSON.parse(fs.readFileSync(filePath, "utf-8")) };
61
+ return { doc: JSON.parse(fs7.readFileSync(filePath, "utf-8")) };
55
62
  }
56
63
  async function validate(file) {
57
- const filePath = path2.resolve(file);
58
- if (!fs.existsSync(filePath)) {
64
+ const filePath = path7.resolve(file);
65
+ if (!fs7.existsSync(filePath)) {
59
66
  console.error(`File not found: ${filePath}`);
60
67
  return false;
61
68
  }
@@ -68,11 +75,11 @@ async function validate(file) {
68
75
  }
69
76
  if (!loaded) return false;
70
77
  const { doc, bundle } = loaded;
71
- if (!fs.existsSync(SCHEMA_PATH)) {
78
+ if (!fs7.existsSync(SCHEMA_PATH)) {
72
79
  console.error(`Schema not found at ${SCHEMA_PATH}`);
73
80
  return false;
74
81
  }
75
- const schema = JSON.parse(fs.readFileSync(SCHEMA_PATH, "utf-8"));
82
+ const schema = JSON.parse(fs7.readFileSync(SCHEMA_PATH, "utf-8"));
76
83
  const ajv = new Ajv({ allErrors: true, strict: false });
77
84
  addFormats(ajv);
78
85
  const validateFn = ajv.compile(schema);
@@ -81,7 +88,7 @@ async function validate(file) {
81
88
  const d = doc;
82
89
  const pageCount = Array.isArray(d.pages) ? d.pages.length : 0;
83
90
  const elCount = Array.isArray(d.pages) ? d.pages.reduce((acc, p) => acc + (Array.isArray(p?.elements) ? p.elements.length : 0), 0) : 0;
84
- console.log(`\u2713 Valid: ${path2.basename(filePath)}`);
91
+ console.log(`\u2713 Valid: ${path7.basename(filePath)}`);
85
92
  console.log(` Format: ${d.$jdf}${bundle ? " (jdfx bundle)" : ""}`);
86
93
  console.log(` Title: ${d.meta?.title}`);
87
94
  console.log(` Pages: ${pageCount}`);
@@ -94,7 +101,7 @@ async function validate(file) {
94
101
  }
95
102
  return true;
96
103
  }
97
- console.error(`\u2717 Invalid: ${path2.basename(filePath)}`);
104
+ console.error(`\u2717 Invalid: ${path7.basename(filePath)}`);
98
105
  for (const err of validateFn.errors || []) {
99
106
  const loc = err.instancePath || "(root)";
100
107
  console.error(` ${loc} \u2014 ${err.message}`);
@@ -109,15 +116,15 @@ function hashBytes(bytes) {
109
116
  return createHash("sha1").update(bytes).digest("hex").slice(0, 16);
110
117
  }
111
118
  function rewriteResourceRefs(doc, oldKey, newKey) {
112
- function walk(els) {
119
+ function walk2(els) {
113
120
  if (!els) return;
114
121
  for (const el of els) {
115
122
  if (el?.resource === oldKey) el.resource = newKey;
116
- if (el?.elements) walk(el.elements);
117
- if (el?.children) walk(el.children);
123
+ if (el?.elements) walk2(el.elements);
124
+ if (el?.children) walk2(el.children);
118
125
  }
119
126
  }
120
- for (const page of doc.pages || []) walk(page.elements);
127
+ for (const page of doc.pages || []) walk2(page.elements);
121
128
  }
122
129
  function extractAssets(doc) {
123
130
  const assets = [];
@@ -149,10 +156,10 @@ function extractAssets(doc) {
149
156
  assets.push({ id, bytes, mimeType, ext });
150
157
  return id;
151
158
  }
152
- function walk(els) {
159
+ function walk2(els) {
153
160
  if (!els) return;
154
161
  for (const el of els) {
155
- if (el?.type === "image" && typeof el.src === "string" && el.src.startsWith("data:")) {
162
+ if ((el?.type === "image" || el?.type === "video") && typeof el.src === "string" && el.src.startsWith("data:")) {
156
163
  const m = el.src.match(/^data:([^;]+);base64,(.*)$/);
157
164
  if (m) {
158
165
  const mimeType = m[1];
@@ -162,19 +169,21 @@ function extractAssets(doc) {
162
169
  el.resource = id;
163
170
  }
164
171
  }
165
- if (el?.elements) walk(el.elements);
166
- if (el?.children) walk(el.children);
172
+ if (el?.elements) walk2(el.elements);
173
+ if (el?.children) walk2(el.children);
167
174
  }
168
175
  }
169
- for (const page of cloned.pages || []) walk(page.elements);
170
- if (cloned.resources?.images) {
171
- for (const [key, res] of Object.entries(cloned.resources.images)) {
176
+ for (const page of cloned.pages || []) walk2(page.elements);
177
+ for (const bucket of ["images", "videos"]) {
178
+ const store = cloned.resources?.[bucket];
179
+ if (!store) continue;
180
+ for (const [key, res] of Object.entries(store)) {
172
181
  if (!res || typeof res !== "object" || !("data" in res) || !res.data) continue;
173
182
  const data = String(res.data);
174
183
  const m = data.match(/^data:([^;]+);base64,(.*)$/);
175
184
  const b64 = m ? m[2] : data;
176
- const mimeType = m ? m[1] : res.mimeType || "image/png";
177
- const ext = mimeType.split("/")[1]?.replace("jpeg", "jpg") || "bin";
185
+ const mimeType = m ? m[1] : res.mimeType || (bucket === "videos" ? "video/mp4" : "image/png");
186
+ const ext = mimeType.split("/")[1]?.replace("jpeg", "jpg").replace("quicktime", "mov") || "bin";
178
187
  const bytes = decodeBase64(b64);
179
188
  const h = hashBytes(bytes);
180
189
  let canonicalId = hashToId.get(h);
@@ -188,10 +197,10 @@ function extractAssets(doc) {
188
197
  delete updated.data;
189
198
  updated.src = "embedded";
190
199
  if (canonicalId !== key) {
191
- delete cloned.resources.images[key];
200
+ delete store[key];
192
201
  rewriteResourceRefs(cloned, key, canonicalId);
193
202
  } else {
194
- cloned.resources.images[key] = updated;
203
+ store[key] = updated;
195
204
  }
196
205
  }
197
206
  }
@@ -221,21 +230,22 @@ async function packJdfx(doc) {
221
230
  return { bytes, manifest };
222
231
  }
223
232
  function shouldUseJdfx(doc) {
224
- const images = doc.resources?.images ?? {};
225
- for (const v of Object.values(images)) {
226
- if (v && typeof v === "object" && "data" in v && v.data) return true;
233
+ for (const store of [doc.resources?.images ?? {}, doc.resources?.videos ?? {}]) {
234
+ for (const v of Object.values(store)) {
235
+ if (v && typeof v === "object" && "data" in v && v.data) return true;
236
+ }
227
237
  }
228
- function walk(els) {
238
+ function walk2(els) {
229
239
  if (!els) return false;
230
240
  for (const el of els) {
231
- if (el?.type === "image" && typeof el.src === "string" && el.src.startsWith("data:")) return true;
232
- if (el?.elements && walk(el.elements)) return true;
233
- if (el?.children && walk(el.children)) return true;
241
+ if ((el?.type === "image" || el?.type === "video") && typeof el.src === "string" && el.src.startsWith("data:")) return true;
242
+ if (el?.elements && walk2(el.elements)) return true;
243
+ if (el?.children && walk2(el.children)) return true;
234
244
  }
235
245
  return false;
236
246
  }
237
247
  for (const page of doc.pages || []) {
238
- if (walk(page.elements)) return true;
248
+ if (walk2(page.elements)) return true;
239
249
  }
240
250
  return false;
241
251
  }
@@ -294,13 +304,13 @@ function stripInline(text) {
294
304
  return parseInline(text).map((r) => r.text).join("");
295
305
  }
296
306
  async function importMarkdown(inputPath, outputPath) {
297
- const input = path2.resolve(inputPath);
307
+ const input = path7.resolve(inputPath);
298
308
  console.log(`Importing: ${input}`);
299
- const content = fs.readFileSync(input, "utf-8");
300
- const doc = convertMarkdownToJdf(content, path2.basename(input, path2.extname(input)), path2.dirname(input));
309
+ const content = fs7.readFileSync(input, "utf-8");
310
+ const doc = convertMarkdownToJdf(content, path7.basename(input, path7.extname(input)), path7.dirname(input));
301
311
  let output;
302
312
  if (outputPath) {
303
- output = path2.resolve(outputPath);
313
+ output = path7.resolve(outputPath);
304
314
  } else {
305
315
  const stem = input.replace(/\.(md|markdown)$/i, "");
306
316
  output = stem + (shouldUseJdfx(doc) ? ".jdfx" : ".jdf");
@@ -308,11 +318,11 @@ async function importMarkdown(inputPath, outputPath) {
308
318
  console.log(`Output: ${output}`);
309
319
  if (output.toLowerCase().endsWith(".jdfx")) {
310
320
  const { bytes, manifest } = await packJdfx(doc);
311
- fs.writeFileSync(output, bytes);
321
+ fs7.writeFileSync(output, bytes);
312
322
  console.log(`
313
323
  Done! Created ${doc.pages.length} page(s), ${manifest.assets.length} asset(s) bundled`);
314
324
  } else {
315
- fs.writeFileSync(output, JSON.stringify(doc, null, 2));
325
+ fs7.writeFileSync(output, JSON.stringify(doc, null, 2));
316
326
  console.log(`
317
327
  Done! Created ${doc.pages.length} page(s)`);
318
328
  }
@@ -329,10 +339,10 @@ var MIME_BY_EXT2 = {
329
339
  };
330
340
  function resolveImageSrc(src, baseDir) {
331
341
  if (/^(https?:|data:|file:)/i.test(src)) return src;
332
- const abs = path2.isAbsolute(src) ? src : path2.resolve(baseDir, src);
342
+ const abs = path7.isAbsolute(src) ? src : path7.resolve(baseDir, src);
333
343
  try {
334
- const bytes = fs.readFileSync(abs);
335
- const ext = path2.extname(abs).slice(1).toLowerCase();
344
+ const bytes = fs7.readFileSync(abs);
345
+ const ext = path7.extname(abs).slice(1).toLowerCase();
336
346
  const mime = MIME_BY_EXT2[ext] || "application/octet-stream";
337
347
  return `data:${mime};base64,${bytes.toString("base64")}`;
338
348
  } catch {
@@ -613,8 +623,232 @@ function convertMarkdownToJdf(md, title, baseDir = process.cwd()) {
613
623
  };
614
624
  }
615
625
 
616
- // ../../packages/jdf-pdf-import/src/core.ts
626
+ // ../../packages/jdf-pdf-import/src/tables.ts
617
627
  var PT_TO_MM = 0.352778;
628
+ function calibrateGlyphWidth(runs) {
629
+ const ks = [];
630
+ for (const r of runs) {
631
+ const t = r.text;
632
+ if (/\s$/.test(t) || t.trim().length < 3 || r.width <= 0) continue;
633
+ ks.push(r.width / (t.length * r.fontSize * PT_TO_MM));
634
+ }
635
+ if (ks.length < 3) return 0.55;
636
+ ks.sort((a, b) => a - b);
637
+ return Math.min(0.7, Math.max(0.4, ks[Math.floor(ks.length / 2)]));
638
+ }
639
+ function hasStretchedSpaces(runs, k) {
640
+ let n = 0, wide = 0;
641
+ for (const r of runs) {
642
+ if (!/\s$/.test(r.text) || r.text.trim().length === 0) continue;
643
+ const em = r.fontSize * PT_TO_MM;
644
+ const est = r.text.trim().length * em * k + em * 0.25;
645
+ n++;
646
+ if (r.width > est * 1.4) wide++;
647
+ }
648
+ return n >= 4 && wide / n >= 0.3;
649
+ }
650
+ function textExtent(r, k, stretched) {
651
+ const em = r.fontSize * PT_TO_MM;
652
+ if (!/\s$/.test(r.text)) return Math.max(em * 0.5, r.width);
653
+ const chars = Math.max(1, r.text.trim().length);
654
+ const est = chars * em * k + em * 0.25;
655
+ if (stretched) return Math.max(em * 0.5, Math.min(r.width, est));
656
+ return Math.max(em * 0.5, r.width > est * 1.4 ? est : r.width);
657
+ }
658
+ function groupRows(runs, skip) {
659
+ const k = calibrateGlyphWidth(runs);
660
+ const stretched = hasStretchedSpaces(runs, k);
661
+ const idx = runs.map((_, i) => i).filter((i) => !skip(runs[i]) && runs[i].text.trim().length > 0);
662
+ idx.sort((a, b) => runs[a].y - runs[b].y || runs[a].x - runs[b].x);
663
+ const rows = [];
664
+ for (const i of idx) {
665
+ const r = runs[i];
666
+ const tol = Math.max(0.8, r.fontSize * PT_TO_MM * 0.35);
667
+ const last = rows[rows.length - 1];
668
+ const cell = { run: r, idx: i, x0: r.x, x1: r.x + textExtent(r, k, stretched) };
669
+ if (last && Math.abs(last.y - r.y) <= tol) {
670
+ last.cells.push(cell);
671
+ last.h = Math.max(last.h, r.height);
672
+ } else {
673
+ rows.push({ y: r.y, h: r.height, cells: [cell] });
674
+ }
675
+ }
676
+ for (const row of rows) {
677
+ row.cells.sort((a, b) => a.x0 - b.x0);
678
+ const merged = [];
679
+ for (const c of row.cells) {
680
+ const last = merged[merged.length - 1];
681
+ const em = c.run.fontSize * PT_TO_MM;
682
+ if (last && c.x0 - last.x1 <= em * 1) {
683
+ last.x1 = Math.max(last.x1, c.x1);
684
+ last.run = { ...last.run, text: `${last.run.text.replace(/\s+$/, "")} ${c.run.text.replace(/^\s+/, "")}`, width: last.x1 - last.x0 };
685
+ last.extra = [...last.extra ?? [], c.idx];
686
+ } else merged.push({ ...c });
687
+ }
688
+ row.cells = merged;
689
+ }
690
+ return rows;
691
+ }
692
+ function columnBands(rows) {
693
+ const cells = rows.flatMap((r) => r.cells);
694
+ const sorted = cells.slice().sort((a, b) => a.x0 - b.x0);
695
+ const bands = [];
696
+ for (const c of sorted) {
697
+ const last = bands[bands.length - 1];
698
+ if (last && c.x0 <= last.x1 - 0.2) {
699
+ last.x1 = Math.max(last.x1, c.x1);
700
+ last.members.push(c);
701
+ } else bands.push({ x0: c.x0, x1: c.x1, members: [c] });
702
+ }
703
+ for (const b of bands) {
704
+ const rowsSeen = /* @__PURE__ */ new Set();
705
+ for (const m of b.members) {
706
+ const row = rows.find((r) => r.cells.includes(m));
707
+ if (rowsSeen.has(row)) return null;
708
+ rowsSeen.add(row);
709
+ }
710
+ }
711
+ return bands.map(({ x0, x1 }) => ({ x0, x1 }));
712
+ }
713
+ var numeric = (s) => /^[\s$€£¥+\-−–]*[\d.,]+\s*(%|ms|s|k|m|b|M|K|B|x|×)?\s*(\/\w+)?$/i.test(s.trim()) || /^[+\-−]?\d/.test(s.trim()) && /\d$/.test(s.trim().replace(/[%)]$/, ""));
714
+ function detectTables(runs, shapes, pageWidthMm) {
715
+ const out = [];
716
+ const used = /* @__PURE__ */ new Set();
717
+ const rows = groupRows(runs, () => false);
718
+ const pageW = pageWidthMm;
719
+ let i = 0;
720
+ while (i < rows.length) {
721
+ if (rows[i].cells.length < 2) {
722
+ i++;
723
+ continue;
724
+ }
725
+ let j = i;
726
+ let cur = columnBands([rows[i]]);
727
+ let best = null;
728
+ while (j + 1 < rows.length && cur) {
729
+ const next = rows[j + 1];
730
+ const gap = next.y - (rows[j].y + rows[j].h);
731
+ const rowH = Math.max(rows[j].h, next.h);
732
+ if (gap > rowH * 2.2) break;
733
+ if (next.cells.length === 1) {
734
+ const c = next.cells[0];
735
+ const inBand = cur.findIndex((b) => c.x0 < b.x1 - 0.2 && c.x1 > b.x0 + 0.2);
736
+ const spansSeveral = cur.filter((b) => c.x0 < b.x1 - 0.2 && c.x1 > b.x0 + 0.2).length > 1;
737
+ if (inBand <= 0 || spansSeveral) break;
738
+ j++;
739
+ continue;
740
+ }
741
+ const nb = columnBands(rows.slice(i, j + 2));
742
+ if (!nb || nb.length < 2) break;
743
+ if (nb.length > cur.length && j - i >= 2) break;
744
+ cur = nb;
745
+ j++;
746
+ if (cur.length >= 2) best = { j, bands: cur };
747
+ }
748
+ const multiRows = best ? rows.slice(i, best.j + 1).filter((r) => r.cells.length >= 2).length : 0;
749
+ const blockRows = best ? rows.slice(i, best.j + 1) : [];
750
+ const bbox = blockRows.length ? {
751
+ x0: Math.min(...best.bands.map((b) => b.x0)) - 5,
752
+ x1: Math.max(...best.bands.map((b) => b.x1)) + 5,
753
+ // Cell padding puts backgrounds/borders well above the first baseline and below the last.
754
+ y0: blockRows[0].y - blockRows[0].h * 2.5,
755
+ y1: blockRows[blockRows.length - 1].y + blockRows[blockRows.length - 1].h * 3
756
+ } : null;
757
+ const gridShapes = bbox ? shapes.map((s, k) => ({ s, k })).filter(({ s }) => s.x >= bbox.x0 - 1 && s.x + s.width <= bbox.x1 + 1 && s.y >= bbox.y0 - 1 && s.y + s.height <= bbox.y1 + 1 && (s.kind === "line" || s.kind === "rect")) : [];
758
+ const hasLattice = gridShapes.length >= 3;
759
+ if (!best || multiRows < (hasLattice ? 2 : 3) || best.bands.length < 2) {
760
+ i++;
761
+ continue;
762
+ }
763
+ const bands = best.bands;
764
+ const cellText = (row, b) => row.cells.filter((c) => c.x0 < bands[b].x1 - 0.2 && c.x1 > bands[b].x0 + 0.2).map((c) => c.run.text.trim()).join(" ").trim();
765
+ const grid = [];
766
+ const lineIdx = [];
767
+ for (const row of blockRows) {
768
+ if (row.cells.length === 1 && grid.length) {
769
+ const c = row.cells[0];
770
+ const b = bands.findIndex((bb) => c.x0 < bb.x1 - 0.2 && c.x1 > bb.x0 + 0.2);
771
+ if (b > 0) {
772
+ grid[grid.length - 1][b] = (grid[grid.length - 1][b] + " " + c.run.text.trim()).trim();
773
+ lineIdx.push(c.idx, ...c.extra ?? []);
774
+ continue;
775
+ }
776
+ }
777
+ grid.push(bands.map((_, b) => cellText(row, b)));
778
+ for (const c of row.cells) {
779
+ lineIdx.push(c.idx);
780
+ for (const k of c.extra ?? []) lineIdx.push(k);
781
+ }
782
+ }
783
+ if (lineIdx.some((k) => used.has(k))) {
784
+ i = best.j + 1;
785
+ continue;
786
+ }
787
+ const first = blockRows[0];
788
+ const tableW = bbox.x1 - bbox.x0;
789
+ const rowFill = (row) => {
790
+ const cy = row.y + row.h * 0.5;
791
+ const rects = gridShapes.filter(({ s }) => s.kind === "rect" && s.fill && s.fill.toLowerCase() !== "#ffffff" && s.height >= row.h * 0.6 && s.height < row.h * 4.5 && s.y <= cy && s.y + s.height >= cy);
792
+ const covered = rects.reduce((a, { s }) => a + s.width, 0);
793
+ if (!rects.length || covered < tableW * 0.5) return null;
794
+ const counts = /* @__PURE__ */ new Map();
795
+ for (const { s } of rects) counts.set(s.fill, (counts.get(s.fill) ?? 0) + s.width);
796
+ const fill = [...counts.entries()].sort((a, b) => b[1] - a[1])[0][0];
797
+ return { fill, rects };
798
+ };
799
+ const headerBg = rowFill(first);
800
+ const firstBold = first.cells.every((c) => c.run.bold);
801
+ const bodyFills = blockRows.slice(1).map(rowFill);
802
+ const headerDistinct = !!headerBg && !bodyFills.every((f) => f?.fill === headerBg.fill);
803
+ const isHeader = headerDistinct || firstBold && !blockRows.slice(1).every((r) => r.cells.every((c) => c.run.bold));
804
+ const columns = bands.map((b, k) => {
805
+ const vals = grid.slice(isHeader ? 1 : 0).map((r) => r[k]).filter(Boolean);
806
+ const numericShare = vals.length ? vals.filter(numeric).length / vals.length : 0;
807
+ const col = { width: Math.round((b.x1 - b.x0) * 10) / 10 };
808
+ if (numericShare >= 0.7) col.align = "right";
809
+ return col;
810
+ });
811
+ const x0 = Math.max(0, bands[0].x0 - 2.5);
812
+ const x1 = Math.min(pageW, bands[bands.length - 1].x1 + 2.5);
813
+ for (let k = 0; k < bands.length; k++) {
814
+ const left = k === 0 ? x0 : (bands[k - 1].x1 + bands[k].x0) / 2;
815
+ const right = k === bands.length - 1 ? x1 : (bands[k].x1 + bands[k + 1].x0) / 2;
816
+ columns[k].width = Math.round((right - left) * 10) / 10;
817
+ }
818
+ const rowFills = (isHeader ? bodyFills : [headerBg, ...bodyFills]).map((f) => f?.fill ?? null);
819
+ const odd = rowFills.filter((_, k) => k % 2 === 1), even = rowFills.filter((_, k) => k % 2 === 0);
820
+ const altColor = odd.length && odd[0] && odd.every((f) => f === odd[0]) && even.every((f) => f !== odd[0]) ? odd[0] : void 0;
821
+ const borderShape = gridShapes.find(({ s }) => s.kind === "line" || s.kind === "rect" && (s.height < 0.6 || s.width < 0.6) && (s.fill || s.stroke));
822
+ const borderColor = borderShape ? borderShape.s.stroke || borderShape.s.fill : void 0;
823
+ const fontSize = Math.round(first.cells[0].run.fontSize * 10) / 10;
824
+ const y0 = headerBg ? Math.min(...headerBg.rects.map(({ s }) => s.y)) : first.y - first.h * 0.5;
825
+ const element = {
826
+ type: "table",
827
+ position: { x: Math.round(x0 * 100) / 100, y: Math.round(Math.max(0, y0) * 100) / 100 },
828
+ width: Math.round((x1 - x0) * 100) / 100,
829
+ columns,
830
+ rows: isHeader ? grid.slice(1) : grid,
831
+ style: { fontSize }
832
+ };
833
+ if (isHeader) {
834
+ element.headers = grid[0];
835
+ const hs = { fontWeight: "bold" };
836
+ if (headerBg) hs.backgroundColor = headerBg.fill;
837
+ const hc = first.cells[0].run.color;
838
+ if (hc && hc !== "#000000") hs.color = hc;
839
+ element.headerStyle = hs;
840
+ }
841
+ if (altColor) element.alternatingRowColor = altColor;
842
+ element.borders = borderColor ? { outer: true, inner: true, color: borderColor, width: 1 } : false;
843
+ for (const k of lineIdx) used.add(k);
844
+ out.push({ element, lineIdx, shapeIdx: gridShapes.map(({ k }) => k) });
845
+ i = best.j + 1;
846
+ }
847
+ return out;
848
+ }
849
+
850
+ // ../../packages/jdf-pdf-import/src/core.ts
851
+ var PT_TO_MM2 = 0.352778;
618
852
  function classifyFont(name) {
619
853
  const n = (name || "").toLowerCase();
620
854
  const bold = /bold|black|heavy|semibold|demibold|extrabold/.test(n);
@@ -640,6 +874,38 @@ function rgbToHex(r, g, b) {
640
874
  const h = (n) => clampByte(n).toString(16).padStart(2, "0");
641
875
  return `#${h(r)}${h(g)}${h(b)}`;
642
876
  }
877
+ function averageStops(stops) {
878
+ if (!Array.isArray(stops) || stops.length === 0) return null;
879
+ let r = 0, g = 0, b = 0, n = 0;
880
+ for (const st of stops) {
881
+ const css = Array.isArray(st) ? st[1] : null;
882
+ const m = typeof css === "string" && css.match(/^#([0-9a-f]{2})([0-9a-f]{2})([0-9a-f]{2})$/i);
883
+ if (!m) continue;
884
+ r += parseInt(m[1], 16);
885
+ g += parseInt(m[2], 16);
886
+ b += parseInt(m[3], 16);
887
+ n++;
888
+ }
889
+ return n ? rgbToHex(r / n, g / n, b / n) : null;
890
+ }
891
+ function patternToColor(page, arg) {
892
+ if (!Array.isArray(arg)) return null;
893
+ if (arg[0] === "TilingPattern") {
894
+ const c = arg[1];
895
+ return Array.isArray(c) && c.length >= 3 ? rgbToHex(c[0], c[1], c[2]) : null;
896
+ }
897
+ if (arg[0] === "Shading") {
898
+ const id = arg[1];
899
+ try {
900
+ const store = typeof id === "string" && id.startsWith("g_") ? page.commonObjs : page.objs;
901
+ if (typeof store?.has === "function" && !store.has(id)) return null;
902
+ const ir = store.get(id);
903
+ if (Array.isArray(ir) && ir[0] === "RadialAxial") return averageStops(ir[3]);
904
+ } catch {
905
+ }
906
+ }
907
+ return null;
908
+ }
643
909
  function multiplyCtm(a, b) {
644
910
  return [
645
911
  a[0] * b[0] + a[1] * b[2],
@@ -658,7 +924,7 @@ async function walkOps(page, OPS, viewport) {
658
924
  const [vx, vy] = viewport.convertToViewportPoint(x, y);
659
925
  return { x: vx, y: vy };
660
926
  };
661
- const opList = await page.getOperatorList();
927
+ const opList = await page.getOperatorList({ annotationMode: annotationModeDisable });
662
928
  const fnArr = opList.fnArray;
663
929
  const argsArr = opList.argsArray;
664
930
  const gs = {
@@ -668,20 +934,60 @@ async function walkOps(page, OPS, viewport) {
668
934
  lineWidth: 1,
669
935
  fillAlpha: 1,
670
936
  strokeAlpha: 1,
671
- textRenderingMode: 0
937
+ textRenderingMode: 0,
938
+ // Text state (PDF 9.3) — needed to know where each showText lands.
939
+ fontSize: 0,
940
+ charSpacing: 0,
941
+ wordSpacing: 0,
942
+ hscale: 1,
943
+ leading: 0,
944
+ rise: 0
672
945
  };
946
+ const snapshot = () => ({ ...gs, ctm: [...gs.ctm] });
673
947
  const stack = [];
674
- const textColors = [];
675
- const textOpacities = [];
676
- const textRenderingModes = [];
948
+ const textOps = [];
677
949
  const shapes = [];
678
950
  const imagePositions = [];
679
- let textIdx = 0;
951
+ let tm = [1, 0, 0, 1, 0, 0];
952
+ let tlm = [1, 0, 0, 1, 0, 0];
953
+ function pushImage(name, ctm, inline, maskFill) {
954
+ const corners = [tx(ctm, 0, 0), tx(ctm, 1, 0), tx(ctm, 1, 1), tx(ctm, 0, 1)];
955
+ const vpCorners = corners.map((p) => toViewport(p.x, p.y));
956
+ const xs = vpCorners.map((p) => p.x), ys = vpCorners.map((p) => p.y);
957
+ const minX = Math.min(...xs), maxX = Math.max(...xs);
958
+ const minY = Math.min(...ys), maxY = Math.max(...ys);
959
+ imagePositions.push({
960
+ name,
961
+ x: minX * PT_TO_MM2,
962
+ y: minY * PT_TO_MM2,
963
+ w: (maxX - minX) * PT_TO_MM2,
964
+ h: (maxY - minY) * PT_TO_MM2,
965
+ inline,
966
+ maskFill
967
+ });
968
+ }
680
969
  let pathSegments = [];
681
970
  let pathRect = null;
971
+ let pathRects = [];
682
972
  let pathStart = null;
683
973
  let pathLast = null;
684
974
  function flushPath(isFill, isStroke) {
975
+ for (const r of pathRects) {
976
+ const tl = toViewport(r.x, r.y + r.h);
977
+ const br = toViewport(r.x + r.w, r.y);
978
+ shapes.push({
979
+ kind: "rect",
980
+ x: Math.min(tl.x, br.x) * PT_TO_MM2,
981
+ y: Math.min(tl.y, br.y) * PT_TO_MM2,
982
+ width: Math.abs(br.x - tl.x) * PT_TO_MM2,
983
+ height: Math.abs(br.y - tl.y) * PT_TO_MM2,
984
+ fill: isFill ? gs.fill : void 0,
985
+ stroke: isStroke ? gs.stroke : void 0,
986
+ strokeWidth: isStroke ? gs.lineWidth * PT_TO_MM2 : void 0,
987
+ opacity: isFill ? gs.fillAlpha : gs.strokeAlpha
988
+ });
989
+ }
990
+ pathRects = [];
685
991
  if (pathRect) {
686
992
  const tl = toViewport(pathRect.x, pathRect.y + pathRect.h);
687
993
  const br = toViewport(pathRect.x + pathRect.w, pathRect.y);
@@ -691,13 +997,13 @@ async function walkOps(page, OPS, viewport) {
691
997
  const h = Math.abs(br.y - tl.y);
692
998
  shapes.push({
693
999
  kind: "rect",
694
- x: x * PT_TO_MM,
695
- y: y * PT_TO_MM,
696
- width: w * PT_TO_MM,
697
- height: h * PT_TO_MM,
1000
+ x: x * PT_TO_MM2,
1001
+ y: y * PT_TO_MM2,
1002
+ width: w * PT_TO_MM2,
1003
+ height: h * PT_TO_MM2,
698
1004
  fill: isFill ? gs.fill : void 0,
699
1005
  stroke: isStroke ? gs.stroke : void 0,
700
- strokeWidth: isStroke ? gs.lineWidth * PT_TO_MM : void 0,
1006
+ strokeWidth: isStroke ? gs.lineWidth * PT_TO_MM2 : void 0,
701
1007
  opacity: isFill ? gs.fillAlpha : gs.strokeAlpha
702
1008
  });
703
1009
  } else if (pathSegments.length === 2 && pathSegments[0].type === "M" && pathSegments[1].type === "L") {
@@ -709,35 +1015,35 @@ async function walkOps(page, OPS, viewport) {
709
1015
  const minY = Math.min(va.y, vb.y);
710
1016
  const maxX = Math.max(va.x, vb.x);
711
1017
  const maxY = Math.max(va.y, vb.y);
712
- const x1Local = (va.x - minX) * PT_TO_MM;
713
- const y1Local = (va.y - minY) * PT_TO_MM;
714
- const x2Local = (vb.x - minX) * PT_TO_MM;
715
- const y2Local = (vb.y - minY) * PT_TO_MM;
716
- const wLocal = Math.max(0.05, (maxX - minX) * PT_TO_MM);
717
- const hLocal = Math.max(0.05, (maxY - minY) * PT_TO_MM);
1018
+ const x1Local = (va.x - minX) * PT_TO_MM2;
1019
+ const y1Local = (va.y - minY) * PT_TO_MM2;
1020
+ const x2Local = (vb.x - minX) * PT_TO_MM2;
1021
+ const y2Local = (vb.y - minY) * PT_TO_MM2;
1022
+ const wLocal = Math.max(0.05, (maxX - minX) * PT_TO_MM2);
1023
+ const hLocal = Math.max(0.05, (maxY - minY) * PT_TO_MM2);
718
1024
  const dx = Math.abs(va.x - vb.x);
719
1025
  const dy = Math.abs(va.y - vb.y);
720
1026
  const axisAligned = dx < 0.5 || dy < 0.5;
721
1027
  if (axisAligned) {
722
1028
  shapes.push({
723
1029
  kind: "line",
724
- x: minX * PT_TO_MM,
725
- y: minY * PT_TO_MM,
1030
+ x: minX * PT_TO_MM2,
1031
+ y: minY * PT_TO_MM2,
726
1032
  width: wLocal,
727
1033
  height: hLocal,
728
1034
  stroke: isStroke ? gs.stroke : void 0,
729
- strokeWidth: isStroke ? gs.lineWidth * PT_TO_MM : void 0,
1035
+ strokeWidth: isStroke ? gs.lineWidth * PT_TO_MM2 : void 0,
730
1036
  opacity: gs.strokeAlpha
731
1037
  });
732
1038
  } else {
733
1039
  shapes.push({
734
1040
  kind: "path",
735
- x: minX * PT_TO_MM,
736
- y: minY * PT_TO_MM,
1041
+ x: minX * PT_TO_MM2,
1042
+ y: minY * PT_TO_MM2,
737
1043
  width: wLocal,
738
1044
  height: hLocal,
739
1045
  stroke: isStroke ? gs.stroke : void 0,
740
- strokeWidth: isStroke ? gs.lineWidth * PT_TO_MM : void 0,
1046
+ strokeWidth: isStroke ? gs.lineWidth * PT_TO_MM2 : void 0,
741
1047
  opacity: gs.strokeAlpha,
742
1048
  path: `M ${x1Local.toFixed(2)} ${y1Local.toFixed(2)} L ${x2Local.toFixed(2)} ${y2Local.toFixed(2)}`
743
1049
  });
@@ -770,20 +1076,20 @@ async function walkOps(page, OPS, viewport) {
770
1076
  if (seg.type === "Z") return "Z";
771
1077
  const p = [];
772
1078
  for (let i = 0; i < seg.pts.length; i += 2) {
773
- p.push(((seg.pts[i] - minX) * PT_TO_MM).toFixed(2));
774
- p.push(((seg.pts[i + 1] - minY) * PT_TO_MM).toFixed(2));
1079
+ p.push(((seg.pts[i] - minX) * PT_TO_MM2).toFixed(2));
1080
+ p.push(((seg.pts[i + 1] - minY) * PT_TO_MM2).toFixed(2));
775
1081
  }
776
1082
  return `${seg.type} ${p.join(" ")}`;
777
1083
  }).join(" ");
778
1084
  shapes.push({
779
1085
  kind: "path",
780
- x: minX * PT_TO_MM,
781
- y: minY * PT_TO_MM,
782
- width: bw * PT_TO_MM,
783
- height: bh * PT_TO_MM,
1086
+ x: minX * PT_TO_MM2,
1087
+ y: minY * PT_TO_MM2,
1088
+ width: bw * PT_TO_MM2,
1089
+ height: bh * PT_TO_MM2,
784
1090
  fill: isFill ? gs.fill : void 0,
785
1091
  stroke: isStroke ? gs.stroke : void 0,
786
- strokeWidth: isStroke ? gs.lineWidth * PT_TO_MM : void 0,
1092
+ strokeWidth: isStroke ? gs.lineWidth * PT_TO_MM2 : void 0,
787
1093
  opacity: isFill ? gs.fillAlpha : gs.strokeAlpha,
788
1094
  path: d
789
1095
  });
@@ -798,10 +1104,51 @@ async function walkOps(page, OPS, viewport) {
798
1104
  const fn = fnArr[i];
799
1105
  const args = argsArr[i] || [];
800
1106
  if (fn === OPS.save) {
801
- stack.push({ ctm: [...gs.ctm], fill: gs.fill, stroke: gs.stroke, lineWidth: gs.lineWidth, fillAlpha: gs.fillAlpha, strokeAlpha: gs.strokeAlpha, textRenderingMode: gs.textRenderingMode });
1107
+ stack.push(snapshot());
802
1108
  } else if (fn === OPS.restore) {
803
1109
  const s = stack.pop();
804
1110
  if (s) Object.assign(gs, s);
1111
+ } else if (fn === OPS.paintFormXObjectBegin) {
1112
+ stack.push(snapshot());
1113
+ const matrix = args[0];
1114
+ if (Array.isArray(matrix) && matrix.length === 6) gs.ctm = multiplyCtm(matrix, gs.ctm);
1115
+ } else if (fn === OPS.paintFormXObjectEnd) {
1116
+ const s = stack.pop();
1117
+ if (s) Object.assign(gs, s);
1118
+ } else if (fn === OPS.beginText) {
1119
+ tm = [1, 0, 0, 1, 0, 0];
1120
+ tlm = [1, 0, 0, 1, 0, 0];
1121
+ } else if (fn === OPS.setTextMatrix) {
1122
+ tm = [...args];
1123
+ tlm = [...tm];
1124
+ } else if (fn === OPS.moveText) {
1125
+ tlm = multiplyCtm([1, 0, 0, 1, args[0], args[1]], tlm);
1126
+ tm = [...tlm];
1127
+ } else if (fn === OPS.setLeadingMoveText) {
1128
+ gs.leading = -args[1];
1129
+ tlm = multiplyCtm([1, 0, 0, 1, args[0], args[1]], tlm);
1130
+ tm = [...tlm];
1131
+ } else if (fn === OPS.nextLine) {
1132
+ tlm = multiplyCtm([1, 0, 0, 1, 0, -gs.leading], tlm);
1133
+ tm = [...tlm];
1134
+ } else if (fn === OPS.setLeading) {
1135
+ gs.leading = args[0];
1136
+ } else if (fn === OPS.setFont) {
1137
+ gs.fontSize = typeof args[1] === "number" ? args[1] : gs.fontSize;
1138
+ } else if (fn === OPS.setCharSpacing) {
1139
+ gs.charSpacing = args[0];
1140
+ } else if (fn === OPS.setWordSpacing) {
1141
+ gs.wordSpacing = args[0];
1142
+ } else if (fn === OPS.setHScale) {
1143
+ gs.hscale = (args[0] ?? 100) / 100;
1144
+ } else if (fn === OPS.setTextRise) {
1145
+ gs.rise = args[0];
1146
+ } else if (fn === OPS.setFillColorN) {
1147
+ const c = patternToColor(page, args[0]);
1148
+ if (c) gs.fill = c;
1149
+ } else if (fn === OPS.setStrokeColorN) {
1150
+ const c = patternToColor(page, args[0]);
1151
+ if (c) gs.stroke = c;
805
1152
  } else if (fn === OPS.transform) {
806
1153
  gs.ctm = multiplyCtm(args, gs.ctm);
807
1154
  } else if (fn === OPS.setFillRGBColor) {
@@ -835,11 +1182,23 @@ async function walkOps(page, OPS, viewport) {
835
1182
  else if (key === "CA") gs.strokeAlpha = val;
836
1183
  }
837
1184
  }
838
- } else if (fn === OPS.showText || fn === OPS.showSpacedText || fn === OPS.nextLineShowText || fn === OPS.nextLineSetSpacingShowText) {
839
- textColors[textIdx] = gs.fill;
840
- textOpacities[textIdx] = gs.fillAlpha;
841
- textRenderingModes[textIdx] = gs.textRenderingMode;
842
- textIdx++;
1185
+ } else if (fn === OPS.showText) {
1186
+ const trm = multiplyCtm(tm, gs.ctm);
1187
+ const origin = tx(trm, 0, gs.rise);
1188
+ const vp = toViewport(origin.x, origin.y);
1189
+ const scale = Math.hypot(trm[2], trm[3]) || 1;
1190
+ textOps.push({ x: vp.x, y: vp.y, fontSize: gs.fontSize * scale, fill: gs.fill, alpha: gs.fillAlpha, mode: gs.textRenderingMode });
1191
+ let advance = 0;
1192
+ const glyphs = Array.isArray(args[0]) ? args[0] : [];
1193
+ for (const g of glyphs) {
1194
+ if (typeof g === "number") {
1195
+ advance += -g / 1e3 * gs.fontSize * gs.hscale;
1196
+ } else if (g && typeof g === "object") {
1197
+ const w = typeof g.width === "number" ? g.width : 0;
1198
+ advance += (w / 1e3 * gs.fontSize + gs.charSpacing + (g.isSpace ? gs.wordSpacing : 0)) * gs.hscale;
1199
+ }
1200
+ }
1201
+ tm = multiplyCtm([1, 0, 0, 1, advance, 0], tm);
843
1202
  } else if (fn === OPS.rectangle) {
844
1203
  const [x, y, w, h] = args;
845
1204
  const p1 = tx(gs.ctm, x, y);
@@ -892,6 +1251,12 @@ async function walkOps(page, OPS, viewport) {
892
1251
  } else if (op === OPS.closePath) {
893
1252
  pathSegments.push({ type: "Z", pts: [] });
894
1253
  if (pathStart) pathLast = { ...pathStart };
1254
+ } else if (op === OPS.rectangle) {
1255
+ const x = pathArgs[ai], y = pathArgs[ai + 1], w = pathArgs[ai + 2], h = pathArgs[ai + 3];
1256
+ ai += 4;
1257
+ const p1 = tx(gs.ctm, x, y);
1258
+ const p3 = tx(gs.ctm, x + w, y + h);
1259
+ pathRects.push({ x: Math.min(p1.x, p3.x), y: Math.min(p1.y, p3.y), w: Math.abs(p3.x - p1.x), h: Math.abs(p3.y - p1.y) });
895
1260
  }
896
1261
  }
897
1262
  } else if (fn === OPS.fill || fn === OPS.stroke || fn === OPS.fillStroke || fn === OPS.eoFill || fn === OPS.eoFillStroke || fn === OPS.closeFillStroke || fn === OPS.closeStroke || fn === OPS.closeEOFillStroke) {
@@ -900,59 +1265,128 @@ async function walkOps(page, OPS, viewport) {
900
1265
  flushPath(isFill, isStroke);
901
1266
  } else if (fn === OPS.endPath || fn === OPS.clip || fn === OPS.eoClip) {
902
1267
  pathSegments = [];
1268
+ pathRects = [];
903
1269
  pathRect = null;
904
1270
  pathStart = null;
905
1271
  pathLast = null;
906
- } else if (fn === OPS.paintImageXObject || fn === OPS.paintImageMaskXObject || fn === OPS.paintInlineImageXObject) {
907
- const name = args[0];
908
- const c = gs.ctm;
909
- const corners = [tx(c, 0, 0), tx(c, 1, 0), tx(c, 1, 1), tx(c, 0, 1)];
910
- const vpCorners = corners.map((p) => toViewport(p.x, p.y));
911
- const xs = vpCorners.map((p) => p.x), ys = vpCorners.map((p) => p.y);
912
- const minX = Math.min(...xs), maxX = Math.max(...xs);
913
- const minY = Math.min(...ys), maxY = Math.max(...ys);
914
- imagePositions.push({
915
- name,
916
- x: minX * PT_TO_MM,
917
- y: minY * PT_TO_MM,
918
- w: (maxX - minX) * PT_TO_MM,
919
- h: (maxY - minY) * PT_TO_MM
920
- });
1272
+ } else if (fn === OPS.paintImageXObject) {
1273
+ pushImage(String(args[0]), gs.ctm);
1274
+ } else if (fn === OPS.paintImageXObjectRepeat) {
1275
+ const [name, scaleX, scaleY, positions] = args;
1276
+ if (Array.isArray(positions)) {
1277
+ for (let k = 0; k + 1 < positions.length; k += 2) {
1278
+ pushImage(String(name), multiplyCtm([scaleX, 0, 0, scaleY, positions[k], positions[k + 1]], gs.ctm));
1279
+ }
1280
+ }
1281
+ } else if (fn === OPS.paintInlineImageXObject) {
1282
+ const img = args[0];
1283
+ if (img && typeof img === "object") pushImage(`inline-${imagePositions.length}`, gs.ctm, img);
1284
+ } else if (fn === OPS.paintImageMaskXObject) {
1285
+ const img = args[0];
1286
+ if (img && typeof img === "object") pushImage(`mask-${imagePositions.length}`, gs.ctm, img, gs.fill);
1287
+ } else if (fn === OPS.paintImageMaskXObjectRepeat) {
1288
+ const [img, scaleX, skewX, skewY, scaleY, positions] = args;
1289
+ if (img && typeof img === "object" && Array.isArray(positions)) {
1290
+ for (let k = 0; k + 1 < positions.length; k += 2) {
1291
+ pushImage(`mask-${imagePositions.length}`, multiplyCtm([scaleX, skewX, skewY, scaleY, positions[k], positions[k + 1]], gs.ctm), img, gs.fill);
1292
+ }
1293
+ }
1294
+ } else if (fn === OPS.paintImageMaskXObjectGroup) {
1295
+ const group = args[0];
1296
+ if (Array.isArray(group)) {
1297
+ for (const img of group) {
1298
+ if (img && typeof img === "object" && Array.isArray(img.transform)) {
1299
+ pushImage(`mask-${imagePositions.length}`, multiplyCtm(img.transform, gs.ctm), img, gs.fill);
1300
+ }
1301
+ }
1302
+ }
921
1303
  }
922
1304
  }
923
- return { textColors, textOpacities, textRenderingModes, shapes, imagePositions };
1305
+ return { textOps, shapes, imagePositions };
924
1306
  }
925
1307
  async function extractImages(page, positions, runtime, dataUrlCache) {
1308
+ const resolveObj = (id) => new Promise((resolve) => {
1309
+ let settled2 = false;
1310
+ const timer = setTimeout(() => done(null), 250);
1311
+ const done = (v) => {
1312
+ if (settled2) return;
1313
+ settled2 = true;
1314
+ clearTimeout(timer);
1315
+ resolve(v);
1316
+ };
1317
+ try {
1318
+ const store = id.startsWith("g_") ? page.commonObjs : page.objs;
1319
+ store.get(id, (img) => done(img));
1320
+ } catch {
1321
+ try {
1322
+ page.objs.get(id, (img) => done(img));
1323
+ } catch {
1324
+ done(null);
1325
+ }
1326
+ }
1327
+ });
1328
+ const maskToRgba = (data, width, height, fill) => {
1329
+ const m = fill.match(/^#([0-9a-f]{2})([0-9a-f]{2})([0-9a-f]{2})$/i);
1330
+ const r = m ? parseInt(m[1], 16) : 0, g = m ? parseInt(m[2], 16) : 0, b = m ? parseInt(m[3], 16) : 0;
1331
+ const out = new Uint8ClampedArray(width * height * 4);
1332
+ const rowBytes = width + 7 >> 3;
1333
+ for (let y = 0; y < height; y++) {
1334
+ for (let x = 0; x < width; x++) {
1335
+ const byte = data[y * rowBytes + (x >> 3)] ?? 255;
1336
+ const bit = byte >> 7 - (x & 7) & 1;
1337
+ const o = (y * width + x) * 4;
1338
+ out[o] = r;
1339
+ out[o + 1] = g;
1340
+ out[o + 2] = b;
1341
+ out[o + 3] = bit ? 0 : 255;
1342
+ }
1343
+ }
1344
+ return out;
1345
+ };
1346
+ const encode = async (imgObj, maskFill) => {
1347
+ if (!imgObj || !imgObj.width || !imgObj.height) return null;
1348
+ let data = imgObj.data;
1349
+ if (typeof data === "string") data = (await resolveObj(data))?.data ?? null;
1350
+ if (maskFill) {
1351
+ if (!data) return null;
1352
+ return runtime.encodePng(imgObj.width, imgObj.height, 3, maskToRgba(data, imgObj.width, imgObj.height, maskFill));
1353
+ }
1354
+ if (data) {
1355
+ let kind = imgObj.kind || 0;
1356
+ if (!kind) {
1357
+ const px = imgObj.width * imgObj.height;
1358
+ if (data.length === px * 4) kind = 3;
1359
+ else if (data.length === px * 3) kind = 2;
1360
+ else if (data.length === (imgObj.width + 7 >> 3) * imgObj.height) kind = 1;
1361
+ }
1362
+ return runtime.encodePng(imgObj.width, imgObj.height, kind, data);
1363
+ }
1364
+ if (imgObj.bitmap) {
1365
+ try {
1366
+ const { canvas, context } = runtime.createCanvas(imgObj.width, imgObj.height);
1367
+ context.drawImage(imgObj.bitmap, 0, 0);
1368
+ if (typeof canvas.toDataURL === "function") return canvas.toDataURL("image/png");
1369
+ if (typeof canvas.toBuffer === "function") return `data:image/png;base64,${canvas.toBuffer("image/png").toString("base64")}`;
1370
+ } catch {
1371
+ }
1372
+ }
1373
+ return null;
1374
+ };
926
1375
  const tasks = positions.map(async (pos) => {
1376
+ if (pos.inline) {
1377
+ const dataUrl2 = await encode(pos.inline, pos.maskFill);
1378
+ return dataUrl2 ? { pos, dataUrl: dataUrl2 } : null;
1379
+ }
927
1380
  if (dataUrlCache.has(pos.name)) {
928
1381
  return { pos, dataUrl: dataUrlCache.get(pos.name) };
929
1382
  }
930
1383
  let imgObj = null;
931
1384
  try {
932
- imgObj = await new Promise((resolve) => {
933
- let settled2 = false;
934
- const timer = setTimeout(() => done(null), 250);
935
- const done = (v) => {
936
- if (settled2) return;
937
- settled2 = true;
938
- clearTimeout(timer);
939
- resolve(v);
940
- };
941
- try {
942
- page.objs.get(pos.name, (img) => done(img));
943
- } catch {
944
- try {
945
- page.commonObjs.get(pos.name, (img) => done(img));
946
- } catch {
947
- done(null);
948
- }
949
- }
950
- });
1385
+ imgObj = await resolveObj(pos.name);
951
1386
  } catch {
952
1387
  imgObj = null;
953
1388
  }
954
- if (!imgObj || !imgObj.data || !imgObj.width || !imgObj.height) return null;
955
- const dataUrl = runtime.encodePng(imgObj.width, imgObj.height, imgObj.kind || 0, imgObj.data);
1389
+ const dataUrl = await encode(imgObj);
956
1390
  if (!dataUrl) return null;
957
1391
  dataUrlCache.set(pos.name, dataUrl);
958
1392
  return { pos, dataUrl };
@@ -960,7 +1394,20 @@ async function extractImages(page, positions, runtime, dataUrlCache) {
960
1394
  const settled = await Promise.all(tasks);
961
1395
  return settled.filter((x) => x !== null);
962
1396
  }
963
- async function extractLinks(page, viewport) {
1397
+ async function resolveDestPage(doc, dest) {
1398
+ try {
1399
+ let d = dest;
1400
+ if (typeof d === "string") d = await doc.getDestination(d);
1401
+ if (Array.isArray(d) && d[0] != null) {
1402
+ if (typeof d[0] === "number") return d[0];
1403
+ const idx = await doc.getPageIndex(d[0]);
1404
+ if (typeof idx === "number") return idx;
1405
+ }
1406
+ } catch {
1407
+ }
1408
+ return void 0;
1409
+ }
1410
+ async function extractLinks(doc, page, viewport) {
964
1411
  const out = [];
965
1412
  let annots = [];
966
1413
  try {
@@ -983,13 +1430,15 @@ async function extractLinks(page, viewport) {
983
1430
  const xMax = Math.max(c1.x, c2.x);
984
1431
  const yMax = Math.max(c1.y, c2.y);
985
1432
  const rectMm = {
986
- x: xMin * PT_TO_MM,
987
- y: yMin * PT_TO_MM,
988
- w: (xMax - xMin) * PT_TO_MM,
989
- h: (yMax - yMin) * PT_TO_MM
1433
+ x: xMin * PT_TO_MM2,
1434
+ y: yMin * PT_TO_MM2,
1435
+ w: (xMax - xMin) * PT_TO_MM2,
1436
+ h: (yMax - yMin) * PT_TO_MM2
990
1437
  };
991
1438
  const url = a.url || a.unsafeUrl;
992
- out.push({ rectMm, url });
1439
+ const destPage = url ? void 0 : await resolveDestPage(doc, a.dest);
1440
+ if (!url && destPage == null) continue;
1441
+ out.push({ rectMm, url, destPage });
993
1442
  }
994
1443
  return out;
995
1444
  }
@@ -1022,10 +1471,10 @@ async function extractFormWidgets(page, viewport) {
1022
1471
  })).filter((o) => o.value !== "") : [];
1023
1472
  out.push({
1024
1473
  rectMm: {
1025
- x: xMin * PT_TO_MM,
1026
- y: yMin * PT_TO_MM,
1027
- w: (xMax - xMin) * PT_TO_MM,
1028
- h: (yMax - yMin) * PT_TO_MM
1474
+ x: xMin * PT_TO_MM2,
1475
+ y: yMin * PT_TO_MM2,
1476
+ w: (xMax - xMin) * PT_TO_MM2,
1477
+ h: (yMax - yMin) * PT_TO_MM2
1029
1478
  },
1030
1479
  fieldType: a.fieldType || "",
1031
1480
  fieldName: a.fieldName || `field-${out.length + 1}`,
@@ -1045,26 +1494,62 @@ async function extractFormWidgets(page, viewport) {
1045
1494
  async function flattenOutline(doc, outline) {
1046
1495
  if (!outline) return [];
1047
1496
  const out = [];
1048
- async function walk(items) {
1497
+ async function walk2(items, depth) {
1049
1498
  for (const item of items) {
1050
- try {
1051
- let dest = item.dest;
1052
- if (typeof dest === "string") {
1053
- dest = await doc.getDestination(dest);
1054
- }
1055
- if (Array.isArray(dest) && dest[0]) {
1056
- const ref = dest[0];
1057
- const idx = await doc.getPageIndex(ref);
1058
- if (typeof idx === "number") out.push({ title: item.title, pageIndex: idx });
1059
- }
1060
- } catch {
1499
+ const idx = await resolveDestPage(doc, item.dest);
1500
+ if (idx != null && typeof item.title === "string" && item.title.trim()) {
1501
+ out.push({ title: item.title.trim(), pageIndex: idx, depth });
1061
1502
  }
1062
- if (item.items?.length) await walk(item.items);
1503
+ if (item.items?.length) await walk2(item.items, depth + 1);
1063
1504
  }
1064
1505
  }
1065
- await walk(outline);
1506
+ await walk2(outline, 1);
1066
1507
  return out;
1067
1508
  }
1509
+ var normTitle = (s) => s.toLowerCase().replace(/[\s\u00a0]+/g, " ").replace(/[^\p{L}\p{N} ]/gu, "").trim();
1510
+ function applyOutline(pages, outline) {
1511
+ for (const entry of outline) {
1512
+ const page = pages[entry.pageIndex];
1513
+ if (!page) continue;
1514
+ const want = normTitle(entry.title);
1515
+ if (!want) continue;
1516
+ let best = null;
1517
+ let bestScore = 0;
1518
+ for (const el of page.elements) {
1519
+ if (el.type !== "text") continue;
1520
+ const have = normTitle(el.content || "");
1521
+ if (!have) continue;
1522
+ let score = 0;
1523
+ if (have === want) score = 3;
1524
+ else if (have.startsWith(want) || want.startsWith(have)) score = 2;
1525
+ else if (have.length >= 6 && want.includes(have)) score = 1;
1526
+ if (score > bestScore || score === bestScore && best && score > 0 && (el.position?.y ?? 0) < (best.position?.y ?? 0)) {
1527
+ best = el;
1528
+ bestScore = score;
1529
+ }
1530
+ }
1531
+ if (best && bestScore > 0) {
1532
+ const level = Math.min(6, Math.max(1, entry.depth));
1533
+ if (!best.heading) best.heading = level;
1534
+ best.tocEntry = entry.title;
1535
+ best.tocLevel = level;
1536
+ }
1537
+ }
1538
+ }
1539
+ function pdfDateToIso(v) {
1540
+ if (typeof v !== "string") return void 0;
1541
+ const m = v.match(/^D:(\d{4})(\d{2})?(\d{2})?(\d{2})?(\d{2})?(\d{2})?([Zz+-])?(\d{2})?'?(\d{2})?/);
1542
+ if (!m) {
1543
+ const t2 = Date.parse(v);
1544
+ return Number.isFinite(t2) ? new Date(t2).toISOString() : void 0;
1545
+ }
1546
+ const [, Y, Mo = "01", D = "01", h = "00", mi = "00", s = "00", sign, oh = "00", om = "00"] = m;
1547
+ const tz = !sign || sign === "Z" || sign === "z" ? "Z" : `${sign}${oh}:${om}`;
1548
+ const iso = `${Y}-${Mo}-${D}T${h}:${mi}:${s}${tz}`;
1549
+ const t = Date.parse(iso);
1550
+ return Number.isFinite(t) ? new Date(t).toISOString() : void 0;
1551
+ }
1552
+ var annotationModeDisable = 0;
1068
1553
  async function importPdfToJdf(source, title, runtime, options = {}) {
1069
1554
  const pdfjs = options.pdfjs || runtime.pdfjs;
1070
1555
  if (!pdfjs) {
@@ -1085,7 +1570,7 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
1085
1570
  } else {
1086
1571
  data = source;
1087
1572
  }
1088
- const doc = await pdfjs.getDocument({
1573
+ const loadingTask = pdfjs.getDocument({
1089
1574
  data,
1090
1575
  // The runtime adapter declares whether it supports a real Web Worker.
1091
1576
  // We don't sniff `typeof Worker` here because Node 22+ exposes a global
@@ -1094,15 +1579,55 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
1094
1579
  // newer Node. Browser entry leaves this unset (= false = real worker
1095
1580
  // via GlobalWorkerOptions.workerSrc); node entry sets `true`.
1096
1581
  disableWorker: runtime.disableWorker === true,
1097
- isEvalSupported: false
1098
- }).promise;
1582
+ isEvalSupported: false,
1583
+ // Keep going past malformed content streams instead of failing the page.
1584
+ stopAtErrors: false,
1585
+ // Force the classic pixel-array image path on every host so the reader
1586
+ // (WKWebView has OffscreenCanvas) and the CLI produce identical PNGs.
1587
+ isOffscreenCanvasSupported: false,
1588
+ password: options.password,
1589
+ // Standard-14 font metrics + CJK CMaps: without these PDF.js falls back
1590
+ // to guesses for non-embedded fonts and logs a warning per page.
1591
+ ...runtime.standardFontDataUrl ? { standardFontDataUrl: runtime.standardFontDataUrl } : {},
1592
+ ...runtime.cMapUrl ? { cMapUrl: runtime.cMapUrl, cMapPacked: true } : {}
1593
+ });
1594
+ let passwordTried = typeof options.password === "string";
1595
+ loadingTask.onPassword = (updatePassword, reason) => {
1596
+ const retry = reason === 2 || passwordTried;
1597
+ if (!options.onPassword) {
1598
+ updatePassword(new Error(retry ? "[@jdf/pdf-import] Wrong password for encrypted PDF" : "[@jdf/pdf-import] PDF is password-protected \u2014 pass `password`"));
1599
+ return;
1600
+ }
1601
+ passwordTried = true;
1602
+ options.onPassword(retry).then((pw) => {
1603
+ if (pw == null) updatePassword(new Error("[@jdf/pdf-import] Password entry cancelled"));
1604
+ else updatePassword(pw);
1605
+ }).catch((e) => updatePassword(e instanceof Error ? e : new Error(String(e))));
1606
+ };
1607
+ let doc;
1608
+ try {
1609
+ doc = await loadingTask.promise;
1610
+ } catch (e) {
1611
+ if (e?.name === "PasswordException") {
1612
+ const msg = /no password/i.test(e.message || "") ? "PDF is password-protected \u2014 pass a password (CLI: --password)" : e.message || "PDF is password-protected";
1613
+ throw new Error(`[@jdf/pdf-import] ${msg}`);
1614
+ }
1615
+ throw e;
1616
+ }
1099
1617
  const pages = [];
1100
1618
  const imageResources = {};
1101
1619
  let imgCounter = 0;
1102
1620
  const dataUrlCache = /* @__PURE__ */ new Map();
1103
1621
  const resourceKeyByName = /* @__PURE__ */ new Map();
1104
- const outline = await doc.getOutline().catch(() => null);
1105
- await flattenOutline(doc, outline);
1622
+ const outline = await flattenOutline(doc, await doc.getOutline().catch(() => null));
1623
+ let pdfInfo = null;
1624
+ let pdfMetadata = null;
1625
+ try {
1626
+ const md = await doc.getMetadata();
1627
+ pdfInfo = md?.info ?? null;
1628
+ pdfMetadata = md?.metadata ?? null;
1629
+ } catch {
1630
+ }
1106
1631
  for (let pi = 1; pi <= doc.numPages; pi++) {
1107
1632
  let findLinkForRun2 = function(r) {
1108
1633
  const cx = r.x + r.width / 2;
@@ -1124,7 +1649,7 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
1124
1649
  } catch {
1125
1650
  }
1126
1651
  const ops = await walkOps(page, OPS, viewport);
1127
- const links = await extractLinks(page, viewport);
1652
+ const links = await extractLinks(doc, page, viewport);
1128
1653
  const formWidgets = await extractFormWidgets(page, viewport);
1129
1654
  const textContent = await page.getTextContent({ disableCombineTextItems: false });
1130
1655
  const items = textContent.items;
@@ -1167,9 +1692,53 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
1167
1692
  const n = typeof v === "number" ? v : Number(v);
1168
1693
  return Number.isFinite(n) ? n : fallback;
1169
1694
  };
1170
- items.forEach((it, idx) => {
1695
+ const opBins = /* @__PURE__ */ new Map();
1696
+ for (const op of ops.textOps) {
1697
+ const key = `${Math.round(op.x)},${Math.round(op.y)}`;
1698
+ const arr = opBins.get(key);
1699
+ if (arr) arr.push(op);
1700
+ else opBins.set(key, [op]);
1701
+ }
1702
+ const findOp = (x, y, fontSize) => {
1703
+ let best = null;
1704
+ let bestD = Infinity;
1705
+ const bx = Math.round(x), by = Math.round(y);
1706
+ for (let dy = -1; dy <= 1; dy++) {
1707
+ for (let dx = -1; dx <= 1; dx++) {
1708
+ const arr = opBins.get(`${bx + dx},${by + dy}`);
1709
+ if (!arr) continue;
1710
+ for (const op of arr) {
1711
+ const d = Math.hypot(op.x - x, op.y - y);
1712
+ if (d < bestD) {
1713
+ bestD = d;
1714
+ best = op;
1715
+ }
1716
+ }
1717
+ }
1718
+ }
1719
+ if (best) return best;
1720
+ const tol = Math.max(2, fontSize * 0.6);
1721
+ for (const op of ops.textOps) {
1722
+ if (Math.abs(op.y - y) > tol) continue;
1723
+ const d = Math.abs(op.x - x) + Math.abs(op.y - y) * 4;
1724
+ if (d < bestD) {
1725
+ bestD = d;
1726
+ best = op;
1727
+ }
1728
+ }
1729
+ if (best) return best;
1730
+ for (const op of ops.textOps) {
1731
+ const d = Math.hypot(op.x - x, op.y - y);
1732
+ if (d < bestD) {
1733
+ bestD = d;
1734
+ best = op;
1735
+ }
1736
+ }
1737
+ return bestD < 40 ? best : null;
1738
+ };
1739
+ const keepInvisible = options.invisibleText !== "drop";
1740
+ items.forEach((it) => {
1171
1741
  if (!it.str || !it.str.length) return;
1172
- if ((ops.textRenderingModes[idx] ?? 0) === 3) return;
1173
1742
  const tr = it.transform;
1174
1743
  const fontSize = safeNum(Math.hypot(safeNum(tr?.[2], 0), safeNum(tr?.[3], 0)), 0) || safeNum(it.height, 0) || 10;
1175
1744
  const baseX = safeNum(tr?.[4], 0);
@@ -1177,24 +1746,35 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
1177
1746
  const conv = viewport.convertToViewportPoint(baseX, baseY);
1178
1747
  const vx = safeNum(conv?.[0], 0);
1179
1748
  const vy = safeNum(conv?.[1], 0);
1749
+ const op = findOp(vx, vy, fontSize);
1750
+ const mode = op?.mode ?? 0;
1751
+ if (mode === 7) return;
1752
+ const invisible = mode === 3;
1753
+ if (invisible && !keepInvisible) return;
1180
1754
  const ascent = it.height ? safeNum(it.height, fontSize) * 0.78 : fontSize * 0.78;
1181
1755
  const yTop = vy - ascent;
1182
1756
  const w = safeNum(it.width, 0);
1183
1757
  runs.push({
1184
1758
  text: it.str,
1185
- x: safeNum(vx * PT_TO_MM, 0),
1186
- y: safeNum(yTop * PT_TO_MM, 0),
1759
+ x: safeNum(vx * PT_TO_MM2, 0),
1760
+ y: safeNum(yTop * PT_TO_MM2, 0),
1187
1761
  fontSize: safeNum(fontSize, 10),
1188
1762
  fontName: it.fontName,
1189
- width: safeNum(w * PT_TO_MM, 0),
1190
- height: safeNum((it.height || fontSize) * PT_TO_MM, fontSize * PT_TO_MM),
1191
- color: ops.textColors[idx] || "#000000",
1192
- opacity: safeNum(ops.textOpacities[idx], 1)
1763
+ width: safeNum(w * PT_TO_MM2, 0),
1764
+ height: safeNum((it.height || fontSize) * PT_TO_MM2, fontSize * PT_TO_MM2),
1765
+ color: op?.fill || "#000000",
1766
+ opacity: invisible ? 0 : safeNum(op?.alpha, 1)
1193
1767
  });
1194
1768
  });
1195
1769
  runs.sort((a, b) => a.y - b.y || a.x - b.x);
1196
1770
  const lines = [];
1197
1771
  const Y_TOL = 0.6;
1772
+ const kGlyph = calibrateGlyphWidth(runs);
1773
+ const stretchedSpaces = hasStretchedSpaces(runs, kGlyph);
1774
+ const fontKey = (name) => {
1775
+ const c = fontMap.get(name) || classifyFont(name || "");
1776
+ return `${c.family}|${c.weight || ""}|${c.style || ""}`;
1777
+ };
1198
1778
  for (const r of runs) {
1199
1779
  if (!r.text.length) continue;
1200
1780
  const last = lines[lines.length - 1];
@@ -1203,24 +1783,48 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
1203
1783
  continue;
1204
1784
  }
1205
1785
  const sameLine = Math.abs(last.y - r.y) <= Y_TOL;
1206
- const sameStyle = Math.abs(last.fontSize - r.fontSize) < 0.4 && last.fontName === r.fontName && last.color === r.color && Math.abs(last.opacity - r.opacity) < 0.05;
1207
- const gapMm = r.x - (last.x + last.width);
1208
- const emMm = r.fontSize * PT_TO_MM;
1209
- const mergeOk = sameLine && sameStyle && gapMm >= -0.2 && gapMm <= emMm * 0.45;
1786
+ const sameStyle = Math.abs(last.fontSize - r.fontSize) < 0.4 && (last.fontName === r.fontName || fontKey(last.fontName) === fontKey(r.fontName)) && last.color === r.color && Math.abs(last.opacity - r.opacity) < 0.05;
1787
+ const emMm = r.fontSize * PT_TO_MM2;
1788
+ const extent = (t) => {
1789
+ if (!/\s$/.test(t.text)) return t.width;
1790
+ const em = t.fontSize * PT_TO_MM2;
1791
+ const est = Math.max(1, t.text.trim().length) * em * kGlyph + em * 0.25;
1792
+ if (stretchedSpaces) return Math.min(t.width, est);
1793
+ return t.width > est * 1.4 ? est : t.width;
1794
+ };
1795
+ const gapMm = r.x - (last.x + extent(last));
1796
+ const mergeOk = sameLine && sameStyle && gapMm >= -emMm * 0.5 && gapMm <= emMm * 0.45;
1210
1797
  if (mergeOk) {
1211
1798
  const lastEndsSpace = /\s$/.test(last.text);
1212
1799
  const currStartsSpace = /^\s/.test(r.text);
1213
1800
  const sep = gapMm > emMm * 0.08 && !lastEndsSpace && !currStartsSpace ? " " : "";
1214
1801
  last.text = last.text + sep + r.text;
1215
1802
  const newExtent = r.x - last.x + r.width;
1216
- last.width = Math.max(last.width, newExtent);
1803
+ last.width = Math.max(extent(last), newExtent);
1217
1804
  } else {
1218
1805
  lines.push({ ...r });
1219
1806
  }
1220
1807
  }
1221
1808
  const elements = [];
1222
- for (const sh of ops.shapes) {
1223
- if (sh.width < 0.3 && sh.height < 0.3) continue;
1809
+ const tRuns = lines.map((l) => {
1810
+ const cls = fontMap.get(l.fontName) || classifyFont(l.fontName || "");
1811
+ return { text: l.text, x: l.x, y: l.y, width: l.width, height: l.height, fontSize: l.fontSize, fontName: l.fontName, color: l.color, bold: cls.weight === "bold" };
1812
+ });
1813
+ const detected = options.detectTables === false ? [] : detectTables(tRuns, ops.shapes, pageW * PT_TO_MM2);
1814
+ const consumedLines = /* @__PURE__ */ new Set();
1815
+ const consumedShapes = /* @__PURE__ */ new Set();
1816
+ const tableAtLine = /* @__PURE__ */ new Map();
1817
+ for (const t of detected) {
1818
+ for (const k of t.lineIdx) consumedLines.add(k);
1819
+ for (const k of t.shapeIdx) consumedShapes.add(k);
1820
+ tableAtLine.set(Math.min(...t.lineIdx), t.element);
1821
+ }
1822
+ const pageWmm = pageW * PT_TO_MM2, pageHmm = pageH * PT_TO_MM2;
1823
+ ops.shapes.forEach((sh, shapeIdx) => {
1824
+ if (consumedShapes.has(shapeIdx)) return;
1825
+ if (sh.width < 0.3 && sh.height < 0.3) return;
1826
+ if (sh.x + sh.width <= 0 || sh.y + sh.height <= 0 || sh.x >= pageWmm || sh.y >= pageHmm) return;
1827
+ if (sh.kind === "rect" && sh.fill && !sh.stroke && sh.width * sh.height >= pageWmm * pageHmm * 0.9) return;
1224
1828
  const shapeType = sh.kind;
1225
1829
  const shape = {
1226
1830
  type: "shape",
@@ -1236,7 +1840,7 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
1236
1840
  shape.style = { opacity: Math.round(sh.opacity * 100) / 100 };
1237
1841
  }
1238
1842
  elements.push(shape);
1239
- }
1843
+ });
1240
1844
  const imgs = await extractImages(page, ops.imagePositions, runtime, dataUrlCache);
1241
1845
  for (const { pos, dataUrl } of imgs) {
1242
1846
  let resourceKey = resourceKeyByName.get(pos.name);
@@ -1259,7 +1863,16 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
1259
1863
  fit: "fill"
1260
1864
  });
1261
1865
  }
1866
+ const sizeChars = /* @__PURE__ */ new Map();
1262
1867
  for (const l of lines) {
1868
+ const k = Math.round(l.fontSize * 2) / 2;
1869
+ sizeChars.set(k, (sizeChars.get(k) ?? 0) + l.text.length);
1870
+ }
1871
+ const bodyFontSize = [...sizeChars.entries()].sort((a, b) => b[1] - a[1])[0]?.[0] ?? 0;
1872
+ lines.forEach((l, lineIdx) => {
1873
+ const tableEl = tableAtLine.get(lineIdx);
1874
+ if (tableEl) elements.push(tableEl);
1875
+ if (consumedLines.has(lineIdx)) return;
1263
1876
  const cls = fontMap.get(l.fontName) || classifyFont(l.fontName || "");
1264
1877
  const style = {
1265
1878
  fontSize: Math.round(l.fontSize * 10) / 10,
@@ -1270,8 +1883,7 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
1270
1883
  if (l.color !== "#000000") style.color = l.color;
1271
1884
  if (l.opacity < 0.999) style.opacity = Math.round(l.opacity * 100) / 100;
1272
1885
  const link = findLinkForRun2(l);
1273
- const pageWmm = pageW * PT_TO_MM;
1274
- const measured = Math.max(l.width + l.fontSize * PT_TO_MM * 0.4, l.fontSize * PT_TO_MM);
1886
+ const measured = Math.max(l.width + l.fontSize * PT_TO_MM2 * 0.4, l.fontSize * PT_TO_MM2);
1275
1887
  const remaining = Math.max(measured, pageWmm - l.x);
1276
1888
  const elWidth = Math.min(measured, remaining);
1277
1889
  const text = {
@@ -1281,18 +1893,26 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
1281
1893
  width: Math.max(2, Math.round(elWidth * 100) / 100),
1282
1894
  style
1283
1895
  };
1284
- if (cls.weight === "bold") {
1285
- if (l.fontSize >= 22) text.heading = 1;
1286
- else if (l.fontSize >= 17) text.heading = 2;
1287
- else if (l.fontSize >= 16) text.heading = 3;
1896
+ if (cls.weight === "bold" && l.text.trim().length <= 120 && !consumedLines.has(lineIdx)) {
1897
+ const ratio = bodyFontSize > 0 ? l.fontSize / bodyFontSize : 1;
1898
+ if (l.fontSize >= 22 || ratio >= 1.8) text.heading = 1;
1899
+ else if (l.fontSize >= 17 || ratio >= 1.35) text.heading = 2;
1900
+ else if (l.fontSize >= 16 || ratio >= 1.2) text.heading = 3;
1288
1901
  }
1289
1902
  if (text.heading) text.tocEntry = text.content;
1290
1903
  if (link) {
1291
1904
  if (link.url) text.link = link.url;
1292
1905
  else if (link.destPage != null) text.link = { type: "internal", target: `#page-${link.destPage + 1}` };
1293
1906
  }
1907
+ const prev = elements[elements.length - 1];
1908
+ if (text.heading && prev && prev.type === "text" && prev.heading === text.heading && !link && !prev.link && Math.abs(prev.style?.fontSize - style.fontSize) < 0.5 && Math.abs(prev.position.x - text.position.x) < 1 && text.position.y - prev.position.y < l.fontSize * PT_TO_MM2 * 2.2 && text.position.y > prev.position.y) {
1909
+ prev.content = `${prev.content} ${text.content}`.replace(/\s+/g, " ");
1910
+ prev.tocEntry = prev.content;
1911
+ prev.width = Math.max(prev.width ?? 0, text.width ?? 0);
1912
+ return;
1913
+ }
1294
1914
  elements.push(text);
1295
- }
1915
+ });
1296
1916
  for (const w of formWidgets) {
1297
1917
  if (w.pushButton) continue;
1298
1918
  const baseEl = {
@@ -1327,19 +1947,41 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
1327
1947
  }
1328
1948
  pages.push({
1329
1949
  id: `page-${pi}`,
1330
- pageSize: { width: Math.round(pageW * PT_TO_MM * 100) / 100, height: Math.round(pageH * PT_TO_MM * 100) / 100 },
1950
+ pageSize: { width: Math.round(pageW * PT_TO_MM2 * 100) / 100, height: Math.round(pageH * PT_TO_MM2 * 100) / 100 },
1331
1951
  margins: { top: 0, right: 0, bottom: 0, left: 0 },
1332
1952
  elements
1333
1953
  });
1334
1954
  }
1955
+ applyOutline(pages, outline);
1956
+ const meta = {
1957
+ title,
1958
+ pageSize: pages[0]?.pageSize || "A4",
1959
+ unit: "mm",
1960
+ margins: { top: 0, right: 0, bottom: 0, left: 0 }
1961
+ };
1962
+ if (pdfInfo) {
1963
+ if (typeof pdfInfo.Author === "string" && pdfInfo.Author.trim()) meta.author = pdfInfo.Author.trim();
1964
+ const created = pdfDateToIso(pdfInfo.CreationDate);
1965
+ const modified = pdfDateToIso(pdfInfo.ModDate);
1966
+ if (created) meta.created = created;
1967
+ if (modified) meta.modified = modified;
1968
+ if (typeof pdfInfo.Keywords === "string") {
1969
+ const kws = pdfInfo.Keywords.split(/[,;]+/).map((k) => k.trim()).filter(Boolean);
1970
+ if (kws.length) meta.keywords = kws;
1971
+ }
1972
+ if (typeof pdfInfo.Language === "string" && pdfInfo.Language.trim()) meta.language = pdfInfo.Language.trim();
1973
+ }
1974
+ if (!meta.language && pdfMetadata && typeof pdfMetadata.get === "function") {
1975
+ try {
1976
+ const lang = pdfMetadata.get("dc:language");
1977
+ const first = Array.isArray(lang) ? lang[0] : lang;
1978
+ if (typeof first === "string" && first.trim()) meta.language = first.trim();
1979
+ } catch {
1980
+ }
1981
+ }
1335
1982
  const result = {
1336
1983
  $jdf: "1.0.0",
1337
- meta: {
1338
- title,
1339
- pageSize: pages[0]?.pageSize || "A4",
1340
- unit: "mm",
1341
- margins: { top: 0, right: 0, bottom: 0, left: 0 }
1342
- },
1984
+ meta,
1343
1985
  pages
1344
1986
  };
1345
1987
  if (Object.keys(imageResources).length > 0) {
@@ -1351,27 +1993,31 @@ async function importPdfToJdf(source, title, runtime, options = {}) {
1351
1993
  // ../../packages/jdf-pdf-import/src/node.ts
1352
1994
  var pdfjsModule = null;
1353
1995
  var pdfjsLoadPromise = null;
1996
+ var pdfjsAssetDirs = null;
1354
1997
  async function loadNodePdfJs() {
1355
1998
  if (pdfjsModule) return pdfjsModule;
1356
1999
  if (pdfjsLoadPromise) return pdfjsLoadPromise;
1357
2000
  pdfjsLoadPromise = (async () => {
1358
2001
  const { createRequire } = await import('module');
1359
2002
  const require_ = createRequire(import.meta.url);
1360
- const origWarn = console.warn;
1361
- console.warn = (...args) => {
1362
- if (typeof args[0] === "string" && args[0].includes("legacy")) return;
1363
- origWarn.apply(console, args);
2003
+ const origLog = console.log;
2004
+ console.log = (...args) => {
2005
+ if (typeof args[0] === "string" && args[0].includes("`legacy` build")) return;
2006
+ origLog.apply(console, args);
1364
2007
  };
1365
2008
  let lib;
1366
2009
  try {
1367
2010
  lib = await import('pdfjs-dist/build/pdf.mjs');
1368
2011
  } finally {
1369
- console.warn = origWarn;
2012
+ console.log = origLog;
1370
2013
  }
1371
2014
  const workerPath = require_.resolve("pdfjs-dist/build/pdf.worker.mjs");
1372
2015
  if (lib.GlobalWorkerOptions) {
1373
2016
  lib.GlobalWorkerOptions.workerSrc = workerPath;
1374
2017
  }
2018
+ const { dirname, join } = await import('path');
2019
+ const pkgDir = dirname(dirname(workerPath));
2020
+ pdfjsAssetDirs = { standardFontDataUrl: join(pkgDir, "standard_fonts") + "/", cMapUrl: join(pkgDir, "cmaps") + "/" };
1375
2021
  pdfjsModule = lib;
1376
2022
  return lib;
1377
2023
  })();
@@ -1382,6 +2028,10 @@ async function loadCanvas() {
1382
2028
  if (canvasModule) return canvasModule;
1383
2029
  try {
1384
2030
  canvasModule = await import('@napi-rs/canvas');
2031
+ const g = globalThis;
2032
+ if (typeof g.DOMMatrix === "undefined" && canvasModule.DOMMatrix) g.DOMMatrix = canvasModule.DOMMatrix;
2033
+ if (typeof g.Path2D === "undefined" && canvasModule.Path2D) g.Path2D = canvasModule.Path2D;
2034
+ if (typeof g.ImageData === "undefined" && canvasModule.ImageData) g.ImageData = canvasModule.ImageData;
1385
2035
  return canvasModule;
1386
2036
  } catch (err) {
1387
2037
  throw new Error(
@@ -1433,6 +2083,7 @@ async function importPdfToJdf2(source, title, options = {}) {
1433
2083
  const runtime = {
1434
2084
  pdfjs,
1435
2085
  disableWorker: true,
2086
+ ...pdfjsAssetDirs ?? {},
1436
2087
  createCanvas(width, height) {
1437
2088
  const canvas = canvasMod.createCanvas(width, height);
1438
2089
  const context = canvas.getContext("2d");
@@ -1449,19 +2100,22 @@ async function importPdfToJdf2(source, title, options = {}) {
1449
2100
 
1450
2101
  // src/commands/import-pdf.ts
1451
2102
  async function importPdf(inputPath, outputPath, options = {}) {
1452
- const input = path2.resolve(inputPath);
1453
- if (!fs.existsSync(input)) {
2103
+ const input = path7.resolve(inputPath);
2104
+ if (!fs7.existsSync(input)) {
1454
2105
  console.error(`File not found: ${input}`);
1455
2106
  process.exit(1);
1456
2107
  }
1457
2108
  console.log(`Importing: ${input}`);
1458
- const title = path2.basename(input, path2.extname(input));
2109
+ const title = path7.basename(input, path7.extname(input));
1459
2110
  const t0 = Date.now();
1460
- const doc = await importPdfToJdf2(input, title);
2111
+ const doc = await importPdfToJdf2(input, title, {
2112
+ password: options.password,
2113
+ invisibleText: options.dropInvisibleText ? "drop" : "keep"
2114
+ });
1461
2115
  console.log(`Parsed in ${((Date.now() - t0) / 1e3).toFixed(1)}s \u2014 ${doc.pages.length} page(s)`);
1462
2116
  let output;
1463
2117
  if (outputPath) {
1464
- output = path2.resolve(outputPath);
2118
+ output = path7.resolve(outputPath);
1465
2119
  } else {
1466
2120
  const stem = input.replace(/\.pdf$/i, "");
1467
2121
  const wantJdfx = !options.forceJson && shouldUseJdfx(doc);
@@ -1470,11 +2124,11 @@ async function importPdf(inputPath, outputPath, options = {}) {
1470
2124
  console.log(`Output: ${output}`);
1471
2125
  if (output.toLowerCase().endsWith(".jdfx")) {
1472
2126
  const { bytes, manifest } = await packJdfx(doc);
1473
- fs.writeFileSync(output, bytes);
2127
+ fs7.writeFileSync(output, bytes);
1474
2128
  console.log(`
1475
2129
  Done! Created ${doc.pages.length} page(s), ${manifest.assets.length} asset(s) bundled`);
1476
2130
  } else {
1477
- fs.writeFileSync(output, JSON.stringify(doc, null, 2));
2131
+ fs7.writeFileSync(output, JSON.stringify(doc, null, 2));
1478
2132
  console.log(`
1479
2133
  Done! Created ${doc.pages.length} page(s)`);
1480
2134
  }
@@ -1487,23 +2141,23 @@ var ImportJsonError = class extends Error {
1487
2141
  }
1488
2142
  };
1489
2143
  async function importJson(inputPath, outputPath, options = {}) {
1490
- const input = path2.resolve(inputPath);
1491
- if (!fs.existsSync(input)) {
2144
+ const input = path7.resolve(inputPath);
2145
+ if (!fs7.existsSync(input)) {
1492
2146
  throw new ImportJsonError(`File not found: ${input}`);
1493
2147
  }
1494
2148
  console.log(`Importing: ${input}`);
1495
- const raw = fs.readFileSync(input, "utf-8");
2149
+ const raw = fs7.readFileSync(input, "utf-8");
1496
2150
  let parsed;
1497
2151
  try {
1498
2152
  parsed = JSON.parse(raw);
1499
2153
  } catch (e) {
1500
2154
  throw new ImportJsonError(`Not valid JSON: ${e.message}`);
1501
2155
  }
1502
- const title = path2.basename(input, path2.extname(input));
2156
+ const title = path7.basename(input, path7.extname(input));
1503
2157
  const doc = normaliseToJdf(parsed, title);
1504
2158
  let output;
1505
2159
  if (outputPath) {
1506
- output = path2.resolve(outputPath);
2160
+ output = path7.resolve(outputPath);
1507
2161
  } else {
1508
2162
  const stem = input.replace(/\.json$/i, "");
1509
2163
  const wantJdfx = !options.forceJson && shouldUseJdfx(doc);
@@ -1512,11 +2166,11 @@ async function importJson(inputPath, outputPath, options = {}) {
1512
2166
  console.log(`Output: ${output}`);
1513
2167
  if (output.toLowerCase().endsWith(".jdfx")) {
1514
2168
  const { bytes, manifest } = await packJdfx(doc);
1515
- fs.writeFileSync(output, bytes);
2169
+ fs7.writeFileSync(output, bytes);
1516
2170
  console.log(`
1517
2171
  Done! Created ${doc.pages.length} page(s), ${manifest.assets.length} asset(s) bundled`);
1518
2172
  } else {
1519
- fs.writeFileSync(output, JSON.stringify(doc, null, 2));
2173
+ fs7.writeFileSync(output, JSON.stringify(doc, null, 2));
1520
2174
  console.log(`
1521
2175
  Done! Created ${doc.pages.length} page(s)`);
1522
2176
  }
@@ -1592,6 +2246,47 @@ function wrapElements(elements, title, meta) {
1592
2246
  ]
1593
2247
  };
1594
2248
  }
2249
+ var DEFAULT_TRANSCRIPT_WINDOW = 45;
2250
+ var fmtTime = (sec) => {
2251
+ const s = Math.max(0, Math.round(sec));
2252
+ const h = Math.floor(s / 3600), m = Math.floor(s % 3600 / 60), r = s % 60;
2253
+ return h ? `${h}:${String(m).padStart(2, "0")}:${String(r).padStart(2, "0")}` : `${String(m).padStart(2, "0")}:${String(r).padStart(2, "0")}`;
2254
+ };
2255
+ function transcriptChunks(el, elementId2, page, crumb, windowSec, maxTokens) {
2256
+ const segs = Array.isArray(el?.transcript?.segments) ? el.transcript.segments : [];
2257
+ if (!segs.length) return [];
2258
+ const chapters = Array.isArray(el?.chapters) ? [...el.chapters].sort((a, b) => a.t - b.t) : [];
2259
+ const chapterAt = (t) => {
2260
+ let cur = null;
2261
+ for (const c of chapters) {
2262
+ if (c.t <= t + 1e-6) cur = c;
2263
+ else break;
2264
+ }
2265
+ return cur;
2266
+ };
2267
+ const out = [];
2268
+ let win = [];
2269
+ const flush = () => {
2270
+ if (!win.length) return;
2271
+ const t0 = win[0].t0, t1 = win[win.length - 1].t1;
2272
+ const body = win.map((sg) => sg.speaker ? `${sg.speaker}: ${sg.text}` : sg.text).join(" ").replace(/\s+/g, " ").trim();
2273
+ const text = `[${fmtTime(t0)}\u2013${fmtTime(t1)}] ${body}`;
2274
+ const chapter = chapterAt(t0);
2275
+ const path9 = [...crumb, ...el.title ? [String(el.title)] : [], ...chapter ? [String(chapter.title)] : []];
2276
+ out.push({ id: `${elementId2}@${Math.round(t0)}`, text, path: path9, page, types: ["video"], tokens: estimateTokens(text), hash: hashText(text), media: { element: elementId2, t0, t1 } });
2277
+ win = [];
2278
+ };
2279
+ for (const sg of segs) {
2280
+ if (typeof sg?.text !== "string" || !sg.text.trim()) continue;
2281
+ const startsNewChapter = win.length && chapterAt(sg.t0) !== chapterAt(win[0].t0);
2282
+ const spansWindow = win.length && sg.t1 - win[0].t0 > windowSec;
2283
+ const overBudget = win.length && estimateTokens(win.map((w) => w.text).join(" ") + sg.text) > maxTokens;
2284
+ if (startsNewChapter || spansWindow || overBudget) flush();
2285
+ win.push(sg);
2286
+ }
2287
+ flush();
2288
+ return out;
2289
+ }
1595
2290
  var DEFAULT_MAX_TOKENS = 512;
1596
2291
  function estimateTokens(text) {
1597
2292
  return Math.ceil(text.length / 4);
@@ -1614,11 +2309,11 @@ function serializeElement(el) {
1614
2309
  case "richtext":
1615
2310
  return (e.runs || []).map((r) => r.text ?? "").join("").trim();
1616
2311
  case "list": {
1617
- const walk = (items, depth = 0) => (items || []).flatMap((it) => {
2312
+ const walk2 = (items, depth = 0) => (items || []).flatMap((it) => {
1618
2313
  const line = " ".repeat(depth) + "- " + String(it.content ?? "").trim();
1619
- return it.children?.length ? [line, ...walk(it.children, depth + 1)] : [line];
2314
+ return it.children?.length ? [line, ...walk2(it.children, depth + 1)] : [line];
1620
2315
  });
1621
- return walk(e.items).join("\n");
2316
+ return walk2(e.items).join("\n");
1622
2317
  }
1623
2318
  case "table": {
1624
2319
  const headers = e.headers ?? e.columns?.map((c) => c.header || "").filter((h) => h) ?? [];
@@ -1649,6 +2344,8 @@ function serializeElement(el) {
1649
2344
  return `${e.checked ? "[x]" : "[ ]"} ${e.label ?? ""}`.trim();
1650
2345
  case "image":
1651
2346
  return e.alt ? `[image: ${e.alt}]` : "";
2347
+ case "video":
2348
+ return e.title ? `[video: ${e.title}]` : "";
1652
2349
  case "toc":
1653
2350
  case "shape":
1654
2351
  case "signature":
@@ -1688,8 +2385,15 @@ function makeChunk(group, breadcrumb) {
1688
2385
  function chunkDocument(doc, options = {}) {
1689
2386
  const strategy = options.strategy ?? "section";
1690
2387
  const maxTokens = options.maxTokens ?? DEFAULT_MAX_TOKENS;
2388
+ const windowSec = options.transcriptWindowSec ?? DEFAULT_TRANSCRIPT_WINDOW;
1691
2389
  const flat = flatten(doc);
1692
2390
  const chunks = [];
2391
+ const withTranscripts = (group, crumb2, c) => {
2392
+ if (c) chunks.push(c);
2393
+ for (const f of group) {
2394
+ if (f.el.type === "video") chunks.push(...transcriptChunks(f.el, f.id, f.page, crumb2, windowSec, maxTokens));
2395
+ }
2396
+ };
1693
2397
  if (strategy === "element") {
1694
2398
  const crumb2 = [];
1695
2399
  for (const f of flat) {
@@ -1698,8 +2402,7 @@ function chunkDocument(doc, options = {}) {
1698
2402
  crumb2.length = Math.max(0, lvl - 1);
1699
2403
  crumb2[lvl - 1] = serializeElement(f.el);
1700
2404
  }
1701
- const c = makeChunk([f], crumb2);
1702
- if (c) chunks.push(c);
2405
+ withTranscripts([f], crumb2, makeChunk([f], crumb2));
1703
2406
  }
1704
2407
  return chunks;
1705
2408
  }
@@ -1708,8 +2411,7 @@ function chunkDocument(doc, options = {}) {
1708
2411
  let buf2 = [];
1709
2412
  let bufTokens = 0;
1710
2413
  const flush = () => {
1711
- const c = makeChunk(buf2, crumb2);
1712
- if (c) chunks.push(c);
2414
+ withTranscripts(buf2, crumb2, makeChunk(buf2, crumb2));
1713
2415
  buf2 = [];
1714
2416
  bufTokens = 0;
1715
2417
  };
@@ -1736,22 +2438,21 @@ function chunkDocument(doc, options = {}) {
1736
2438
  for (const f of buf) {
1737
2439
  const t = estimateTokens(serializeElement(f.el));
1738
2440
  if (subTokens + t > maxTokens && sub.length > 0) {
1739
- const c2 = makeChunk(sub, crumb);
1740
- if (c2) chunks.push(c2);
2441
+ withTranscripts(sub, crumb, makeChunk(sub, crumb));
1741
2442
  sub = [];
1742
2443
  subTokens = 0;
1743
2444
  }
1744
2445
  sub.push(f);
1745
2446
  subTokens += t;
1746
2447
  }
1747
- const c = makeChunk(sub, crumb);
1748
- if (c) chunks.push(c);
2448
+ withTranscripts(sub, crumb, makeChunk(sub, crumb));
1749
2449
  buf = [];
1750
2450
  };
1751
2451
  for (const f of flat) {
1752
2452
  const lvl = headingLevel(f.el);
1753
2453
  if (lvl != null) {
1754
- flushSection();
2454
+ const onlyHeadings = buf.length > 0 && buf.every((b) => headingLevel(b.el) != null);
2455
+ if (!onlyHeadings) flushSection();
1755
2456
  crumb.length = Math.max(0, lvl - 1);
1756
2457
  crumb[lvl - 1] = serializeElement(f.el);
1757
2458
  }
@@ -1762,34 +2463,34 @@ function chunkDocument(doc, options = {}) {
1762
2463
  }
1763
2464
  async function loadJdf(filePath) {
1764
2465
  if (filePath.toLowerCase().endsWith(".jdfx")) {
1765
- const zip = await JSZip.loadAsync(fs.readFileSync(filePath));
2466
+ const zip = await JSZip.loadAsync(fs7.readFileSync(filePath));
1766
2467
  const docFile = zip.file(JDFX_DOCUMENT_PATH);
1767
2468
  if (!docFile) throw new Error(`Bundle missing ${JDFX_DOCUMENT_PATH}`);
1768
2469
  return JSON.parse(await docFile.async("string"));
1769
2470
  }
1770
- return JSON.parse(fs.readFileSync(filePath, "utf-8"));
2471
+ return JSON.parse(fs7.readFileSync(filePath, "utf-8"));
1771
2472
  }
1772
2473
  async function chunkFile(inputPath, opts = {}) {
1773
- const input = path2.resolve(inputPath);
1774
- if (!fs.existsSync(input)) throw new Error(`File not found: ${input}`);
2474
+ const input = path7.resolve(inputPath);
2475
+ if (!fs7.existsSync(input)) throw new Error(`File not found: ${input}`);
1775
2476
  const doc = await loadJdf(input);
1776
2477
  const strategy = opts.strategy ?? "section";
1777
- const chunks = chunkDocument(doc, { strategy, maxTokens: opts.maxTokens });
2478
+ const chunks = chunkDocument(doc, { strategy, maxTokens: opts.maxTokens, transcriptWindowSec: opts.transcriptWindowSec });
1778
2479
  const format = opts.format ?? "jsonl";
1779
2480
  console.log(`Chunking: ${input}`);
1780
2481
  console.log(`Strategy: ${strategy}${opts.maxTokens ? ` (max ${opts.maxTokens} tokens)` : ""}`);
1781
2482
  if (format === "inline") {
1782
- const out = opts.output ? path2.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".jdf");
2483
+ const out = opts.output ? path7.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".jdf");
1783
2484
  const withIndex = { ...doc, index: { chunker: `jdf-${strategy}-v1`, chunks } };
1784
- fs.writeFileSync(out, JSON.stringify(withIndex, null, 2));
2485
+ fs7.writeFileSync(out, JSON.stringify(withIndex, null, 2));
1785
2486
  console.log(`Output: ${out} (${chunks.length} chunks in "index" block)`);
1786
2487
  } else if (format === "json") {
1787
- const out = opts.output ? path2.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".chunks.json");
1788
- fs.writeFileSync(out, JSON.stringify(chunks, null, 2));
2488
+ const out = opts.output ? path7.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".chunks.json");
2489
+ fs7.writeFileSync(out, JSON.stringify(chunks, null, 2));
1789
2490
  console.log(`Output: ${out} (${chunks.length} chunks)`);
1790
2491
  } else {
1791
- const out = opts.output ? path2.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".chunks.jsonl");
1792
- fs.writeFileSync(out, chunks.map((c) => JSON.stringify(c)).join("\n") + "\n");
2492
+ const out = opts.output ? path7.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".chunks.jsonl");
2493
+ fs7.writeFileSync(out, chunks.map((c) => JSON.stringify(c)).join("\n") + "\n");
1793
2494
  console.log(`Output: ${out} (${chunks.length} chunks)`);
1794
2495
  }
1795
2496
  const totalTokens = chunks.reduce((a, c) => a + c.tokens, 0);
@@ -1961,17 +2662,17 @@ async function embedOpenAI(model, inputs) {
1961
2662
  }
1962
2663
  async function loadJdf2(filePath) {
1963
2664
  if (filePath.toLowerCase().endsWith(".jdfx")) {
1964
- const zip = await JSZip.loadAsync(fs.readFileSync(filePath));
2665
+ const zip = await JSZip.loadAsync(fs7.readFileSync(filePath));
1965
2666
  const docFile = zip.file(JDFX_DOCUMENT_PATH);
1966
2667
  if (!docFile) throw new Error(`Bundle missing ${JDFX_DOCUMENT_PATH}`);
1967
2668
  return JSON.parse(await docFile.async("string"));
1968
2669
  }
1969
- return JSON.parse(fs.readFileSync(filePath, "utf-8"));
2670
+ return JSON.parse(fs7.readFileSync(filePath, "utf-8"));
1970
2671
  }
1971
2672
  function loadCache(cachePath) {
1972
2673
  try {
1973
- if (!fs.existsSync(cachePath)) return null;
1974
- return JSON.parse(fs.readFileSync(cachePath, "utf-8"));
2674
+ if (!fs7.existsSync(cachePath)) return null;
2675
+ return JSON.parse(fs7.readFileSync(cachePath, "utf-8"));
1975
2676
  } catch {
1976
2677
  return null;
1977
2678
  }
@@ -1982,15 +2683,15 @@ function batched(items, size) {
1982
2683
  return out;
1983
2684
  }
1984
2685
  async function embedFile(inputPath, opts = {}) {
1985
- const input = path2.resolve(inputPath);
1986
- if (!fs.existsSync(input)) throw new Error(`File not found: ${input}`);
2686
+ const input = path7.resolve(inputPath);
2687
+ if (!fs7.existsSync(input)) throw new Error(`File not found: ${input}`);
1987
2688
  const provider = opts.provider ?? "ollama";
1988
2689
  const model = opts.model ?? DEFAULT_MODEL[provider];
1989
2690
  const strategy = opts.strategy ?? "section";
1990
2691
  const doc = await loadJdf2(input);
1991
- const chunks = chunkDocument(doc, { strategy, maxTokens: opts.maxTokens });
1992
- const output = opts.output ? path2.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".embeddings.json");
1993
- const cachePath = opts.cache ? path2.resolve(opts.cache) : output;
2692
+ const chunks = chunkDocument(doc, { strategy, maxTokens: opts.maxTokens, transcriptWindowSec: opts.transcriptWindowSec });
2693
+ const output = opts.output ? path7.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".embeddings.json");
2694
+ const cachePath = opts.cache ? path7.resolve(opts.cache) : output;
1994
2695
  console.log(`Embedding: ${input}`);
1995
2696
  console.log(`Provider: ${provider} / ${model}${provider === "ollama" ? " (local \u2014 no data leaves this machine)" : " (remote API)"}`);
1996
2697
  console.log(`Strategy: ${strategy} \u2192 ${chunks.length} chunks`);
@@ -2029,11 +2730,295 @@ async function embedFile(inputPath, opts = {}) {
2029
2730
  chunker: `jdf-${strategy}-v1`,
2030
2731
  vectors
2031
2732
  };
2032
- fs.writeFileSync(output, JSON.stringify(sidecar));
2733
+ fs7.writeFileSync(output, JSON.stringify(sidecar));
2033
2734
  console.log(`
2034
2735
  Done! ${Object.keys(vectors).length} vectors (${dims}-dim) \u2192 ${output}`);
2035
2736
  return sidecar;
2036
2737
  }
2738
+ var toSec = (ts) => {
2739
+ const m = ts.trim().replace(",", ".").match(/^(?:(\d+):)?(\d{1,2}):(\d{2}(?:\.\d+)?)$/);
2740
+ if (!m) throw new Error(`bad timestamp "${ts}"`);
2741
+ return (m[1] ? Number(m[1]) * 3600 : 0) + Number(m[2]) * 60 + Number(m[3]);
2742
+ };
2743
+ function parseSubtitles(text, filename = "") {
2744
+ const trimmed = text.replace(/^/, "").trim();
2745
+ if (trimmed.startsWith("{") || trimmed.startsWith("[")) {
2746
+ const j = JSON.parse(trimmed);
2747
+ const arr = Array.isArray(j) ? j : Array.isArray(j.segments) ? j.segments : [];
2748
+ return arr.map((sg) => ({
2749
+ t0: Number(sg.t0 ?? sg.start ?? sg.from ?? 0),
2750
+ t1: Number(sg.t1 ?? sg.end ?? sg.to ?? 0),
2751
+ text: String(sg.text ?? "").trim(),
2752
+ ...sg.speaker ? { speaker: String(sg.speaker) } : {}
2753
+ })).filter((sg) => sg.text);
2754
+ }
2755
+ const segs = [];
2756
+ for (const block of trimmed.split(/\r?\n\r?\n+/)) {
2757
+ const lines = block.split(/\r?\n/).filter((l) => l.trim() !== "" && l.trim() !== "WEBVTT");
2758
+ const ti = lines.findIndex((l) => l.includes("-->"));
2759
+ if (ti < 0) continue;
2760
+ const [a, b] = lines[ti].split("-->").map((x) => x.trim().split(/\s+/)[0]);
2761
+ const body = lines.slice(ti + 1).join(" ").replace(/<[^>]+>/g, "").replace(/\s+/g, " ").trim();
2762
+ if (!body) continue;
2763
+ segs.push({ t0: toSec(a), t1: toSec(b), text: body });
2764
+ }
2765
+ if (!segs.length) throw new Error(`no cues found in ${filename || "subtitle input"} (expected SRT, WebVTT or JSON segments)`);
2766
+ return segs;
2767
+ }
2768
+ function parseChapters(text) {
2769
+ const t = text.trim();
2770
+ if (t.startsWith("[")) return JSON.parse(t).map((c) => ({ t: Number(c.t ?? c.start ?? 0), title: String(c.title ?? "") }));
2771
+ return t.split(/\r?\n/).map((l) => l.trim()).filter(Boolean).map((l) => {
2772
+ const m = l.match(/^(\S+)\s+(.+)$/);
2773
+ if (!m) throw new Error(`bad chapter line "${l}" (expected "mm:ss Title")`);
2774
+ return { t: toSec(m[1]), title: m[2].trim() };
2775
+ });
2776
+ }
2777
+ async function loadDoc(file) {
2778
+ if (file.toLowerCase().endsWith(".jdfx")) {
2779
+ const zip = await JSZip.loadAsync(fs7.readFileSync(file));
2780
+ const f = zip.file(JDFX_DOCUMENT_PATH);
2781
+ if (!f) throw new Error(`Bundle missing ${JDFX_DOCUMENT_PATH}`);
2782
+ const doc = JSON.parse(await f.async("string"));
2783
+ const manifest = zip.file("manifest.json") ? JSON.parse(await zip.file("manifest.json").async("string")) : { assets: [] };
2784
+ for (const a of manifest.assets ?? []) {
2785
+ const af = zip.file(a.path);
2786
+ if (!af) continue;
2787
+ const data = (await af.async("nodebuffer")).toString("base64");
2788
+ const res = { src: "embedded", mimeType: a.mimeType, data };
2789
+ doc.resources ??= {};
2790
+ if (/^video\//i.test(a.mimeType || "")) (doc.resources.videos ??= {})[a.id] = res;
2791
+ else (doc.resources.images ??= {})[a.id] = res;
2792
+ }
2793
+ return { doc, bundle: true, zip };
2794
+ }
2795
+ return { doc: JSON.parse(fs7.readFileSync(file, "utf-8")), bundle: false };
2796
+ }
2797
+ function findVideos(doc) {
2798
+ const out = [];
2799
+ const walk2 = (els, page) => {
2800
+ for (const el of els ?? []) {
2801
+ if (el?.type === "video") out.push({ el, page, index: out.length });
2802
+ if (el?.elements) walk2(el.elements, page);
2803
+ }
2804
+ };
2805
+ doc.pages.forEach((p, i) => walk2(p.elements, i + 1));
2806
+ return out;
2807
+ }
2808
+ async function clipToTempFile(doc, el, docDir) {
2809
+ const tmp = path7.join(fs7.mkdtempSync(path7.join(__require("os").tmpdir(), "jdf-transcribe-")), "clip.mp4");
2810
+ const res = el.resource ? doc.resources?.videos?.[el.resource] ?? doc.resources?.images?.[el.resource] : void 0;
2811
+ if (res?.data) {
2812
+ fs7.writeFileSync(tmp, Buffer.from(res.data.replace(/^data:[^,]*,/, ""), "base64"));
2813
+ return tmp;
2814
+ }
2815
+ if (res?.path) return path7.resolve(docDir, res.path);
2816
+ const src = el.src;
2817
+ if (!src) return null;
2818
+ if (src.startsWith("data:")) {
2819
+ fs7.writeFileSync(tmp, Buffer.from(src.replace(/^data:[^,]*,/, ""), "base64"));
2820
+ return tmp;
2821
+ }
2822
+ if (/^https?:\/\//i.test(src)) {
2823
+ const r = await fetch(src);
2824
+ if (!r.ok) throw new Error(`download failed ${r.status}: ${src}`);
2825
+ fs7.writeFileSync(tmp, Buffer.from(await r.arrayBuffer()));
2826
+ return tmp;
2827
+ }
2828
+ const local = path7.resolve(docDir, src);
2829
+ return fs7.existsSync(local) ? local : null;
2830
+ }
2831
+ function whisperCli(clip, model, language, prompt2) {
2832
+ const ffmpeg = spawnSync("ffmpeg", ["-version"]);
2833
+ if (ffmpeg.error) throw new Error("ffmpeg not found \u2014 needed to extract audio for whisper-cli (brew install ffmpeg)");
2834
+ const wav = clip.replace(/\.[^.]+$/, "") + ".16k.wav";
2835
+ const ex = spawnSync("ffmpeg", ["-y", "-i", clip, "-vn", "-ac", "1", "-ar", "16000", "-f", "wav", wav], { encoding: "utf-8" });
2836
+ if (ex.status !== 0) throw new Error(`ffmpeg failed: ${ex.stderr.slice(-400)}`);
2837
+ const args = ["-f", wav, "-oj", "-of", wav.replace(/\.wav$/, "")];
2838
+ if (model) args.push("-m", model);
2839
+ if (language) args.push("-l", language);
2840
+ if (prompt2) args.push("--prompt", prompt2);
2841
+ const run = spawnSync("whisper-cli", args, { encoding: "utf-8" });
2842
+ if (run.error) throw new Error("whisper-cli not found \u2014 install whisper.cpp (brew install whisper-cpp) or use --from / --provider openai");
2843
+ if (run.status !== 0) throw new Error(`whisper-cli failed: ${run.stderr.slice(-400)}`);
2844
+ const j = JSON.parse(fs7.readFileSync(wav.replace(/\.wav$/, "") + ".json", "utf-8"));
2845
+ const segs = j.transcription ?? j.segments ?? [];
2846
+ const ms = (x) => typeof x === "number" ? x / 1e3 : toSec(String(x).replace(",", "."));
2847
+ return segs.map((sg) => ({ t0: ms(sg.offsets?.from ?? sg.start), t1: ms(sg.offsets?.to ?? sg.end), text: String(sg.text ?? "").trim() })).filter((sg) => sg.text);
2848
+ }
2849
+ async function openaiTranscribe(clip, model, language, prompt2) {
2850
+ const key = process.env.OPENAI_API_KEY;
2851
+ if (!key) throw new Error("OPENAI_API_KEY is not set");
2852
+ const base = process.env.OPENAI_BASE_URL || "https://api.openai.com/v1";
2853
+ const form = new FormData();
2854
+ form.append("file", new Blob([fs7.readFileSync(clip)]), path7.basename(clip));
2855
+ form.append("model", model || "whisper-1");
2856
+ form.append("response_format", "verbose_json");
2857
+ form.append("timestamp_granularities[]", "segment");
2858
+ if (language) form.append("language", language);
2859
+ if (prompt2) form.append("prompt", prompt2);
2860
+ const r = await fetch(`${base}/audio/transcriptions`, { method: "POST", headers: { Authorization: `Bearer ${key}` }, body: form });
2861
+ if (!r.ok) throw new Error(`OpenAI transcription failed ${r.status}: ${(await r.text()).slice(0, 300)}`);
2862
+ const j = await r.json();
2863
+ return (j.segments ?? []).map((sg) => ({ t0: Number(sg.start), t1: Number(sg.end), text: String(sg.text).trim() })).filter((sg) => sg.text);
2864
+ }
2865
+ async function transcribeFile(inputPath, opts = {}) {
2866
+ const input = path7.resolve(inputPath);
2867
+ if (!fs7.existsSync(input)) throw new Error(`File not found: ${input}`);
2868
+ const { doc, bundle } = await loadDoc(input);
2869
+ const videos = findVideos(doc);
2870
+ if (!videos.length) throw new Error("document has no video element");
2871
+ let target = videos[0];
2872
+ if (opts.element != null) {
2873
+ const byId = videos.find((v) => v.el.id === opts.element);
2874
+ const byIdx = /^\d+$/.test(opts.element) ? videos[Number(opts.element)] : void 0;
2875
+ target = byId ?? byIdx ?? (() => {
2876
+ throw new Error(`no video element "${opts.element}" (have: ${videos.map((v) => v.el.id ?? `#${v.index}`).join(", ")})`);
2877
+ })();
2878
+ } else if (videos.length > 1) {
2879
+ throw new Error(`document has ${videos.length} videos \u2014 pick one with --element <id|index>`);
2880
+ }
2881
+ let segments;
2882
+ let source;
2883
+ if (opts.from) {
2884
+ segments = parseSubtitles(fs7.readFileSync(path7.resolve(opts.from), "utf-8"), opts.from);
2885
+ source = `${path7.extname(opts.from).slice(1).toLowerCase() || "file"}-import`;
2886
+ } else {
2887
+ const provider = opts.provider ?? "whisper-cli";
2888
+ const clip = await clipToTempFile(doc, target.el, path7.dirname(input));
2889
+ if (!clip) throw new Error("could not locate the clip bytes (no bundled asset, data URL, local path or http URL) \u2014 use --from to import subtitles instead");
2890
+ segments = provider === "openai" ? await openaiTranscribe(clip, opts.model, opts.language, opts.prompt) : whisperCli(clip, opts.model, opts.language, opts.prompt);
2891
+ source = provider === "openai" ? `openai:${opts.model || "whisper-1"}` : `whisper-cli${opts.model ? ":" + path7.basename(opts.model) : ""}`;
2892
+ }
2893
+ segments.sort((a, b) => a.t0 - b.t0);
2894
+ const transcript = { ...opts.language ? { language: opts.language } : {}, source, created: (/* @__PURE__ */ new Date()).toISOString(), segments };
2895
+ target.el.transcript = transcript;
2896
+ if (opts.chapters) target.el.chapters = parseChapters(fs7.readFileSync(path7.resolve(opts.chapters), "utf-8"));
2897
+ if (!target.el.id) target.el.id = `video-${target.index + 1}`;
2898
+ const output = opts.output ? path7.resolve(opts.output) : input;
2899
+ if (output.toLowerCase().endsWith(".jdfx") || bundle && !opts.output) {
2900
+ const { bytes } = await packJdfx(doc);
2901
+ fs7.writeFileSync(output, bytes);
2902
+ } else {
2903
+ fs7.writeFileSync(output, JSON.stringify(doc, null, 2));
2904
+ }
2905
+ const dur = segments.length ? segments[segments.length - 1].t1 : 0;
2906
+ console.log(`Transcribed: ${path7.basename(input)} \u2192 element "${target.el.id}" (${segments.length} segments, ${Math.round(dur)} s, source ${source})`);
2907
+ if (target.el.chapters) console.log(`Chapters: ${target.el.chapters.length}`);
2908
+ console.log(`Output: ${output}
2909
+ Next: jdf chunk ${path7.basename(output)} # transcript \u2192 time-windowed chunks with media.t0/t1`);
2910
+ return transcript;
2911
+ }
2912
+ var CONFIG_NAME = "jdf.rag.json";
2913
+ var OUT_DIR = ".jdf-rag";
2914
+ function walk(dir, acc = []) {
2915
+ for (const ent of fs7.readdirSync(dir, { withFileTypes: true })) {
2916
+ if (ent.name === "node_modules" || ent.name === OUT_DIR || ent.name.startsWith(".")) continue;
2917
+ const p = path7.join(dir, ent.name);
2918
+ if (ent.isDirectory()) walk(p, acc);
2919
+ else if (/\.(jdf|jdfx)$/i.test(ent.name)) acc.push(p);
2920
+ }
2921
+ return acc.sort();
2922
+ }
2923
+ async function readDoc(file) {
2924
+ if (file.toLowerCase().endsWith(".jdfx")) {
2925
+ const zip = await JSZip.loadAsync(fs7.readFileSync(file));
2926
+ return JSON.parse(await zip.file(JDFX_DOCUMENT_PATH).async("string"));
2927
+ }
2928
+ return JSON.parse(fs7.readFileSync(file, "utf-8"));
2929
+ }
2930
+ function videosIn(doc) {
2931
+ const out = [];
2932
+ const w = (els) => {
2933
+ for (const el of els ?? []) {
2934
+ if (el?.type === "video") out.push({ id: el.id, hasTranscript: !!el.transcript?.segments?.length });
2935
+ if (el?.elements) w(el.elements);
2936
+ }
2937
+ };
2938
+ for (const p of doc.pages ?? []) w(p.elements);
2939
+ return out;
2940
+ }
2941
+ async function ragFolder(dirPath, cli = {}) {
2942
+ const dir = path7.resolve(dirPath);
2943
+ if (!fs7.existsSync(dir) || !fs7.statSync(dir).isDirectory()) throw new Error(`Not a directory: ${dir}`);
2944
+ const cfgPath = path7.join(dir, CONFIG_NAME);
2945
+ const cfg = fs7.existsSync(cfgPath) ? JSON.parse(fs7.readFileSync(cfgPath, "utf-8")) : {};
2946
+ const opts = { ...cfg, ...Object.fromEntries(Object.entries(cli).filter(([, v]) => v !== void 0)) };
2947
+ const provider = opts.provider ?? "ollama";
2948
+ const transcribe = opts.transcribe ?? "none";
2949
+ const outDir = path7.resolve(opts.out ?? path7.join(dir, OUT_DIR));
2950
+ const files = walk(dir);
2951
+ console.log(`jdf rag: ${dir}
2952
+ files: ${files.length} (.jdf/.jdfx)${fs7.existsSync(cfgPath) ? `
2953
+ config: ${CONFIG_NAME}` : ""}
2954
+ embeddings: ${opts.noEmbed ? "skipped (--no-embed)" : `${provider}${opts.model ? " / " + opts.model : ""}`}
2955
+ transcribe: ${transcribe}${opts.dryRun ? "\n DRY RUN \u2014 nothing written" : ""}
2956
+ `);
2957
+ if (!files.length) {
2958
+ console.log("Nothing to do.");
2959
+ return;
2960
+ }
2961
+ const manifest = { dir, created: (/* @__PURE__ */ new Date()).toISOString(), provider: opts.noEmbed ? null : provider, model: opts.model ?? null, strategy: opts.strategy ?? "section", transcribe, files: [], totals: { files: files.length, chunks: 0, videoChunks: 0, videos: 0, transcribed: 0, untranscribed: 0 } };
2962
+ const indexLines = [];
2963
+ for (const file of files) {
2964
+ const rel = path7.relative(dir, file);
2965
+ const doc = await readDoc(file);
2966
+ const vids = videosIn(doc);
2967
+ let transcribedHere = 0;
2968
+ for (const v of vids) {
2969
+ if (v.hasTranscript) continue;
2970
+ if (transcribe === "none") {
2971
+ manifest.totals.untranscribed++;
2972
+ continue;
2973
+ }
2974
+ if (opts.dryRun) {
2975
+ transcribedHere++;
2976
+ continue;
2977
+ }
2978
+ try {
2979
+ await transcribeFile(file, { provider: transcribe, element: v.id, model: opts.transcribeModel, language: opts.language, prompt: opts.prompt });
2980
+ transcribedHere++;
2981
+ } catch (e) {
2982
+ console.warn(` ! ${rel}: transcription failed for video ${v.id ?? "#?"}: ${e.message}`);
2983
+ manifest.totals.untranscribed++;
2984
+ }
2985
+ }
2986
+ manifest.totals.videos += vids.length;
2987
+ manifest.totals.transcribed += transcribedHere;
2988
+ if (opts.dryRun) {
2989
+ manifest.files.push({ file: rel, videos: vids.length, wouldTranscribe: transcribedHere });
2990
+ continue;
2991
+ }
2992
+ const chunkOpts = { strategy: opts.strategy, maxTokens: opts.maxTokens, transcriptWindowSec: opts.transcriptWindowSec };
2993
+ const chunkOut = path7.join(outDir, "chunks", rel.replace(/\.(jdf|jdfx)$/i, ".chunks.jsonl"));
2994
+ fs7.mkdirSync(path7.dirname(chunkOut), { recursive: true });
2995
+ let chunks;
2996
+ if (opts.noEmbed) {
2997
+ chunks = await chunkFile(file, { ...chunkOpts, format: "jsonl", output: chunkOut });
2998
+ } else {
2999
+ const side = await embedFile(file, { ...chunkOpts, provider, model: opts.model, incremental: true });
3000
+ chunks = await chunkFile(file, { ...chunkOpts, format: "jsonl", output: chunkOut });
3001
+ manifest.files.push({ file: rel, chunks: chunks.length, vectors: Object.keys(side.vectors).length, sidecar: path7.relative(dir, file.replace(/\.(jdf|jdfx)$/i, ".embeddings.json")), videos: vids.length, transcribed: transcribedHere });
3002
+ }
3003
+ if (opts.noEmbed) manifest.files.push({ file: rel, chunks: chunks.length, videos: vids.length, transcribed: transcribedHere });
3004
+ for (const c of chunks) {
3005
+ indexLines.push(JSON.stringify({ file: rel, ...c }));
3006
+ manifest.totals.chunks++;
3007
+ if (c.media) manifest.totals.videoChunks++;
3008
+ }
3009
+ }
3010
+ if (!opts.dryRun) {
3011
+ fs7.mkdirSync(outDir, { recursive: true });
3012
+ fs7.writeFileSync(path7.join(outDir, "index.jsonl"), indexLines.join("\n") + (indexLines.length ? "\n" : ""));
3013
+ fs7.writeFileSync(path7.join(outDir, "manifest.json"), JSON.stringify(manifest, null, 2) + "\n");
3014
+ }
3015
+ const t = manifest.totals;
3016
+ console.log(`
3017
+ Done. ${t.files} files \u2192 ${t.chunks} chunks (${t.videoChunks} from video transcripts); videos ${t.videos}, transcribed now ${t.transcribed}, still without transcript ${t.untranscribed}${t.untranscribed && transcribe === "none" ? " (pass --transcribe whisper-cli|openai, or jdf transcribe --from subs.srt)" : ""}.`);
3018
+ if (!opts.dryRun) console.log(`Index: ${path7.join(outDir, "index.jsonl")}
3019
+ Report: ${path7.join(outDir, "manifest.json")}${opts.noEmbed ? "" : `
3020
+ Vectors: one <file>.embeddings.json next to each document (incremental \u2014 re-run any time)`}`);
3021
+ }
2037
3022
 
2038
3023
  // src/index.ts
2039
3024
  var HELP = `jdf \u2014 JSON Document Format CLI
@@ -2045,12 +3030,18 @@ The CLI exists for these workflows:
2045
3030
  into a validated .jdf (or .jdfx) you can ship.
2046
3031
  \u2022 JDF \u2192 chunks turn a document into retrieval-ready chunks (RAG).
2047
3032
  \u2022 JDF \u2192 vectors embed those chunks, incrementally, for a vector store.
3033
+ \u2022 video \u2192 text attach a time-stamped transcript to a video element so
3034
+ RAG retrieves "video at 02:13", not just "a video".
3035
+ \u2022 folder \u2192 index one command over a directory of .jdf/.jdfx: transcribe,
3036
+ chunk, embed incrementally, write .jdf-rag/index.jsonl.
2048
3037
 
2049
3038
  Usage:
2050
3039
  jdf validate <file.jdf>
2051
- jdf convert <file.{pdf,json,md}> [-o output.{jdf,jdfx}] [--json]
3040
+ jdf convert <file.{pdf,json,md}> [-o output.{jdf,jdfx}] [--json] [--password PW] [--drop-invisible-text]
2052
3041
  jdf chunk <file.{jdf,jdfx}> [--strategy section|element|fixed] [--format jsonl|json|inline] [--max-tokens N] [-o out]
2053
3042
  jdf embed <file.{jdf,jdfx}> [--provider ollama|openai] [--model NAME] [--strategy \u2026] [--incremental] [-o out]
3043
+ jdf transcribe <file.{jdf,jdfx}> [--from subs.srt|.vtt|.json] [--provider whisper-cli|openai] [--element ID] [--chapters FILE] [-o out]
3044
+ jdf rag <dir> [--provider ollama|openai] [--model NAME] [--transcribe none|whisper-cli|openai] [--no-embed] [--dry-run] [--out DIR]
2054
3045
  jdf --help
2055
3046
 
2056
3047
  Commands:
@@ -2058,18 +3049,38 @@ Commands:
2058
3049
  convert Convert a PDF, JSON, or Markdown file into JDF (alias: import)
2059
3050
  chunk Split a JDF document into retrieval-ready chunks (offline, deterministic)
2060
3051
  embed Compute embeddings for the chunks (local via Ollama by default)
3052
+ transcribe Store time-stamped text on a video element (import SRT/VTT/JSON, or run Whisper)
3053
+ rag Make a whole folder retrieval-ready (finds .jdf/.jdfx, transcribes, chunks, embeds, indexes)
2061
3054
 
2062
3055
  Flags:
2063
3056
  -o, --output <path> Explicit output path
2064
3057
  --json convert: force pure JSON .jdf output (inline base64
2065
3058
  instead of a .jdfx zip bundle)
3059
+ --password <pw> convert(pdf): password for an encrypted PDF
3060
+ --drop-invisible-text
3061
+ convert(pdf): omit invisible (OCR-layer) text; by
3062
+ default it is kept with opacity 0 so RAG / search
3063
+ still see the words of a scanned PDF
2066
3064
  --strategy <s> chunk/embed: section (default) | element | fixed
2067
3065
  --format <f> chunk: jsonl (default) | json | inline
2068
3066
  --max-tokens <n> chunk/embed: soft cap per chunk (default 512)
2069
3067
  --provider <p> embed: ollama (default, local) | openai (remote API)
2070
3068
  --model <name> embed: model id (default: nomic-embed-text / text-embedding-3-small)
2071
3069
  --incremental embed: skip chunks whose content hash is unchanged
3070
+ --cache <path> embed: sidecar to reuse vectors from (default: the
3071
+ output path itself)
2072
3072
  --no-auto-start embed(ollama): don't auto-launch Ollama via Docker
3073
+ --from <file> transcribe: import subtitles (.srt / .vtt / JSON segments) \u2014 offline, no model
3074
+ --element <id|n> transcribe: which video element (id, or 0-based index); default the only one
3075
+ --chapters <file> transcribe: JSON [{t,title}] or "mm:ss Title" lines \u2192 chapter breadcrumbs
3076
+ --language <tag> transcribe: BCP-47 language hint for Whisper
3077
+ --prompt <text> transcribe: Whisper vocabulary hint (names, acronyms) \u2014 not an instruction
3078
+ --window <sec> chunk/embed/rag: transcript window per video chunk (default 45)
3079
+ --transcribe <p> rag: none (default) | whisper-cli | openai \u2014 for videos that have no transcript yet
3080
+ --no-embed rag: chunk + index only
3081
+ --dry-run rag: list what would happen, write nothing
3082
+ --out <dir> rag: index folder (default <dir>/.jdf-rag)
3083
+ rag reads defaults from <dir>/jdf.rag.json (same keys as the flags; flags win)
2073
3084
 
2074
3085
  Environment (embed):
2075
3086
  ollama: OLLAMA_HOST (default http://localhost:11434)
@@ -2083,8 +3094,11 @@ Examples:
2083
3094
  jdf chunk report.jdf --format inline # embed the chunk index into the .jdf
2084
3095
  jdf embed report.jdf # local embeddings via Ollama (auto-setup)
2085
3096
  jdf embed report.jdf --provider openai --incremental
3097
+ jdf transcribe talk.jdfx --from talk.srt --chapters chapters.txt # then: jdf chunk talk.jdfx
3098
+ jdf transcribe talk.jdfx --provider openai --language en --prompt "JDF, jdfx, Ollama"
3099
+ jdf rag ./knowledge-base --transcribe openai # whole folder \u2192 .jdf-rag/index.jsonl
2086
3100
  `;
2087
- var BOOLEAN_FLAGS = /* @__PURE__ */ new Set(["help", "h", "json", "verbose", "skip-validate", "incremental", "no-auto-start"]);
3101
+ var BOOLEAN_FLAGS = /* @__PURE__ */ new Set(["help", "h", "json", "verbose", "skip-validate", "incremental", "no-auto-start", "drop-invisible-text", "no-embed", "dry-run"]);
2088
3102
  function parseArgs(argv) {
2089
3103
  const positional = [];
2090
3104
  const flags = {};
@@ -2164,7 +3178,11 @@ async function main() {
2164
3178
  await importMarkdown(input, output);
2165
3179
  process.exit(0);
2166
3180
  } else if (lower.endsWith(".pdf")) {
2167
- await importPdf(input, output, { forceJson });
3181
+ await importPdf(input, output, {
3182
+ forceJson,
3183
+ password: typeof flags.password === "string" ? flags.password : void 0,
3184
+ dropInvisibleText: flags["drop-invisible-text"] === true
3185
+ });
2168
3186
  process.exit(0);
2169
3187
  } else if (lower.endsWith(".json")) {
2170
3188
  await importJson(input, output, { forceJson });
@@ -2184,14 +3202,55 @@ async function main() {
2184
3202
  strategy: typeof flags.strategy === "string" ? flags.strategy : void 0,
2185
3203
  format: typeof flags.format === "string" ? flags.format : void 0,
2186
3204
  maxTokens: typeof flags["max-tokens"] === "string" ? parseInt(flags["max-tokens"], 10) : void 0,
3205
+ transcriptWindowSec: typeof flags.window === "string" ? parseInt(flags.window, 10) : void 0,
2187
3206
  output: typeof flags.output === "string" ? flags.output : void 0
2188
3207
  });
2189
3208
  process.exit(0);
2190
3209
  }
3210
+ case "transcribe": {
3211
+ const input = positional[0];
3212
+ if (!input) {
3213
+ console.error("Usage: jdf transcribe <file.{jdf,jdfx}> [--from subs.srt|.vtt|.json] [--provider whisper-cli|openai] [--model M] [--language tag] [--element id|n] [--chapters file] [-o out]");
3214
+ process.exit(1);
3215
+ }
3216
+ await transcribeFile(input, {
3217
+ from: typeof flags.from === "string" ? flags.from : void 0,
3218
+ provider: typeof flags.provider === "string" ? flags.provider : void 0,
3219
+ model: typeof flags.model === "string" ? flags.model : void 0,
3220
+ language: typeof flags.language === "string" ? flags.language : void 0,
3221
+ prompt: typeof flags.prompt === "string" ? flags.prompt : void 0,
3222
+ element: typeof flags.element === "string" ? flags.element : void 0,
3223
+ chapters: typeof flags.chapters === "string" ? flags.chapters : void 0,
3224
+ output: typeof flags.output === "string" ? flags.output : void 0
3225
+ });
3226
+ process.exit(0);
3227
+ }
3228
+ case "rag": {
3229
+ const input = positional[0];
3230
+ if (!input) {
3231
+ console.error("Usage: jdf rag <dir> [--provider ollama|openai] [--model NAME] [--strategy \u2026] [--window sec] [--transcribe none|whisper-cli|openai] [--language tag] [--prompt text] [--no-embed] [--dry-run] [--out DIR]");
3232
+ process.exit(1);
3233
+ }
3234
+ await ragFolder(input, {
3235
+ provider: typeof flags.provider === "string" ? flags.provider : void 0,
3236
+ model: typeof flags.model === "string" ? flags.model : void 0,
3237
+ strategy: typeof flags.strategy === "string" ? flags.strategy : void 0,
3238
+ maxTokens: typeof flags["max-tokens"] === "string" ? parseInt(flags["max-tokens"], 10) : void 0,
3239
+ transcriptWindowSec: typeof flags.window === "string" ? parseInt(flags.window, 10) : void 0,
3240
+ transcribe: typeof flags.transcribe === "string" ? flags.transcribe : void 0,
3241
+ transcribeModel: typeof flags["transcribe-model"] === "string" ? flags["transcribe-model"] : void 0,
3242
+ language: typeof flags.language === "string" ? flags.language : void 0,
3243
+ prompt: typeof flags.prompt === "string" ? flags.prompt : void 0,
3244
+ noEmbed: flags["no-embed"] === true,
3245
+ dryRun: flags["dry-run"] === true,
3246
+ out: typeof flags.out === "string" ? flags.out : void 0
3247
+ });
3248
+ process.exit(0);
3249
+ }
2191
3250
  case "embed": {
2192
3251
  const input = positional[0];
2193
3252
  if (!input) {
2194
- console.error("Usage: jdf embed <file.{jdf,jdfx}> [--provider ollama|openai] [--model NAME] [--strategy \u2026] [--incremental] [--no-auto-start] [-o out]");
3253
+ console.error("Usage: jdf embed <file.{jdf,jdfx}> [--provider ollama|openai] [--model NAME] [--strategy \u2026] [--incremental] [--cache prev.embeddings.json] [--no-auto-start] [-o out]");
2195
3254
  process.exit(1);
2196
3255
  }
2197
3256
  await embedFile(input, {
@@ -2200,8 +3259,10 @@ async function main() {
2200
3259
  strategy: typeof flags.strategy === "string" ? flags.strategy : void 0,
2201
3260
  maxTokens: typeof flags["max-tokens"] === "string" ? parseInt(flags["max-tokens"], 10) : void 0,
2202
3261
  incremental: flags.incremental === true,
3262
+ transcriptWindowSec: typeof flags.window === "string" ? parseInt(flags.window, 10) : void 0,
2203
3263
  autoStart: flags["no-auto-start"] !== true,
2204
- output: typeof flags.output === "string" ? flags.output : void 0
3264
+ output: typeof flags.output === "string" ? flags.output : void 0,
3265
+ cache: typeof flags.cache === "string" ? flags.cache : void 0
2205
3266
  });
2206
3267
  process.exit(0);
2207
3268
  }