@uurtech/jdf-cli 0.1.21 → 0.1.23

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -5,8 +5,9 @@ import { fileURLToPath } from 'url';
5
5
  import Ajv from 'ajv';
6
6
  import addFormats from 'ajv-formats';
7
7
  import JSZip from 'jszip';
8
- import { createHash } from 'crypto';
8
+ import crypto, { createHash } from 'crypto';
9
9
  import { readFile } from 'fs/promises';
10
+ import { execFileSync } from 'child_process';
10
11
 
11
12
  // ../../packages/jdf-core/src/manifest.ts
12
13
  var JDFX_MANIFEST_VERSION = "1.0.0";
@@ -240,6 +241,58 @@ function shouldUseJdfx(doc) {
240
241
  }
241
242
 
242
243
  // src/commands/import-md.ts
244
+ function parseInline(text) {
245
+ const runs = [];
246
+ let i = 0;
247
+ let buf = "";
248
+ const flush = (extra) => {
249
+ if (buf) {
250
+ runs.push({ text: buf, ...extra });
251
+ buf = "";
252
+ }
253
+ };
254
+ while (i < text.length) {
255
+ const rest = text.slice(i);
256
+ const link = rest.match(/^\[([^\]]+)\]\(([^)\s]+)[^)]*\)/);
257
+ if (link) {
258
+ flush();
259
+ runs.push({ text: link[1], link: link[2] });
260
+ i += link[0].length;
261
+ continue;
262
+ }
263
+ const code = rest.match(/^`([^`]+)`/);
264
+ if (code) {
265
+ flush();
266
+ runs.push({ text: code[1], fontFamily: "JetBrains Mono", color: "#be185d" });
267
+ i += code[0].length;
268
+ continue;
269
+ }
270
+ const bold = rest.match(/^(\*\*|__)([^]+?)\1/);
271
+ if (bold) {
272
+ flush();
273
+ runs.push(...parseInline(bold[2]).map((r) => ({ ...r, bold: true })));
274
+ i += bold[0].length;
275
+ continue;
276
+ }
277
+ const ital = rest.match(/^(\*|_)([^]+?)\1/);
278
+ if (ital) {
279
+ flush();
280
+ runs.push(...parseInline(ital[2]).map((r) => ({ ...r, italic: true })));
281
+ i += ital[0].length;
282
+ continue;
283
+ }
284
+ buf += text[i];
285
+ i++;
286
+ }
287
+ flush();
288
+ return runs.length ? runs : [{ text }];
289
+ }
290
+ function hasFormatting(runs) {
291
+ return runs.some((r) => r.bold || r.italic || r.link || r.fontFamily || r.color);
292
+ }
293
+ function stripInline(text) {
294
+ return parseInline(text).map((r) => r.text).join("");
295
+ }
243
296
  async function importMarkdown(inputPath, outputPath) {
244
297
  const input = path2.resolve(inputPath);
245
298
  console.log(`Importing: ${input}`);
@@ -330,11 +383,28 @@ function convertMarkdownToJdf(md, title, baseDir = process.cwd()) {
330
383
  i++;
331
384
  continue;
332
385
  }
333
- const imageMatch = line.match(/^\s*!\[([^\]]*)\]\(\s*([^)\s]+)[^)]*\)\s*$/);
334
- if (imageMatch) {
335
- const alt = imageMatch[1];
336
- const src = resolveImageSrc(imageMatch[2], baseDir);
337
- const height2 = 60;
386
+ if (line.match(/^\s*([-*_])(\s*\1){2,}\s*$/)) {
387
+ if (y + 6 > maxY) {
388
+ pages.push({ id: `page-${pageNum}`, elements });
389
+ elements = [];
390
+ y = 5;
391
+ pageNum++;
392
+ }
393
+ elements.push({ type: "shape", shape: "rect", position: { x: 0, y: y + 2 }, width: contentWidth, height: 0.3, fill: "#cbd5e1" });
394
+ y += 6;
395
+ i++;
396
+ continue;
397
+ }
398
+ if (line.includes("|") && i + 1 < lines.length && lines[i + 1].match(/^\s*\|?[\s:|-]+\|?\s*$/) && lines[i + 1].includes("-")) {
399
+ const splitRow = (r) => r.replace(/^\s*\|/, "").replace(/\|\s*$/, "").split("|").map((c) => stripInline(c.trim()));
400
+ const headers = splitRow(line);
401
+ i += 2;
402
+ const rows = [];
403
+ while (i < lines.length && lines[i].includes("|") && lines[i].trim() !== "") {
404
+ rows.push(splitRow(lines[i]));
405
+ i++;
406
+ }
407
+ const height2 = (rows.length + 1) * 7 + 3;
338
408
  if (y + height2 > maxY) {
339
409
  pages.push({ id: `page-${pageNum}`, elements });
340
410
  elements = [];
@@ -342,25 +412,27 @@ function convertMarkdownToJdf(md, title, baseDir = process.cwd()) {
342
412
  pageNum++;
343
413
  }
344
414
  elements.push({
345
- type: "image",
346
- src,
347
- alt,
415
+ type: "table",
416
+ headers,
417
+ rows,
348
418
  position: { x: 0, y },
349
419
  width: contentWidth,
350
- height: height2,
351
- fit: "contain"
420
+ borders: true,
421
+ headerStyle: { backgroundColor: "#f1f5f9", fontWeight: "bold", fontSize: 10, color: "#0f172a" },
422
+ rowStyle: { color: "#334155", fontSize: 10 },
423
+ style: { fontFamily: "Inter", fontSize: 10 }
352
424
  });
353
- y += height2 + 4;
354
- i++;
425
+ y += height2;
355
426
  continue;
356
427
  }
357
- if (line.match(/^\s*[-*+]\s/)) {
358
- const items = [];
359
- while (i < lines.length && lines[i].match(/^\s*[-*+]\s/)) {
360
- items.push({ content: lines[i].replace(/^\s*[-*+]\s+/, "").trim() });
428
+ if (line.match(/^\s*>\s?/)) {
429
+ const quoteLines = [];
430
+ while (i < lines.length && lines[i].match(/^\s*>\s?/)) {
431
+ quoteLines.push(lines[i].replace(/^\s*>\s?/, ""));
361
432
  i++;
362
433
  }
363
- const height2 = items.length * 5 + 3;
434
+ const quote = quoteLines.join(" ").trim();
435
+ const height2 = Math.ceil(quote.length / 70) * 4.5 + 6;
364
436
  if (y + height2 > maxY) {
365
437
  pages.push({ id: `page-${pageNum}`, elements });
366
438
  elements = [];
@@ -368,23 +440,66 @@ function convertMarkdownToJdf(md, title, baseDir = process.cwd()) {
368
440
  pageNum++;
369
441
  }
370
442
  elements.push({
371
- type: "list",
372
- listType: "unordered",
443
+ type: "text",
444
+ content: quote,
373
445
  position: { x: 0, y },
374
446
  width: contentWidth,
375
- style: { fontFamily: "Inter", fontSize: 10, lineHeight: 1.6, color: "#334155" },
376
- items
447
+ style: { fontFamily: "Inter", fontSize: 10, lineHeight: 1.6, color: "#475569", fontStyle: "italic", backgroundColor: "#f8fafc", padding: 10, borderRadius: 6, marginTop: 4 }
377
448
  });
378
449
  y += height2;
379
450
  continue;
380
451
  }
381
- if (line.match(/^\s*\d+\.\s/)) {
382
- const items = [];
383
- while (i < lines.length && lines[i].match(/^\s*\d+\.\s/)) {
384
- items.push({ content: lines[i].replace(/^\s*\d+\.\s+/, "").trim() });
452
+ const imageMatch = line.match(/^\s*!\[([^\]]*)\]\(\s*([^)\s]+)[^)]*\)\s*$/);
453
+ if (imageMatch) {
454
+ const alt = imageMatch[1];
455
+ const src = resolveImageSrc(imageMatch[2], baseDir);
456
+ const height2 = 60;
457
+ if (y + height2 > maxY) {
458
+ pages.push({ id: `page-${pageNum}`, elements });
459
+ elements = [];
460
+ y = 5;
461
+ pageNum++;
462
+ }
463
+ elements.push({
464
+ type: "image",
465
+ src,
466
+ alt,
467
+ position: { x: 0, y },
468
+ width: contentWidth,
469
+ height: height2,
470
+ fit: "contain"
471
+ });
472
+ y += height2 + 4;
473
+ i++;
474
+ continue;
475
+ }
476
+ const listMatch = line.match(/^(\s*)([-*+]|\d+\.)\s/);
477
+ if (listMatch) {
478
+ const raw = [];
479
+ while (i < lines.length) {
480
+ const m = lines[i].match(/^(\s*)([-*+]|\d+\.)\s+(.*)$/);
481
+ if (!m) break;
482
+ raw.push({ indent: m[1].length, ordered: /\d/.test(m[2]), content: m[3].trim() });
385
483
  i++;
386
484
  }
387
- const height2 = items.length * 5 + 3;
485
+ const rootOrdered = raw[0].ordered;
486
+ const root = [];
487
+ const stack = [{ indent: raw[0].indent, items: root }];
488
+ for (const r of raw) {
489
+ while (stack.length > 1 && r.indent < stack[stack.length - 1].indent) stack.pop();
490
+ if (r.indent > stack[stack.length - 1].indent) {
491
+ const parentItems = stack[stack.length - 1].items;
492
+ const parent = parentItems[parentItems.length - 1];
493
+ if (parent) {
494
+ parent.children = parent.children || [];
495
+ parent.listType = r.ordered ? "ordered" : "unordered";
496
+ stack.push({ indent: r.indent, items: parent.children });
497
+ }
498
+ }
499
+ stack[stack.length - 1].items.push({ content: stripInline(r.content) });
500
+ }
501
+ const count = raw.length;
502
+ const height2 = count * 5 + 3;
388
503
  if (y + height2 > maxY) {
389
504
  pages.push({ id: `page-${pageNum}`, elements });
390
505
  elements = [];
@@ -393,11 +508,11 @@ function convertMarkdownToJdf(md, title, baseDir = process.cwd()) {
393
508
  }
394
509
  elements.push({
395
510
  type: "list",
396
- listType: "ordered",
511
+ listType: rootOrdered ? "ordered" : "unordered",
397
512
  position: { x: 0, y },
398
513
  width: contentWidth,
399
514
  style: { fontFamily: "Inter", fontSize: 10, lineHeight: 1.6, color: "#334155" },
400
- items
515
+ items: root
401
516
  });
402
517
  y += height2;
403
518
  continue;
@@ -442,10 +557,15 @@ function convertMarkdownToJdf(md, title, baseDir = process.cwd()) {
442
557
  continue;
443
558
  }
444
559
  let para = "";
445
- while (i < lines.length && lines[i].trim() !== "" && !lines[i].match(/^#{1,6}\s/) && !lines[i].match(/^\s*[-*+]\s/) && !lines[i].match(/^\s*\d+\.\s/) && !lines[i].startsWith("```")) {
560
+ while (i < lines.length && lines[i].trim() !== "" && !lines[i].match(/^#{1,6}\s/) && !lines[i].match(/^\s*([-*+]|\d+\.)\s/) && !lines[i].startsWith("```") && !lines[i].match(/^\s*>\s?/) && !lines[i].match(/^\s*([-*_])(\s*\1){2,}\s*$/) && !lines[i].includes("|")) {
446
561
  para += (para ? " " : "") + lines[i].trim();
447
562
  i++;
448
563
  }
564
+ if (!para) {
565
+ para = lines[i].trim();
566
+ i++;
567
+ if (!para) continue;
568
+ }
449
569
  const paraLines = Math.ceil(para.length / 80);
450
570
  const height = paraLines * 4.5 + 3;
451
571
  if (y + height > maxY) {
@@ -454,13 +574,24 @@ function convertMarkdownToJdf(md, title, baseDir = process.cwd()) {
454
574
  y = 5;
455
575
  pageNum++;
456
576
  }
457
- elements.push({
458
- type: "text",
459
- content: para,
460
- position: { x: 0, y },
461
- width: contentWidth,
462
- style: { fontFamily: "Inter", fontSize: 10, lineHeight: 1.6, color: "#334155" }
463
- });
577
+ const runs = parseInline(para);
578
+ if (hasFormatting(runs)) {
579
+ elements.push({
580
+ type: "richtext",
581
+ runs,
582
+ position: { x: 0, y },
583
+ width: contentWidth,
584
+ style: { fontFamily: "Inter", fontSize: 10, lineHeight: 1.6, color: "#334155" }
585
+ });
586
+ } else {
587
+ elements.push({
588
+ type: "text",
589
+ content: para,
590
+ position: { x: 0, y },
591
+ width: contentWidth,
592
+ style: { fontFamily: "Inter", fontSize: 10, lineHeight: 1.6, color: "#334155" }
593
+ });
594
+ }
464
595
  y += height;
465
596
  }
466
597
  if (elements.length > 0) {
@@ -582,17 +713,35 @@ async function walkOps(page, OPS, viewport) {
582
713
  const y1Local = (va.y - minY) * PT_TO_MM;
583
714
  const x2Local = (vb.x - minX) * PT_TO_MM;
584
715
  const y2Local = (vb.y - minY) * PT_TO_MM;
585
- shapes.push({
586
- kind: "path",
587
- x: minX * PT_TO_MM,
588
- y: minY * PT_TO_MM,
589
- width: Math.max(0.05, (maxX - minX) * PT_TO_MM),
590
- height: Math.max(0.05, (maxY - minY) * PT_TO_MM),
591
- stroke: isStroke ? gs.stroke : void 0,
592
- strokeWidth: isStroke ? gs.lineWidth * PT_TO_MM : void 0,
593
- opacity: gs.strokeAlpha,
594
- path: `M ${x1Local.toFixed(2)} ${y1Local.toFixed(2)} L ${x2Local.toFixed(2)} ${y2Local.toFixed(2)}`
595
- });
716
+ const wLocal = Math.max(0.05, (maxX - minX) * PT_TO_MM);
717
+ const hLocal = Math.max(0.05, (maxY - minY) * PT_TO_MM);
718
+ const dx = Math.abs(va.x - vb.x);
719
+ const dy = Math.abs(va.y - vb.y);
720
+ const axisAligned = dx < 0.5 || dy < 0.5;
721
+ if (axisAligned) {
722
+ shapes.push({
723
+ kind: "line",
724
+ x: minX * PT_TO_MM,
725
+ y: minY * PT_TO_MM,
726
+ width: wLocal,
727
+ height: hLocal,
728
+ stroke: isStroke ? gs.stroke : void 0,
729
+ strokeWidth: isStroke ? gs.lineWidth * PT_TO_MM : void 0,
730
+ opacity: gs.strokeAlpha
731
+ });
732
+ } else {
733
+ shapes.push({
734
+ kind: "path",
735
+ x: minX * PT_TO_MM,
736
+ y: minY * PT_TO_MM,
737
+ width: wLocal,
738
+ height: hLocal,
739
+ stroke: isStroke ? gs.stroke : void 0,
740
+ strokeWidth: isStroke ? gs.lineWidth * PT_TO_MM : void 0,
741
+ opacity: gs.strokeAlpha,
742
+ path: `M ${x1Local.toFixed(2)} ${y1Local.toFixed(2)} L ${x2Local.toFixed(2)} ${y2Local.toFixed(2)}`
743
+ });
744
+ }
596
745
  } else if (pathSegments.length > 0) {
597
746
  const vpSegments = pathSegments.map((seg) => {
598
747
  if (seg.type === "Z") return seg;
@@ -1443,40 +1592,499 @@ function wrapElements(elements, title, meta) {
1443
1592
  ]
1444
1593
  };
1445
1594
  }
1595
+ var DEFAULT_MAX_TOKENS = 512;
1596
+ function estimateTokens(text) {
1597
+ return Math.ceil(text.length / 4);
1598
+ }
1599
+ function hashText(text) {
1600
+ return crypto.createHash("sha256").update(text).digest("hex").slice(0, 12);
1601
+ }
1602
+ function headingLevel(el) {
1603
+ const h = el.heading;
1604
+ if (h === true) return 1;
1605
+ if (typeof h === "number" && h >= 1 && h <= 6) return h;
1606
+ if (typeof el.tocLevel === "number") return el.tocLevel;
1607
+ return null;
1608
+ }
1609
+ function serializeElement(el) {
1610
+ const e = el;
1611
+ switch (e.type) {
1612
+ case "text":
1613
+ return String(e.content ?? "").trim();
1614
+ case "richtext":
1615
+ return (e.runs || []).map((r) => r.text ?? "").join("").trim();
1616
+ case "list": {
1617
+ const walk = (items, depth = 0) => (items || []).flatMap((it) => {
1618
+ const line = " ".repeat(depth) + "- " + String(it.content ?? "").trim();
1619
+ return it.children?.length ? [line, ...walk(it.children, depth + 1)] : [line];
1620
+ });
1621
+ return walk(e.items).join("\n");
1622
+ }
1623
+ case "table": {
1624
+ const headers = e.headers ?? e.columns?.map((c) => c.header || "").filter((h) => h) ?? [];
1625
+ const cell = (c) => typeof c === "string" ? c : c?.content ?? "";
1626
+ const rows = e.rows || [];
1627
+ const lines = rows.map(
1628
+ (row) => row.map((c, i) => {
1629
+ const h = headers[i];
1630
+ const v = cell(c).trim();
1631
+ return h ? `${h}: ${v}` : v;
1632
+ }).filter((s) => s !== "").join(" | ")
1633
+ );
1634
+ return lines.join("\n");
1635
+ }
1636
+ case "collapsible": {
1637
+ const title = String(e.title ?? "").trim();
1638
+ const inner = (e.elements || []).map(serializeElement).filter(Boolean).join("\n");
1639
+ return [title, inner].filter(Boolean).join("\n");
1640
+ }
1641
+ case "input":
1642
+ case "textarea":
1643
+ case "select": {
1644
+ const label = e.label ? `${e.label}: ` : "";
1645
+ const val = e.multiple ? (e.values || []).join(", ") : e.value ?? "";
1646
+ return `${label}${val}`.trim();
1647
+ }
1648
+ case "checkbox":
1649
+ return `${e.checked ? "[x]" : "[ ]"} ${e.label ?? ""}`.trim();
1650
+ case "image":
1651
+ return e.alt ? `[image: ${e.alt}]` : "";
1652
+ case "toc":
1653
+ case "shape":
1654
+ case "signature":
1655
+ return "";
1656
+ default:
1657
+ return "";
1658
+ }
1659
+ }
1660
+ function elementId(el, pageIdx, elIdx) {
1661
+ if (typeof el.id === "string" && el.id.length > 0) return el.id;
1662
+ return `p${pageIdx + 1}e${elIdx}`;
1663
+ }
1664
+ function flatten(doc) {
1665
+ const out = [];
1666
+ (doc.pages || []).forEach((page, pi) => {
1667
+ (page.elements || []).forEach((el, ei) => {
1668
+ out.push({ el, page: pi + 1, id: elementId(el, pi, ei) });
1669
+ });
1670
+ });
1671
+ return out;
1672
+ }
1673
+ function makeChunk(group, breadcrumb) {
1674
+ const parts = group.map((g) => serializeElement(g.el)).filter((s) => s.length > 0);
1675
+ const text = parts.join("\n\n").trim();
1676
+ if (!text) return null;
1677
+ const types = Array.from(new Set(group.map((g) => g.el.type)));
1678
+ return {
1679
+ id: group[0].id,
1680
+ text,
1681
+ path: [...breadcrumb],
1682
+ page: group[0].page,
1683
+ types,
1684
+ tokens: estimateTokens(text),
1685
+ hash: hashText(text)
1686
+ };
1687
+ }
1688
+ function chunkDocument(doc, options = {}) {
1689
+ const strategy = options.strategy ?? "section";
1690
+ const maxTokens = options.maxTokens ?? DEFAULT_MAX_TOKENS;
1691
+ const flat = flatten(doc);
1692
+ const chunks = [];
1693
+ if (strategy === "element") {
1694
+ const crumb2 = [];
1695
+ for (const f of flat) {
1696
+ const lvl = headingLevel(f.el);
1697
+ if (lvl != null) {
1698
+ crumb2.length = Math.max(0, lvl - 1);
1699
+ crumb2[lvl - 1] = serializeElement(f.el);
1700
+ }
1701
+ const c = makeChunk([f], crumb2);
1702
+ if (c) chunks.push(c);
1703
+ }
1704
+ return chunks;
1705
+ }
1706
+ if (strategy === "fixed") {
1707
+ const crumb2 = [];
1708
+ let buf2 = [];
1709
+ let bufTokens = 0;
1710
+ const flush = () => {
1711
+ const c = makeChunk(buf2, crumb2);
1712
+ if (c) chunks.push(c);
1713
+ buf2 = [];
1714
+ bufTokens = 0;
1715
+ };
1716
+ for (const f of flat) {
1717
+ const lvl = headingLevel(f.el);
1718
+ if (lvl != null) {
1719
+ crumb2.length = Math.max(0, lvl - 1);
1720
+ crumb2[lvl - 1] = serializeElement(f.el);
1721
+ }
1722
+ const t = estimateTokens(serializeElement(f.el));
1723
+ if (bufTokens + t > maxTokens && buf2.length > 0) flush();
1724
+ buf2.push(f);
1725
+ bufTokens += t;
1726
+ }
1727
+ flush();
1728
+ return chunks;
1729
+ }
1730
+ const crumb = [];
1731
+ let buf = [];
1732
+ const flushSection = () => {
1733
+ if (buf.length === 0) return;
1734
+ let sub = [];
1735
+ let subTokens = 0;
1736
+ for (const f of buf) {
1737
+ const t = estimateTokens(serializeElement(f.el));
1738
+ if (subTokens + t > maxTokens && sub.length > 0) {
1739
+ const c2 = makeChunk(sub, crumb);
1740
+ if (c2) chunks.push(c2);
1741
+ sub = [];
1742
+ subTokens = 0;
1743
+ }
1744
+ sub.push(f);
1745
+ subTokens += t;
1746
+ }
1747
+ const c = makeChunk(sub, crumb);
1748
+ if (c) chunks.push(c);
1749
+ buf = [];
1750
+ };
1751
+ for (const f of flat) {
1752
+ const lvl = headingLevel(f.el);
1753
+ if (lvl != null) {
1754
+ flushSection();
1755
+ crumb.length = Math.max(0, lvl - 1);
1756
+ crumb[lvl - 1] = serializeElement(f.el);
1757
+ }
1758
+ buf.push(f);
1759
+ }
1760
+ flushSection();
1761
+ return chunks;
1762
+ }
1763
+ async function loadJdf(filePath) {
1764
+ if (filePath.toLowerCase().endsWith(".jdfx")) {
1765
+ const zip = await JSZip.loadAsync(fs.readFileSync(filePath));
1766
+ const docFile = zip.file(JDFX_DOCUMENT_PATH);
1767
+ if (!docFile) throw new Error(`Bundle missing ${JDFX_DOCUMENT_PATH}`);
1768
+ return JSON.parse(await docFile.async("string"));
1769
+ }
1770
+ return JSON.parse(fs.readFileSync(filePath, "utf-8"));
1771
+ }
1772
+ async function chunkFile(inputPath, opts = {}) {
1773
+ const input = path2.resolve(inputPath);
1774
+ if (!fs.existsSync(input)) throw new Error(`File not found: ${input}`);
1775
+ const doc = await loadJdf(input);
1776
+ const strategy = opts.strategy ?? "section";
1777
+ const chunks = chunkDocument(doc, { strategy, maxTokens: opts.maxTokens });
1778
+ const format = opts.format ?? "jsonl";
1779
+ console.log(`Chunking: ${input}`);
1780
+ console.log(`Strategy: ${strategy}${opts.maxTokens ? ` (max ${opts.maxTokens} tokens)` : ""}`);
1781
+ if (format === "inline") {
1782
+ const out = opts.output ? path2.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".jdf");
1783
+ const withIndex = { ...doc, index: { chunker: `jdf-${strategy}-v1`, chunks } };
1784
+ fs.writeFileSync(out, JSON.stringify(withIndex, null, 2));
1785
+ console.log(`Output: ${out} (${chunks.length} chunks in "index" block)`);
1786
+ } else if (format === "json") {
1787
+ const out = opts.output ? path2.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".chunks.json");
1788
+ fs.writeFileSync(out, JSON.stringify(chunks, null, 2));
1789
+ console.log(`Output: ${out} (${chunks.length} chunks)`);
1790
+ } else {
1791
+ const out = opts.output ? path2.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".chunks.jsonl");
1792
+ fs.writeFileSync(out, chunks.map((c) => JSON.stringify(c)).join("\n") + "\n");
1793
+ console.log(`Output: ${out} (${chunks.length} chunks)`);
1794
+ }
1795
+ const totalTokens = chunks.reduce((a, c) => a + c.tokens, 0);
1796
+ console.log(`
1797
+ Done! ${chunks.length} chunks, ~${totalTokens} tokens total.`);
1798
+ return chunks;
1799
+ }
1800
+ var DEFAULT_MODEL = {
1801
+ ollama: "nomic-embed-text",
1802
+ openai: "text-embedding-3-small"
1803
+ };
1804
+ async function embedBatch(provider, model, inputs) {
1805
+ if (inputs.length === 0) return [];
1806
+ switch (provider) {
1807
+ case "ollama":
1808
+ return embedOllama(model, inputs);
1809
+ case "openai":
1810
+ return embedOpenAI(model, inputs);
1811
+ default:
1812
+ throw new Error(`Unknown embedding provider: ${provider}`);
1813
+ }
1814
+ }
1815
+ var OLLAMA_HOST = process.env.OLLAMA_HOST || "http://localhost:11434";
1816
+ var OLLAMA_CONTAINER = "jdf-ollama";
1817
+ async function ollamaUp() {
1818
+ try {
1819
+ const res = await fetch(`${OLLAMA_HOST}/api/tags`, { signal: AbortSignal.timeout(1500) });
1820
+ return res.ok;
1821
+ } catch {
1822
+ return false;
1823
+ }
1824
+ }
1825
+ function dockerReady() {
1826
+ try {
1827
+ execFileSync("docker", ["info"], { stdio: "ignore" });
1828
+ return true;
1829
+ } catch {
1830
+ return false;
1831
+ }
1832
+ }
1833
+ async function ensureOllama(model, autoStart) {
1834
+ if (await ollamaUp()) {
1835
+ await ollamaPull(model);
1836
+ return;
1837
+ }
1838
+ const manualHint = ` Option A \u2014 install Ollama natively (https://ollama.com), then:
1839
+ ollama serve && ollama pull ${model}
1840
+ Option B \u2014 start Docker, then re-run this command (the CLI will launch Ollama for you), or run it yourself:
1841
+ docker run -d --name ${OLLAMA_CONTAINER} -p 11434:11434 -v jdf-ollama:/root/.ollama ollama/ollama
1842
+ docker exec ${OLLAMA_CONTAINER} ollama pull ${model}`;
1843
+ if (!autoStart) {
1844
+ throw new Error(`Ollama isn't running at ${OLLAMA_HOST} and --no-auto-start was given.
1845
+ ${manualHint}`);
1846
+ }
1847
+ if (!dockerReady()) {
1848
+ throw new Error(
1849
+ `Ollama isn't running at ${OLLAMA_HOST}, and the Docker daemon isn't available to auto-start it.
1850
+ ${manualHint}`
1851
+ );
1852
+ }
1853
+ console.log(`Ollama not reachable \u2014 starting it via Docker (container "${OLLAMA_CONTAINER}")\u2026`);
1854
+ try {
1855
+ execFileSync("docker", ["start", OLLAMA_CONTAINER], { stdio: "ignore" });
1856
+ } catch {
1857
+ try {
1858
+ execFileSync("docker", [
1859
+ "run",
1860
+ "-d",
1861
+ "--name",
1862
+ OLLAMA_CONTAINER,
1863
+ "-p",
1864
+ "11434:11434",
1865
+ "-v",
1866
+ "jdf-ollama:/root/.ollama",
1867
+ "ollama/ollama"
1868
+ ], { stdio: "inherit" });
1869
+ } catch (e) {
1870
+ throw new Error(`Failed to start the Ollama container via Docker.
1871
+ ${manualHint}`);
1872
+ }
1873
+ }
1874
+ for (let i = 0; i < 30; i++) {
1875
+ if (await ollamaUp()) {
1876
+ console.log(" Ollama is up.");
1877
+ await ollamaPull(model);
1878
+ return;
1879
+ }
1880
+ await new Promise((r) => setTimeout(r, 1e3));
1881
+ }
1882
+ throw new Error("Ollama container started but never became reachable on :11434.");
1883
+ }
1884
+ async function ollamaPull(model) {
1885
+ try {
1886
+ const show = await fetch(`${OLLAMA_HOST}/api/show`, {
1887
+ method: "POST",
1888
+ headers: { "Content-Type": "application/json" },
1889
+ body: JSON.stringify({ name: model })
1890
+ });
1891
+ if (show.ok) return;
1892
+ } catch {
1893
+ }
1894
+ console.log(`Pulling embedding model "${model}" into Ollama (first run only)\u2026`);
1895
+ const res = await fetch(`${OLLAMA_HOST}/api/pull`, {
1896
+ method: "POST",
1897
+ headers: { "Content-Type": "application/json" },
1898
+ body: JSON.stringify({ name: model, stream: false })
1899
+ });
1900
+ if (!res.ok) throw new Error(`Ollama pull failed (${res.status}) for model "${model}".`);
1901
+ console.log(" Model ready.");
1902
+ }
1903
+ async function embedOllama(model, inputs) {
1904
+ const out = [];
1905
+ for (const text of inputs) {
1906
+ const res = await fetch(`${OLLAMA_HOST}/api/embeddings`, {
1907
+ method: "POST",
1908
+ headers: { "Content-Type": "application/json" },
1909
+ body: JSON.stringify({ model, prompt: text })
1910
+ });
1911
+ if (!res.ok) throw new Error(`Ollama embeddings ${res.status} for model "${model}".`);
1912
+ const json = await res.json();
1913
+ if (!Array.isArray(json.embedding)) throw new Error(`Ollama returned no embedding for model "${model}".`);
1914
+ out.push(json.embedding);
1915
+ }
1916
+ return out;
1917
+ }
1918
+ async function prompt(question, hidden = false) {
1919
+ if (!process.stdin.isTTY) return "";
1920
+ const readline = await import('readline');
1921
+ const rl = readline.createInterface({ input: process.stdin, output: process.stdout });
1922
+ if (hidden) {
1923
+ const out = rl;
1924
+ out._writeToOutput = (s) => {
1925
+ if (s.trim().length) out.output.write("*");
1926
+ else out.output.write(s);
1927
+ };
1928
+ }
1929
+ return new Promise((resolve) => rl.question(question, (a) => {
1930
+ rl.close();
1931
+ if (hidden) process.stdout.write("\n");
1932
+ resolve(a.trim());
1933
+ }));
1934
+ }
1935
+ var openaiCreds = null;
1936
+ async function resolveOpenAICreds() {
1937
+ if (openaiCreds) return openaiCreds;
1938
+ let base = process.env.OPENAI_BASE_URL || "";
1939
+ let key = process.env.OPENAI_API_KEY || "";
1940
+ if (!base) base = await prompt("OpenAI API endpoint [https://api.openai.com/v1]: ") || "https://api.openai.com/v1";
1941
+ if (!key) key = await prompt("OpenAI API key: ", true);
1942
+ if (!key) {
1943
+ throw new Error("No OpenAI API key. Set OPENAI_API_KEY (and optionally OPENAI_BASE_URL) or run interactively so the CLI can prompt.");
1944
+ }
1945
+ openaiCreds = { base: base.replace(/\/$/, ""), key };
1946
+ return openaiCreds;
1947
+ }
1948
+ async function embedOpenAI(model, inputs) {
1949
+ const { base, key } = await resolveOpenAICreds();
1950
+ const res = await fetch(`${base}/embeddings`, {
1951
+ method: "POST",
1952
+ headers: { "Content-Type": "application/json", Authorization: `Bearer ${key}` },
1953
+ body: JSON.stringify({ model, input: inputs })
1954
+ });
1955
+ if (!res.ok) {
1956
+ const body = await res.text().catch(() => "");
1957
+ throw new Error(`OpenAI embeddings API ${res.status}: ${body.slice(0, 200)}`);
1958
+ }
1959
+ const json = await res.json();
1960
+ return json.data.sort((a, b) => a.index - b.index).map((d) => d.embedding);
1961
+ }
1962
+ async function loadJdf2(filePath) {
1963
+ if (filePath.toLowerCase().endsWith(".jdfx")) {
1964
+ const zip = await JSZip.loadAsync(fs.readFileSync(filePath));
1965
+ const docFile = zip.file(JDFX_DOCUMENT_PATH);
1966
+ if (!docFile) throw new Error(`Bundle missing ${JDFX_DOCUMENT_PATH}`);
1967
+ return JSON.parse(await docFile.async("string"));
1968
+ }
1969
+ return JSON.parse(fs.readFileSync(filePath, "utf-8"));
1970
+ }
1971
+ function loadCache(cachePath) {
1972
+ try {
1973
+ if (!fs.existsSync(cachePath)) return null;
1974
+ return JSON.parse(fs.readFileSync(cachePath, "utf-8"));
1975
+ } catch {
1976
+ return null;
1977
+ }
1978
+ }
1979
+ function batched(items, size) {
1980
+ const out = [];
1981
+ for (let i = 0; i < items.length; i += size) out.push(items.slice(i, i + size));
1982
+ return out;
1983
+ }
1984
+ async function embedFile(inputPath, opts = {}) {
1985
+ const input = path2.resolve(inputPath);
1986
+ if (!fs.existsSync(input)) throw new Error(`File not found: ${input}`);
1987
+ const provider = opts.provider ?? "ollama";
1988
+ const model = opts.model ?? DEFAULT_MODEL[provider];
1989
+ const strategy = opts.strategy ?? "section";
1990
+ const doc = await loadJdf2(input);
1991
+ const chunks = chunkDocument(doc, { strategy, maxTokens: opts.maxTokens });
1992
+ const output = opts.output ? path2.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".embeddings.json");
1993
+ const cachePath = opts.cache ? path2.resolve(opts.cache) : output;
1994
+ console.log(`Embedding: ${input}`);
1995
+ console.log(`Provider: ${provider} / ${model}${provider === "ollama" ? " (local \u2014 no data leaves this machine)" : " (remote API)"}`);
1996
+ console.log(`Strategy: ${strategy} \u2192 ${chunks.length} chunks`);
1997
+ if (provider === "ollama") {
1998
+ await ensureOllama(model, opts.autoStart !== false);
1999
+ }
2000
+ const cache = opts.incremental ? loadCache(cachePath) : null;
2001
+ const cachedVectors = cache?.model === model ? cache.vectors : void 0;
2002
+ if (opts.incremental && cache && cache.model !== model) {
2003
+ console.log(` (cache model ${cache.model} \u2260 ${model} \u2014 re-embedding all)`);
2004
+ }
2005
+ const toEmbed = [];
2006
+ const reused = {};
2007
+ for (const c of chunks) {
2008
+ const hit = cachedVectors?.[c.id];
2009
+ if (hit && hit.hash === c.hash) reused[c.id] = hit;
2010
+ else toEmbed.push(c);
2011
+ }
2012
+ if (opts.incremental) {
2013
+ console.log(` Reused ${Object.keys(reused).length} cached, embedding ${toEmbed.length} changed/new.`);
2014
+ }
2015
+ const vectors = { ...reused };
2016
+ let dims = cache?.dims ?? 0;
2017
+ for (const batch of batched(toEmbed, 96)) {
2018
+ const embs = await embedBatch(provider, model, batch.map((c) => c.text));
2019
+ batch.forEach((c, i) => {
2020
+ vectors[c.id] = { hash: c.hash, vector: embs[i] };
2021
+ if (!dims) dims = embs[i].length;
2022
+ });
2023
+ console.log(` Embedded ${Object.keys(vectors).length}/${chunks.length}\u2026`);
2024
+ }
2025
+ const sidecar = {
2026
+ model,
2027
+ provider,
2028
+ dims,
2029
+ chunker: `jdf-${strategy}-v1`,
2030
+ vectors
2031
+ };
2032
+ fs.writeFileSync(output, JSON.stringify(sidecar));
2033
+ console.log(`
2034
+ Done! ${Object.keys(vectors).length} vectors (${dims}-dim) \u2192 ${output}`);
2035
+ return sidecar;
2036
+ }
1446
2037
 
1447
2038
  // src/index.ts
1448
2039
  var HELP = `jdf \u2014 JSON Document Format CLI
1449
2040
 
1450
- The CLI exists for two workflows:
2041
+ The CLI exists for these workflows:
1451
2042
  \u2022 PDF \u2192 JDF legacy documents become a structured JSON tree your
1452
2043
  RAG / agent / pipeline can read natively.
1453
2044
  \u2022 JSON \u2192 JDF LLMs and code emit JSON; this command wraps that JSON
1454
2045
  into a validated .jdf (or .jdfx) you can ship.
2046
+ \u2022 JDF \u2192 chunks turn a document into retrieval-ready chunks (RAG).
2047
+ \u2022 JDF \u2192 vectors embed those chunks, incrementally, for a vector store.
1455
2048
 
1456
2049
  Usage:
1457
2050
  jdf validate <file.jdf>
1458
- jdf import <file.{pdf,json,md}> [-o output.{jdf,jdfx}] [--json]
2051
+ jdf convert <file.{pdf,json,md}> [-o output.{jdf,jdfx}] [--json]
2052
+ jdf chunk <file.{jdf,jdfx}> [--strategy section|element|fixed] [--format jsonl|json|inline] [--max-tokens N] [-o out]
2053
+ jdf embed <file.{jdf,jdfx}> [--provider ollama|openai] [--model NAME] [--strategy \u2026] [--incremental] [-o out]
1459
2054
  jdf --help
1460
2055
 
1461
2056
  Commands:
1462
2057
  validate Validate a .jdf / .jdfx file against the JDF schema
1463
- import Convert a PDF, JSON, or Markdown file into JDF
2058
+ convert Convert a PDF, JSON, or Markdown file into JDF (alias: import)
2059
+ chunk Split a JDF document into retrieval-ready chunks (offline, deterministic)
2060
+ embed Compute embeddings for the chunks (local via Ollama by default)
1464
2061
 
1465
2062
  Flags:
1466
- -o, --output <path> Explicit output path (extension picks .jdf vs .jdfx)
1467
- --json Force pure JSON .jdf output (documents with embedded
1468
- images stay as a single base64-inlined .jdf instead
1469
- of a .jdfx bundle). Useful for RAG / CI consumers
1470
- that prefer one text file over a zip.
2063
+ -o, --output <path> Explicit output path
2064
+ --json convert: force pure JSON .jdf output (inline base64
2065
+ instead of a .jdfx zip bundle)
2066
+ --strategy <s> chunk/embed: section (default) | element | fixed
2067
+ --format <f> chunk: jsonl (default) | json | inline
2068
+ --max-tokens <n> chunk/embed: soft cap per chunk (default 512)
2069
+ --provider <p> embed: ollama (default, local) | openai (remote API)
2070
+ --model <name> embed: model id (default: nomic-embed-text / text-embedding-3-small)
2071
+ --incremental embed: skip chunks whose content hash is unchanged
2072
+ --no-auto-start embed(ollama): don't auto-launch Ollama via Docker
2073
+
2074
+ Environment (embed):
2075
+ ollama: OLLAMA_HOST (default http://localhost:11434)
2076
+ openai: OPENAI_API_KEY (required), OPENAI_BASE_URL (default api.openai.com)
1471
2077
 
1472
2078
  Examples:
1473
2079
  jdf validate spec/examples/hello-world.jdf
1474
- jdf import paper.pdf # PDF \u2192 JDF (or .jdfx for images)
1475
- jdf import contract.pdf --json | jq . # PDF \u2192 pure JSON, pipe-friendly
1476
- jdf import response.json -o response.jdf # LLM JSON output \u2192 validated JDF
1477
- jdf import README.md
2080
+ jdf convert paper.pdf # PDF \u2192 JDF (or .jdfx for images)
2081
+ jdf convert response.json -o response.jdf # LLM JSON output \u2192 validated JDF
2082
+ jdf chunk report.jdf # \u2192 report.chunks.jsonl (RAG-ready)
2083
+ jdf chunk report.jdf --format inline # embed the chunk index into the .jdf
2084
+ jdf embed report.jdf # local embeddings via Ollama (auto-setup)
2085
+ jdf embed report.jdf --provider openai --incremental
1478
2086
  `;
1479
- var BOOLEAN_FLAGS = /* @__PURE__ */ new Set(["help", "h", "json", "verbose", "skip-validate"]);
2087
+ var BOOLEAN_FLAGS = /* @__PURE__ */ new Set(["help", "h", "json", "verbose", "skip-validate", "incremental", "no-auto-start"]);
1480
2088
  function parseArgs(argv) {
1481
2089
  const positional = [];
1482
2090
  const flags = {};
@@ -1540,10 +2148,13 @@ async function main() {
1540
2148
  const ok = await validate(positional[0]);
1541
2149
  process.exit(ok ? 0 : 1);
1542
2150
  }
2151
+ // `convert` is the headline verb; `import` stays as a back-compat alias
2152
+ // so existing scripts and docs keep working.
2153
+ case "convert":
1543
2154
  case "import": {
1544
2155
  const input = positional[0];
1545
2156
  if (!input) {
1546
- console.error("Usage: jdf import <file.{pdf,json,md}> [-o output.jdf] [--json]");
2157
+ console.error("Usage: jdf convert <file.{pdf,json,md}> [-o output.jdf] [--json]");
1547
2158
  process.exit(1);
1548
2159
  }
1549
2160
  const output = typeof flags.output === "string" ? flags.output : void 0;
@@ -1563,6 +2174,37 @@ async function main() {
1563
2174
  process.exit(1);
1564
2175
  }
1565
2176
  }
2177
+ case "chunk": {
2178
+ const input = positional[0];
2179
+ if (!input) {
2180
+ console.error("Usage: jdf chunk <file.{jdf,jdfx}> [--strategy section|element|fixed] [--format jsonl|json|inline] [--max-tokens N] [-o out]");
2181
+ process.exit(1);
2182
+ }
2183
+ await chunkFile(input, {
2184
+ strategy: typeof flags.strategy === "string" ? flags.strategy : void 0,
2185
+ format: typeof flags.format === "string" ? flags.format : void 0,
2186
+ maxTokens: typeof flags["max-tokens"] === "string" ? parseInt(flags["max-tokens"], 10) : void 0,
2187
+ output: typeof flags.output === "string" ? flags.output : void 0
2188
+ });
2189
+ process.exit(0);
2190
+ }
2191
+ case "embed": {
2192
+ const input = positional[0];
2193
+ if (!input) {
2194
+ console.error("Usage: jdf embed <file.{jdf,jdfx}> [--provider ollama|openai] [--model NAME] [--strategy \u2026] [--incremental] [--no-auto-start] [-o out]");
2195
+ process.exit(1);
2196
+ }
2197
+ await embedFile(input, {
2198
+ provider: typeof flags.provider === "string" ? flags.provider : void 0,
2199
+ model: typeof flags.model === "string" ? flags.model : void 0,
2200
+ strategy: typeof flags.strategy === "string" ? flags.strategy : void 0,
2201
+ maxTokens: typeof flags["max-tokens"] === "string" ? parseInt(flags["max-tokens"], 10) : void 0,
2202
+ incremental: flags.incremental === true,
2203
+ autoStart: flags["no-auto-start"] !== true,
2204
+ output: typeof flags.output === "string" ? flags.output : void 0
2205
+ });
2206
+ process.exit(0);
2207
+ }
1566
2208
  default:
1567
2209
  console.error(`Unknown command: ${command}`);
1568
2210
  console.log(HELP);
@@ -26,9 +26,34 @@
26
26
  "type": "array",
27
27
  "minItems": 1,
28
28
  "items": { "$ref": "#/definitions/Page" }
29
- }
29
+ },
30
+ "index": { "$ref": "#/definitions/DocumentIndex" }
30
31
  },
31
32
  "definitions": {
33
+ "DocumentIndex": {
34
+ "type": "object",
35
+ "description": "Optional precomputed RAG chunk index (jdf chunk --format inline). Data-only; renderers ignore it.",
36
+ "required": ["chunker", "chunks"],
37
+ "properties": {
38
+ "chunker": { "type": "string" },
39
+ "chunks": {
40
+ "type": "array",
41
+ "items": {
42
+ "type": "object",
43
+ "required": ["id", "text", "hash"],
44
+ "properties": {
45
+ "id": { "type": "string" },
46
+ "text": { "type": "string" },
47
+ "path": { "type": "array", "items": { "type": "string" } },
48
+ "page": { "type": "integer" },
49
+ "types": { "type": "array", "items": { "type": "string" } },
50
+ "tokens": { "type": "integer" },
51
+ "hash": { "type": "string" }
52
+ }
53
+ }
54
+ }
55
+ }
56
+ },
32
57
  "Meta": {
33
58
  "type": "object",
34
59
  "required": ["title"],
@@ -42,7 +67,8 @@
42
67
  "pageSize": { "$ref": "#/definitions/PageSize" },
43
68
  "pageOrientation": { "$ref": "#/definitions/PageOrientation" },
44
69
  "margins": { "$ref": "#/definitions/Margins" },
45
- "unit": { "$ref": "#/definitions/Unit" }
70
+ "unit": { "$ref": "#/definitions/Unit" },
71
+ "flow": { "type": "boolean", "description": "Document-level default: lay elements out top-to-bottom and auto-paginate overflow in PDF export. Per-page `flow` overrides." }
46
72
  }
47
73
  },
48
74
  "PageSize": {
@@ -335,7 +361,7 @@
335
361
  "headers": { "type": "array", "items": { "type": "string" } },
336
362
  "rows": {
337
363
  "type": "array",
338
- "items": { "type": "array", "items": { "oneOf": [{ "type": "string" }, { "type": "object", "required": ["content"], "properties": { "content": { "type": "string" }, "style": { "$ref": "#/definitions/StyleRef" }, "colspan": { "type": "integer" }, "rowspan": { "type": "integer" } } }] } }
364
+ "items": { "type": "array", "items": { "oneOf": [{ "type": "string" }, { "type": "object", "required": ["content"], "properties": { "content": { "type": "string" }, "style": { "$ref": "#/definitions/StyleRef" }, "align": { "type": "string", "enum": ["left","center","right","justify"] }, "colspan": { "type": "integer" }, "rowspan": { "type": "integer" } } }] } }
339
365
  },
340
366
  "position": { "$ref": "#/definitions/Position" }, "width": { "type": "number" },
341
367
  "headerStyle": { "$ref": "#/definitions/StyleRef" },
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@uurtech/jdf-cli",
3
- "version": "0.1.21",
3
+ "version": "0.1.23",
4
4
  "description": "Command-line tool for the JDF (JSON Document Format) — validate and convert documents.",
5
5
  "license": "MIT",
6
6
  "author": "Ugur Kazdal",