@uurtech/jdf-cli 0.1.22 → 0.1.24

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -5,8 +5,9 @@ import { fileURLToPath } from 'url';
5
5
  import Ajv from 'ajv';
6
6
  import addFormats from 'ajv-formats';
7
7
  import JSZip from 'jszip';
8
- import { createHash } from 'crypto';
8
+ import crypto, { createHash } from 'crypto';
9
9
  import { readFile } from 'fs/promises';
10
+ import { execFileSync } from 'child_process';
10
11
 
11
12
  // ../../packages/jdf-core/src/manifest.ts
12
13
  var JDFX_MANIFEST_VERSION = "1.0.0";
@@ -802,7 +803,7 @@ async function walkOps(page, OPS, viewport) {
802
803
  const s = stack.pop();
803
804
  if (s) Object.assign(gs, s);
804
805
  } else if (fn === OPS.transform) {
805
- gs.ctm = multiplyCtm(gs.ctm, args);
806
+ gs.ctm = multiplyCtm(args, gs.ctm);
806
807
  } else if (fn === OPS.setFillRGBColor) {
807
808
  gs.fill = rgbToHex(args[0], args[1], args[2]);
808
809
  } else if (fn === OPS.setStrokeRGBColor) {
@@ -1591,41 +1592,499 @@ function wrapElements(elements, title, meta) {
1591
1592
  ]
1592
1593
  };
1593
1594
  }
1595
+ var DEFAULT_MAX_TOKENS = 512;
1596
+ function estimateTokens(text) {
1597
+ return Math.ceil(text.length / 4);
1598
+ }
1599
+ function hashText(text) {
1600
+ return crypto.createHash("sha256").update(text).digest("hex").slice(0, 12);
1601
+ }
1602
+ function headingLevel(el) {
1603
+ const h = el.heading;
1604
+ if (h === true) return 1;
1605
+ if (typeof h === "number" && h >= 1 && h <= 6) return h;
1606
+ if (typeof el.tocLevel === "number") return el.tocLevel;
1607
+ return null;
1608
+ }
1609
+ function serializeElement(el) {
1610
+ const e = el;
1611
+ switch (e.type) {
1612
+ case "text":
1613
+ return String(e.content ?? "").trim();
1614
+ case "richtext":
1615
+ return (e.runs || []).map((r) => r.text ?? "").join("").trim();
1616
+ case "list": {
1617
+ const walk = (items, depth = 0) => (items || []).flatMap((it) => {
1618
+ const line = " ".repeat(depth) + "- " + String(it.content ?? "").trim();
1619
+ return it.children?.length ? [line, ...walk(it.children, depth + 1)] : [line];
1620
+ });
1621
+ return walk(e.items).join("\n");
1622
+ }
1623
+ case "table": {
1624
+ const headers = e.headers ?? e.columns?.map((c) => c.header || "").filter((h) => h) ?? [];
1625
+ const cell = (c) => typeof c === "string" ? c : c?.content ?? "";
1626
+ const rows = e.rows || [];
1627
+ const lines = rows.map(
1628
+ (row) => row.map((c, i) => {
1629
+ const h = headers[i];
1630
+ const v = cell(c).trim();
1631
+ return h ? `${h}: ${v}` : v;
1632
+ }).filter((s) => s !== "").join(" | ")
1633
+ );
1634
+ return lines.join("\n");
1635
+ }
1636
+ case "collapsible": {
1637
+ const title = String(e.title ?? "").trim();
1638
+ const inner = (e.elements || []).map(serializeElement).filter(Boolean).join("\n");
1639
+ return [title, inner].filter(Boolean).join("\n");
1640
+ }
1641
+ case "input":
1642
+ case "textarea":
1643
+ case "select": {
1644
+ const label = e.label ? `${e.label}: ` : "";
1645
+ const val = e.multiple ? (e.values || []).join(", ") : e.value ?? "";
1646
+ return `${label}${val}`.trim();
1647
+ }
1648
+ case "checkbox":
1649
+ return `${e.checked ? "[x]" : "[ ]"} ${e.label ?? ""}`.trim();
1650
+ case "image":
1651
+ return e.alt ? `[image: ${e.alt}]` : "";
1652
+ case "toc":
1653
+ case "shape":
1654
+ case "signature":
1655
+ return "";
1656
+ default:
1657
+ return "";
1658
+ }
1659
+ }
1660
+ function elementId(el, pageIdx, elIdx) {
1661
+ if (typeof el.id === "string" && el.id.length > 0) return el.id;
1662
+ return `p${pageIdx + 1}e${elIdx}`;
1663
+ }
1664
+ function flatten(doc) {
1665
+ const out = [];
1666
+ (doc.pages || []).forEach((page, pi) => {
1667
+ (page.elements || []).forEach((el, ei) => {
1668
+ out.push({ el, page: pi + 1, id: elementId(el, pi, ei) });
1669
+ });
1670
+ });
1671
+ return out;
1672
+ }
1673
+ function makeChunk(group, breadcrumb) {
1674
+ const parts = group.map((g) => serializeElement(g.el)).filter((s) => s.length > 0);
1675
+ const text = parts.join("\n\n").trim();
1676
+ if (!text) return null;
1677
+ const types = Array.from(new Set(group.map((g) => g.el.type)));
1678
+ return {
1679
+ id: group[0].id,
1680
+ text,
1681
+ path: [...breadcrumb],
1682
+ page: group[0].page,
1683
+ types,
1684
+ tokens: estimateTokens(text),
1685
+ hash: hashText(text)
1686
+ };
1687
+ }
1688
+ function chunkDocument(doc, options = {}) {
1689
+ const strategy = options.strategy ?? "section";
1690
+ const maxTokens = options.maxTokens ?? DEFAULT_MAX_TOKENS;
1691
+ const flat = flatten(doc);
1692
+ const chunks = [];
1693
+ if (strategy === "element") {
1694
+ const crumb2 = [];
1695
+ for (const f of flat) {
1696
+ const lvl = headingLevel(f.el);
1697
+ if (lvl != null) {
1698
+ crumb2.length = Math.max(0, lvl - 1);
1699
+ crumb2[lvl - 1] = serializeElement(f.el);
1700
+ }
1701
+ const c = makeChunk([f], crumb2);
1702
+ if (c) chunks.push(c);
1703
+ }
1704
+ return chunks;
1705
+ }
1706
+ if (strategy === "fixed") {
1707
+ const crumb2 = [];
1708
+ let buf2 = [];
1709
+ let bufTokens = 0;
1710
+ const flush = () => {
1711
+ const c = makeChunk(buf2, crumb2);
1712
+ if (c) chunks.push(c);
1713
+ buf2 = [];
1714
+ bufTokens = 0;
1715
+ };
1716
+ for (const f of flat) {
1717
+ const lvl = headingLevel(f.el);
1718
+ if (lvl != null) {
1719
+ crumb2.length = Math.max(0, lvl - 1);
1720
+ crumb2[lvl - 1] = serializeElement(f.el);
1721
+ }
1722
+ const t = estimateTokens(serializeElement(f.el));
1723
+ if (bufTokens + t > maxTokens && buf2.length > 0) flush();
1724
+ buf2.push(f);
1725
+ bufTokens += t;
1726
+ }
1727
+ flush();
1728
+ return chunks;
1729
+ }
1730
+ const crumb = [];
1731
+ let buf = [];
1732
+ const flushSection = () => {
1733
+ if (buf.length === 0) return;
1734
+ let sub = [];
1735
+ let subTokens = 0;
1736
+ for (const f of buf) {
1737
+ const t = estimateTokens(serializeElement(f.el));
1738
+ if (subTokens + t > maxTokens && sub.length > 0) {
1739
+ const c2 = makeChunk(sub, crumb);
1740
+ if (c2) chunks.push(c2);
1741
+ sub = [];
1742
+ subTokens = 0;
1743
+ }
1744
+ sub.push(f);
1745
+ subTokens += t;
1746
+ }
1747
+ const c = makeChunk(sub, crumb);
1748
+ if (c) chunks.push(c);
1749
+ buf = [];
1750
+ };
1751
+ for (const f of flat) {
1752
+ const lvl = headingLevel(f.el);
1753
+ if (lvl != null) {
1754
+ flushSection();
1755
+ crumb.length = Math.max(0, lvl - 1);
1756
+ crumb[lvl - 1] = serializeElement(f.el);
1757
+ }
1758
+ buf.push(f);
1759
+ }
1760
+ flushSection();
1761
+ return chunks;
1762
+ }
1763
+ async function loadJdf(filePath) {
1764
+ if (filePath.toLowerCase().endsWith(".jdfx")) {
1765
+ const zip = await JSZip.loadAsync(fs.readFileSync(filePath));
1766
+ const docFile = zip.file(JDFX_DOCUMENT_PATH);
1767
+ if (!docFile) throw new Error(`Bundle missing ${JDFX_DOCUMENT_PATH}`);
1768
+ return JSON.parse(await docFile.async("string"));
1769
+ }
1770
+ return JSON.parse(fs.readFileSync(filePath, "utf-8"));
1771
+ }
1772
+ async function chunkFile(inputPath, opts = {}) {
1773
+ const input = path2.resolve(inputPath);
1774
+ if (!fs.existsSync(input)) throw new Error(`File not found: ${input}`);
1775
+ const doc = await loadJdf(input);
1776
+ const strategy = opts.strategy ?? "section";
1777
+ const chunks = chunkDocument(doc, { strategy, maxTokens: opts.maxTokens });
1778
+ const format = opts.format ?? "jsonl";
1779
+ console.log(`Chunking: ${input}`);
1780
+ console.log(`Strategy: ${strategy}${opts.maxTokens ? ` (max ${opts.maxTokens} tokens)` : ""}`);
1781
+ if (format === "inline") {
1782
+ const out = opts.output ? path2.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".jdf");
1783
+ const withIndex = { ...doc, index: { chunker: `jdf-${strategy}-v1`, chunks } };
1784
+ fs.writeFileSync(out, JSON.stringify(withIndex, null, 2));
1785
+ console.log(`Output: ${out} (${chunks.length} chunks in "index" block)`);
1786
+ } else if (format === "json") {
1787
+ const out = opts.output ? path2.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".chunks.json");
1788
+ fs.writeFileSync(out, JSON.stringify(chunks, null, 2));
1789
+ console.log(`Output: ${out} (${chunks.length} chunks)`);
1790
+ } else {
1791
+ const out = opts.output ? path2.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".chunks.jsonl");
1792
+ fs.writeFileSync(out, chunks.map((c) => JSON.stringify(c)).join("\n") + "\n");
1793
+ console.log(`Output: ${out} (${chunks.length} chunks)`);
1794
+ }
1795
+ const totalTokens = chunks.reduce((a, c) => a + c.tokens, 0);
1796
+ console.log(`
1797
+ Done! ${chunks.length} chunks, ~${totalTokens} tokens total.`);
1798
+ return chunks;
1799
+ }
1800
+ var DEFAULT_MODEL = {
1801
+ ollama: "nomic-embed-text",
1802
+ openai: "text-embedding-3-small"
1803
+ };
1804
+ async function embedBatch(provider, model, inputs) {
1805
+ if (inputs.length === 0) return [];
1806
+ switch (provider) {
1807
+ case "ollama":
1808
+ return embedOllama(model, inputs);
1809
+ case "openai":
1810
+ return embedOpenAI(model, inputs);
1811
+ default:
1812
+ throw new Error(`Unknown embedding provider: ${provider}`);
1813
+ }
1814
+ }
1815
+ var OLLAMA_HOST = process.env.OLLAMA_HOST || "http://localhost:11434";
1816
+ var OLLAMA_CONTAINER = "jdf-ollama";
1817
+ async function ollamaUp() {
1818
+ try {
1819
+ const res = await fetch(`${OLLAMA_HOST}/api/tags`, { signal: AbortSignal.timeout(1500) });
1820
+ return res.ok;
1821
+ } catch {
1822
+ return false;
1823
+ }
1824
+ }
1825
+ function dockerReady() {
1826
+ try {
1827
+ execFileSync("docker", ["info"], { stdio: "ignore" });
1828
+ return true;
1829
+ } catch {
1830
+ return false;
1831
+ }
1832
+ }
1833
+ async function ensureOllama(model, autoStart) {
1834
+ if (await ollamaUp()) {
1835
+ await ollamaPull(model);
1836
+ return;
1837
+ }
1838
+ const manualHint = ` Option A \u2014 install Ollama natively (https://ollama.com), then:
1839
+ ollama serve && ollama pull ${model}
1840
+ Option B \u2014 start Docker, then re-run this command (the CLI will launch Ollama for you), or run it yourself:
1841
+ docker run -d --name ${OLLAMA_CONTAINER} -p 11434:11434 -v jdf-ollama:/root/.ollama ollama/ollama
1842
+ docker exec ${OLLAMA_CONTAINER} ollama pull ${model}`;
1843
+ if (!autoStart) {
1844
+ throw new Error(`Ollama isn't running at ${OLLAMA_HOST} and --no-auto-start was given.
1845
+ ${manualHint}`);
1846
+ }
1847
+ if (!dockerReady()) {
1848
+ throw new Error(
1849
+ `Ollama isn't running at ${OLLAMA_HOST}, and the Docker daemon isn't available to auto-start it.
1850
+ ${manualHint}`
1851
+ );
1852
+ }
1853
+ console.log(`Ollama not reachable \u2014 starting it via Docker (container "${OLLAMA_CONTAINER}")\u2026`);
1854
+ try {
1855
+ execFileSync("docker", ["start", OLLAMA_CONTAINER], { stdio: "ignore" });
1856
+ } catch {
1857
+ try {
1858
+ execFileSync("docker", [
1859
+ "run",
1860
+ "-d",
1861
+ "--name",
1862
+ OLLAMA_CONTAINER,
1863
+ "-p",
1864
+ "11434:11434",
1865
+ "-v",
1866
+ "jdf-ollama:/root/.ollama",
1867
+ "ollama/ollama"
1868
+ ], { stdio: "inherit" });
1869
+ } catch (e) {
1870
+ throw new Error(`Failed to start the Ollama container via Docker.
1871
+ ${manualHint}`);
1872
+ }
1873
+ }
1874
+ for (let i = 0; i < 30; i++) {
1875
+ if (await ollamaUp()) {
1876
+ console.log(" Ollama is up.");
1877
+ await ollamaPull(model);
1878
+ return;
1879
+ }
1880
+ await new Promise((r) => setTimeout(r, 1e3));
1881
+ }
1882
+ throw new Error("Ollama container started but never became reachable on :11434.");
1883
+ }
1884
+ async function ollamaPull(model) {
1885
+ try {
1886
+ const show = await fetch(`${OLLAMA_HOST}/api/show`, {
1887
+ method: "POST",
1888
+ headers: { "Content-Type": "application/json" },
1889
+ body: JSON.stringify({ name: model })
1890
+ });
1891
+ if (show.ok) return;
1892
+ } catch {
1893
+ }
1894
+ console.log(`Pulling embedding model "${model}" into Ollama (first run only)\u2026`);
1895
+ const res = await fetch(`${OLLAMA_HOST}/api/pull`, {
1896
+ method: "POST",
1897
+ headers: { "Content-Type": "application/json" },
1898
+ body: JSON.stringify({ name: model, stream: false })
1899
+ });
1900
+ if (!res.ok) throw new Error(`Ollama pull failed (${res.status}) for model "${model}".`);
1901
+ console.log(" Model ready.");
1902
+ }
1903
+ async function embedOllama(model, inputs) {
1904
+ const out = [];
1905
+ for (const text of inputs) {
1906
+ const res = await fetch(`${OLLAMA_HOST}/api/embeddings`, {
1907
+ method: "POST",
1908
+ headers: { "Content-Type": "application/json" },
1909
+ body: JSON.stringify({ model, prompt: text })
1910
+ });
1911
+ if (!res.ok) throw new Error(`Ollama embeddings ${res.status} for model "${model}".`);
1912
+ const json = await res.json();
1913
+ if (!Array.isArray(json.embedding)) throw new Error(`Ollama returned no embedding for model "${model}".`);
1914
+ out.push(json.embedding);
1915
+ }
1916
+ return out;
1917
+ }
1918
+ async function prompt(question, hidden = false) {
1919
+ if (!process.stdin.isTTY) return "";
1920
+ const readline = await import('readline');
1921
+ const rl = readline.createInterface({ input: process.stdin, output: process.stdout });
1922
+ if (hidden) {
1923
+ const out = rl;
1924
+ out._writeToOutput = (s) => {
1925
+ if (s.trim().length) out.output.write("*");
1926
+ else out.output.write(s);
1927
+ };
1928
+ }
1929
+ return new Promise((resolve) => rl.question(question, (a) => {
1930
+ rl.close();
1931
+ if (hidden) process.stdout.write("\n");
1932
+ resolve(a.trim());
1933
+ }));
1934
+ }
1935
+ var openaiCreds = null;
1936
+ async function resolveOpenAICreds() {
1937
+ if (openaiCreds) return openaiCreds;
1938
+ let base = process.env.OPENAI_BASE_URL || "";
1939
+ let key = process.env.OPENAI_API_KEY || "";
1940
+ if (!base) base = await prompt("OpenAI API endpoint [https://api.openai.com/v1]: ") || "https://api.openai.com/v1";
1941
+ if (!key) key = await prompt("OpenAI API key: ", true);
1942
+ if (!key) {
1943
+ throw new Error("No OpenAI API key. Set OPENAI_API_KEY (and optionally OPENAI_BASE_URL) or run interactively so the CLI can prompt.");
1944
+ }
1945
+ openaiCreds = { base: base.replace(/\/$/, ""), key };
1946
+ return openaiCreds;
1947
+ }
1948
+ async function embedOpenAI(model, inputs) {
1949
+ const { base, key } = await resolveOpenAICreds();
1950
+ const res = await fetch(`${base}/embeddings`, {
1951
+ method: "POST",
1952
+ headers: { "Content-Type": "application/json", Authorization: `Bearer ${key}` },
1953
+ body: JSON.stringify({ model, input: inputs })
1954
+ });
1955
+ if (!res.ok) {
1956
+ const body = await res.text().catch(() => "");
1957
+ throw new Error(`OpenAI embeddings API ${res.status}: ${body.slice(0, 200)}`);
1958
+ }
1959
+ const json = await res.json();
1960
+ return json.data.sort((a, b) => a.index - b.index).map((d) => d.embedding);
1961
+ }
1962
+ async function loadJdf2(filePath) {
1963
+ if (filePath.toLowerCase().endsWith(".jdfx")) {
1964
+ const zip = await JSZip.loadAsync(fs.readFileSync(filePath));
1965
+ const docFile = zip.file(JDFX_DOCUMENT_PATH);
1966
+ if (!docFile) throw new Error(`Bundle missing ${JDFX_DOCUMENT_PATH}`);
1967
+ return JSON.parse(await docFile.async("string"));
1968
+ }
1969
+ return JSON.parse(fs.readFileSync(filePath, "utf-8"));
1970
+ }
1971
+ function loadCache(cachePath) {
1972
+ try {
1973
+ if (!fs.existsSync(cachePath)) return null;
1974
+ return JSON.parse(fs.readFileSync(cachePath, "utf-8"));
1975
+ } catch {
1976
+ return null;
1977
+ }
1978
+ }
1979
+ function batched(items, size) {
1980
+ const out = [];
1981
+ for (let i = 0; i < items.length; i += size) out.push(items.slice(i, i + size));
1982
+ return out;
1983
+ }
1984
+ async function embedFile(inputPath, opts = {}) {
1985
+ const input = path2.resolve(inputPath);
1986
+ if (!fs.existsSync(input)) throw new Error(`File not found: ${input}`);
1987
+ const provider = opts.provider ?? "ollama";
1988
+ const model = opts.model ?? DEFAULT_MODEL[provider];
1989
+ const strategy = opts.strategy ?? "section";
1990
+ const doc = await loadJdf2(input);
1991
+ const chunks = chunkDocument(doc, { strategy, maxTokens: opts.maxTokens });
1992
+ const output = opts.output ? path2.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".embeddings.json");
1993
+ const cachePath = opts.cache ? path2.resolve(opts.cache) : output;
1994
+ console.log(`Embedding: ${input}`);
1995
+ console.log(`Provider: ${provider} / ${model}${provider === "ollama" ? " (local \u2014 no data leaves this machine)" : " (remote API)"}`);
1996
+ console.log(`Strategy: ${strategy} \u2192 ${chunks.length} chunks`);
1997
+ if (provider === "ollama") {
1998
+ await ensureOllama(model, opts.autoStart !== false);
1999
+ }
2000
+ const cache = opts.incremental ? loadCache(cachePath) : null;
2001
+ const cachedVectors = cache?.model === model ? cache.vectors : void 0;
2002
+ if (opts.incremental && cache && cache.model !== model) {
2003
+ console.log(` (cache model ${cache.model} \u2260 ${model} \u2014 re-embedding all)`);
2004
+ }
2005
+ const toEmbed = [];
2006
+ const reused = {};
2007
+ for (const c of chunks) {
2008
+ const hit = cachedVectors?.[c.id];
2009
+ if (hit && hit.hash === c.hash) reused[c.id] = hit;
2010
+ else toEmbed.push(c);
2011
+ }
2012
+ if (opts.incremental) {
2013
+ console.log(` Reused ${Object.keys(reused).length} cached, embedding ${toEmbed.length} changed/new.`);
2014
+ }
2015
+ const vectors = { ...reused };
2016
+ let dims = cache?.dims ?? 0;
2017
+ for (const batch of batched(toEmbed, 96)) {
2018
+ const embs = await embedBatch(provider, model, batch.map((c) => c.text));
2019
+ batch.forEach((c, i) => {
2020
+ vectors[c.id] = { hash: c.hash, vector: embs[i] };
2021
+ if (!dims) dims = embs[i].length;
2022
+ });
2023
+ console.log(` Embedded ${Object.keys(vectors).length}/${chunks.length}\u2026`);
2024
+ }
2025
+ const sidecar = {
2026
+ model,
2027
+ provider,
2028
+ dims,
2029
+ chunker: `jdf-${strategy}-v1`,
2030
+ vectors
2031
+ };
2032
+ fs.writeFileSync(output, JSON.stringify(sidecar));
2033
+ console.log(`
2034
+ Done! ${Object.keys(vectors).length} vectors (${dims}-dim) \u2192 ${output}`);
2035
+ return sidecar;
2036
+ }
1594
2037
 
1595
2038
  // src/index.ts
1596
2039
  var HELP = `jdf \u2014 JSON Document Format CLI
1597
2040
 
1598
- The CLI exists for two workflows:
2041
+ The CLI exists for these workflows:
1599
2042
  \u2022 PDF \u2192 JDF legacy documents become a structured JSON tree your
1600
2043
  RAG / agent / pipeline can read natively.
1601
2044
  \u2022 JSON \u2192 JDF LLMs and code emit JSON; this command wraps that JSON
1602
2045
  into a validated .jdf (or .jdfx) you can ship.
2046
+ \u2022 JDF \u2192 chunks turn a document into retrieval-ready chunks (RAG).
2047
+ \u2022 JDF \u2192 vectors embed those chunks, incrementally, for a vector store.
1603
2048
 
1604
2049
  Usage:
1605
2050
  jdf validate <file.jdf>
1606
- jdf convert <file.{pdf,json,md}> [-o output.{jdf,jdfx}] [--json]
2051
+ jdf convert <file.{pdf,json,md}> [-o output.{jdf,jdfx}] [--json]
2052
+ jdf chunk <file.{jdf,jdfx}> [--strategy section|element|fixed] [--format jsonl|json|inline] [--max-tokens N] [-o out]
2053
+ jdf embed <file.{jdf,jdfx}> [--provider ollama|openai] [--model NAME] [--strategy \u2026] [--incremental] [-o out]
1607
2054
  jdf --help
1608
2055
 
1609
2056
  Commands:
1610
2057
  validate Validate a .jdf / .jdfx file against the JDF schema
1611
- convert Convert a PDF, JSON, or Markdown file into JDF
1612
- (alias: import)
2058
+ convert Convert a PDF, JSON, or Markdown file into JDF (alias: import)
2059
+ chunk Split a JDF document into retrieval-ready chunks (offline, deterministic)
2060
+ embed Compute embeddings for the chunks (local via Ollama by default)
1613
2061
 
1614
2062
  Flags:
1615
- -o, --output <path> Explicit output path (extension picks .jdf vs .jdfx)
1616
- --json Force pure JSON .jdf output (documents with embedded
1617
- images stay as a single base64-inlined .jdf instead
1618
- of a .jdfx bundle). Useful for RAG / CI consumers
1619
- that prefer one text file over a zip.
2063
+ -o, --output <path> Explicit output path
2064
+ --json convert: force pure JSON .jdf output (inline base64
2065
+ instead of a .jdfx zip bundle)
2066
+ --strategy <s> chunk/embed: section (default) | element | fixed
2067
+ --format <f> chunk: jsonl (default) | json | inline
2068
+ --max-tokens <n> chunk/embed: soft cap per chunk (default 512)
2069
+ --provider <p> embed: ollama (default, local) | openai (remote API)
2070
+ --model <name> embed: model id (default: nomic-embed-text / text-embedding-3-small)
2071
+ --incremental embed: skip chunks whose content hash is unchanged
2072
+ --no-auto-start embed(ollama): don't auto-launch Ollama via Docker
2073
+
2074
+ Environment (embed):
2075
+ ollama: OLLAMA_HOST (default http://localhost:11434)
2076
+ openai: OPENAI_API_KEY (required), OPENAI_BASE_URL (default api.openai.com)
1620
2077
 
1621
2078
  Examples:
1622
2079
  jdf validate spec/examples/hello-world.jdf
1623
2080
  jdf convert paper.pdf # PDF \u2192 JDF (or .jdfx for images)
1624
- jdf convert contract.pdf --json | jq . # PDF \u2192 pure JSON, pipe-friendly
1625
2081
  jdf convert response.json -o response.jdf # LLM JSON output \u2192 validated JDF
1626
- jdf convert README.md
2082
+ jdf chunk report.jdf # \u2192 report.chunks.jsonl (RAG-ready)
2083
+ jdf chunk report.jdf --format inline # embed the chunk index into the .jdf
2084
+ jdf embed report.jdf # local embeddings via Ollama (auto-setup)
2085
+ jdf embed report.jdf --provider openai --incremental
1627
2086
  `;
1628
- var BOOLEAN_FLAGS = /* @__PURE__ */ new Set(["help", "h", "json", "verbose", "skip-validate"]);
2087
+ var BOOLEAN_FLAGS = /* @__PURE__ */ new Set(["help", "h", "json", "verbose", "skip-validate", "incremental", "no-auto-start"]);
1629
2088
  function parseArgs(argv) {
1630
2089
  const positional = [];
1631
2090
  const flags = {};
@@ -1715,6 +2174,37 @@ async function main() {
1715
2174
  process.exit(1);
1716
2175
  }
1717
2176
  }
2177
+ case "chunk": {
2178
+ const input = positional[0];
2179
+ if (!input) {
2180
+ console.error("Usage: jdf chunk <file.{jdf,jdfx}> [--strategy section|element|fixed] [--format jsonl|json|inline] [--max-tokens N] [-o out]");
2181
+ process.exit(1);
2182
+ }
2183
+ await chunkFile(input, {
2184
+ strategy: typeof flags.strategy === "string" ? flags.strategy : void 0,
2185
+ format: typeof flags.format === "string" ? flags.format : void 0,
2186
+ maxTokens: typeof flags["max-tokens"] === "string" ? parseInt(flags["max-tokens"], 10) : void 0,
2187
+ output: typeof flags.output === "string" ? flags.output : void 0
2188
+ });
2189
+ process.exit(0);
2190
+ }
2191
+ case "embed": {
2192
+ const input = positional[0];
2193
+ if (!input) {
2194
+ console.error("Usage: jdf embed <file.{jdf,jdfx}> [--provider ollama|openai] [--model NAME] [--strategy \u2026] [--incremental] [--no-auto-start] [-o out]");
2195
+ process.exit(1);
2196
+ }
2197
+ await embedFile(input, {
2198
+ provider: typeof flags.provider === "string" ? flags.provider : void 0,
2199
+ model: typeof flags.model === "string" ? flags.model : void 0,
2200
+ strategy: typeof flags.strategy === "string" ? flags.strategy : void 0,
2201
+ maxTokens: typeof flags["max-tokens"] === "string" ? parseInt(flags["max-tokens"], 10) : void 0,
2202
+ incremental: flags.incremental === true,
2203
+ autoStart: flags["no-auto-start"] !== true,
2204
+ output: typeof flags.output === "string" ? flags.output : void 0
2205
+ });
2206
+ process.exit(0);
2207
+ }
1718
2208
  default:
1719
2209
  console.error(`Unknown command: ${command}`);
1720
2210
  console.log(HELP);
@@ -26,9 +26,34 @@
26
26
  "type": "array",
27
27
  "minItems": 1,
28
28
  "items": { "$ref": "#/definitions/Page" }
29
- }
29
+ },
30
+ "index": { "$ref": "#/definitions/DocumentIndex" }
30
31
  },
31
32
  "definitions": {
33
+ "DocumentIndex": {
34
+ "type": "object",
35
+ "description": "Optional precomputed RAG chunk index (jdf chunk --format inline). Data-only; renderers ignore it.",
36
+ "required": ["chunker", "chunks"],
37
+ "properties": {
38
+ "chunker": { "type": "string" },
39
+ "chunks": {
40
+ "type": "array",
41
+ "items": {
42
+ "type": "object",
43
+ "required": ["id", "text", "hash"],
44
+ "properties": {
45
+ "id": { "type": "string" },
46
+ "text": { "type": "string" },
47
+ "path": { "type": "array", "items": { "type": "string" } },
48
+ "page": { "type": "integer" },
49
+ "types": { "type": "array", "items": { "type": "string" } },
50
+ "tokens": { "type": "integer" },
51
+ "hash": { "type": "string" }
52
+ }
53
+ }
54
+ }
55
+ }
56
+ },
32
57
  "Meta": {
33
58
  "type": "object",
34
59
  "required": ["title"],
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@uurtech/jdf-cli",
3
- "version": "0.1.22",
3
+ "version": "0.1.24",
4
4
  "description": "Command-line tool for the JDF (JSON Document Format) — validate and convert documents.",
5
5
  "license": "MIT",
6
6
  "author": "Ugur Kazdal",