@uurtech/jdf-cli 0.1.22 → 0.1.24
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.js +504 -14
- package/dist/jdf-schema.json +26 -1
- package/package.json +1 -1
package/dist/index.js
CHANGED
|
@@ -5,8 +5,9 @@ import { fileURLToPath } from 'url';
|
|
|
5
5
|
import Ajv from 'ajv';
|
|
6
6
|
import addFormats from 'ajv-formats';
|
|
7
7
|
import JSZip from 'jszip';
|
|
8
|
-
import { createHash } from 'crypto';
|
|
8
|
+
import crypto, { createHash } from 'crypto';
|
|
9
9
|
import { readFile } from 'fs/promises';
|
|
10
|
+
import { execFileSync } from 'child_process';
|
|
10
11
|
|
|
11
12
|
// ../../packages/jdf-core/src/manifest.ts
|
|
12
13
|
var JDFX_MANIFEST_VERSION = "1.0.0";
|
|
@@ -802,7 +803,7 @@ async function walkOps(page, OPS, viewport) {
|
|
|
802
803
|
const s = stack.pop();
|
|
803
804
|
if (s) Object.assign(gs, s);
|
|
804
805
|
} else if (fn === OPS.transform) {
|
|
805
|
-
gs.ctm = multiplyCtm(gs.ctm
|
|
806
|
+
gs.ctm = multiplyCtm(args, gs.ctm);
|
|
806
807
|
} else if (fn === OPS.setFillRGBColor) {
|
|
807
808
|
gs.fill = rgbToHex(args[0], args[1], args[2]);
|
|
808
809
|
} else if (fn === OPS.setStrokeRGBColor) {
|
|
@@ -1591,41 +1592,499 @@ function wrapElements(elements, title, meta) {
|
|
|
1591
1592
|
]
|
|
1592
1593
|
};
|
|
1593
1594
|
}
|
|
1595
|
+
var DEFAULT_MAX_TOKENS = 512;
|
|
1596
|
+
function estimateTokens(text) {
|
|
1597
|
+
return Math.ceil(text.length / 4);
|
|
1598
|
+
}
|
|
1599
|
+
function hashText(text) {
|
|
1600
|
+
return crypto.createHash("sha256").update(text).digest("hex").slice(0, 12);
|
|
1601
|
+
}
|
|
1602
|
+
function headingLevel(el) {
|
|
1603
|
+
const h = el.heading;
|
|
1604
|
+
if (h === true) return 1;
|
|
1605
|
+
if (typeof h === "number" && h >= 1 && h <= 6) return h;
|
|
1606
|
+
if (typeof el.tocLevel === "number") return el.tocLevel;
|
|
1607
|
+
return null;
|
|
1608
|
+
}
|
|
1609
|
+
function serializeElement(el) {
|
|
1610
|
+
const e = el;
|
|
1611
|
+
switch (e.type) {
|
|
1612
|
+
case "text":
|
|
1613
|
+
return String(e.content ?? "").trim();
|
|
1614
|
+
case "richtext":
|
|
1615
|
+
return (e.runs || []).map((r) => r.text ?? "").join("").trim();
|
|
1616
|
+
case "list": {
|
|
1617
|
+
const walk = (items, depth = 0) => (items || []).flatMap((it) => {
|
|
1618
|
+
const line = " ".repeat(depth) + "- " + String(it.content ?? "").trim();
|
|
1619
|
+
return it.children?.length ? [line, ...walk(it.children, depth + 1)] : [line];
|
|
1620
|
+
});
|
|
1621
|
+
return walk(e.items).join("\n");
|
|
1622
|
+
}
|
|
1623
|
+
case "table": {
|
|
1624
|
+
const headers = e.headers ?? e.columns?.map((c) => c.header || "").filter((h) => h) ?? [];
|
|
1625
|
+
const cell = (c) => typeof c === "string" ? c : c?.content ?? "";
|
|
1626
|
+
const rows = e.rows || [];
|
|
1627
|
+
const lines = rows.map(
|
|
1628
|
+
(row) => row.map((c, i) => {
|
|
1629
|
+
const h = headers[i];
|
|
1630
|
+
const v = cell(c).trim();
|
|
1631
|
+
return h ? `${h}: ${v}` : v;
|
|
1632
|
+
}).filter((s) => s !== "").join(" | ")
|
|
1633
|
+
);
|
|
1634
|
+
return lines.join("\n");
|
|
1635
|
+
}
|
|
1636
|
+
case "collapsible": {
|
|
1637
|
+
const title = String(e.title ?? "").trim();
|
|
1638
|
+
const inner = (e.elements || []).map(serializeElement).filter(Boolean).join("\n");
|
|
1639
|
+
return [title, inner].filter(Boolean).join("\n");
|
|
1640
|
+
}
|
|
1641
|
+
case "input":
|
|
1642
|
+
case "textarea":
|
|
1643
|
+
case "select": {
|
|
1644
|
+
const label = e.label ? `${e.label}: ` : "";
|
|
1645
|
+
const val = e.multiple ? (e.values || []).join(", ") : e.value ?? "";
|
|
1646
|
+
return `${label}${val}`.trim();
|
|
1647
|
+
}
|
|
1648
|
+
case "checkbox":
|
|
1649
|
+
return `${e.checked ? "[x]" : "[ ]"} ${e.label ?? ""}`.trim();
|
|
1650
|
+
case "image":
|
|
1651
|
+
return e.alt ? `[image: ${e.alt}]` : "";
|
|
1652
|
+
case "toc":
|
|
1653
|
+
case "shape":
|
|
1654
|
+
case "signature":
|
|
1655
|
+
return "";
|
|
1656
|
+
default:
|
|
1657
|
+
return "";
|
|
1658
|
+
}
|
|
1659
|
+
}
|
|
1660
|
+
function elementId(el, pageIdx, elIdx) {
|
|
1661
|
+
if (typeof el.id === "string" && el.id.length > 0) return el.id;
|
|
1662
|
+
return `p${pageIdx + 1}e${elIdx}`;
|
|
1663
|
+
}
|
|
1664
|
+
function flatten(doc) {
|
|
1665
|
+
const out = [];
|
|
1666
|
+
(doc.pages || []).forEach((page, pi) => {
|
|
1667
|
+
(page.elements || []).forEach((el, ei) => {
|
|
1668
|
+
out.push({ el, page: pi + 1, id: elementId(el, pi, ei) });
|
|
1669
|
+
});
|
|
1670
|
+
});
|
|
1671
|
+
return out;
|
|
1672
|
+
}
|
|
1673
|
+
function makeChunk(group, breadcrumb) {
|
|
1674
|
+
const parts = group.map((g) => serializeElement(g.el)).filter((s) => s.length > 0);
|
|
1675
|
+
const text = parts.join("\n\n").trim();
|
|
1676
|
+
if (!text) return null;
|
|
1677
|
+
const types = Array.from(new Set(group.map((g) => g.el.type)));
|
|
1678
|
+
return {
|
|
1679
|
+
id: group[0].id,
|
|
1680
|
+
text,
|
|
1681
|
+
path: [...breadcrumb],
|
|
1682
|
+
page: group[0].page,
|
|
1683
|
+
types,
|
|
1684
|
+
tokens: estimateTokens(text),
|
|
1685
|
+
hash: hashText(text)
|
|
1686
|
+
};
|
|
1687
|
+
}
|
|
1688
|
+
function chunkDocument(doc, options = {}) {
|
|
1689
|
+
const strategy = options.strategy ?? "section";
|
|
1690
|
+
const maxTokens = options.maxTokens ?? DEFAULT_MAX_TOKENS;
|
|
1691
|
+
const flat = flatten(doc);
|
|
1692
|
+
const chunks = [];
|
|
1693
|
+
if (strategy === "element") {
|
|
1694
|
+
const crumb2 = [];
|
|
1695
|
+
for (const f of flat) {
|
|
1696
|
+
const lvl = headingLevel(f.el);
|
|
1697
|
+
if (lvl != null) {
|
|
1698
|
+
crumb2.length = Math.max(0, lvl - 1);
|
|
1699
|
+
crumb2[lvl - 1] = serializeElement(f.el);
|
|
1700
|
+
}
|
|
1701
|
+
const c = makeChunk([f], crumb2);
|
|
1702
|
+
if (c) chunks.push(c);
|
|
1703
|
+
}
|
|
1704
|
+
return chunks;
|
|
1705
|
+
}
|
|
1706
|
+
if (strategy === "fixed") {
|
|
1707
|
+
const crumb2 = [];
|
|
1708
|
+
let buf2 = [];
|
|
1709
|
+
let bufTokens = 0;
|
|
1710
|
+
const flush = () => {
|
|
1711
|
+
const c = makeChunk(buf2, crumb2);
|
|
1712
|
+
if (c) chunks.push(c);
|
|
1713
|
+
buf2 = [];
|
|
1714
|
+
bufTokens = 0;
|
|
1715
|
+
};
|
|
1716
|
+
for (const f of flat) {
|
|
1717
|
+
const lvl = headingLevel(f.el);
|
|
1718
|
+
if (lvl != null) {
|
|
1719
|
+
crumb2.length = Math.max(0, lvl - 1);
|
|
1720
|
+
crumb2[lvl - 1] = serializeElement(f.el);
|
|
1721
|
+
}
|
|
1722
|
+
const t = estimateTokens(serializeElement(f.el));
|
|
1723
|
+
if (bufTokens + t > maxTokens && buf2.length > 0) flush();
|
|
1724
|
+
buf2.push(f);
|
|
1725
|
+
bufTokens += t;
|
|
1726
|
+
}
|
|
1727
|
+
flush();
|
|
1728
|
+
return chunks;
|
|
1729
|
+
}
|
|
1730
|
+
const crumb = [];
|
|
1731
|
+
let buf = [];
|
|
1732
|
+
const flushSection = () => {
|
|
1733
|
+
if (buf.length === 0) return;
|
|
1734
|
+
let sub = [];
|
|
1735
|
+
let subTokens = 0;
|
|
1736
|
+
for (const f of buf) {
|
|
1737
|
+
const t = estimateTokens(serializeElement(f.el));
|
|
1738
|
+
if (subTokens + t > maxTokens && sub.length > 0) {
|
|
1739
|
+
const c2 = makeChunk(sub, crumb);
|
|
1740
|
+
if (c2) chunks.push(c2);
|
|
1741
|
+
sub = [];
|
|
1742
|
+
subTokens = 0;
|
|
1743
|
+
}
|
|
1744
|
+
sub.push(f);
|
|
1745
|
+
subTokens += t;
|
|
1746
|
+
}
|
|
1747
|
+
const c = makeChunk(sub, crumb);
|
|
1748
|
+
if (c) chunks.push(c);
|
|
1749
|
+
buf = [];
|
|
1750
|
+
};
|
|
1751
|
+
for (const f of flat) {
|
|
1752
|
+
const lvl = headingLevel(f.el);
|
|
1753
|
+
if (lvl != null) {
|
|
1754
|
+
flushSection();
|
|
1755
|
+
crumb.length = Math.max(0, lvl - 1);
|
|
1756
|
+
crumb[lvl - 1] = serializeElement(f.el);
|
|
1757
|
+
}
|
|
1758
|
+
buf.push(f);
|
|
1759
|
+
}
|
|
1760
|
+
flushSection();
|
|
1761
|
+
return chunks;
|
|
1762
|
+
}
|
|
1763
|
+
async function loadJdf(filePath) {
|
|
1764
|
+
if (filePath.toLowerCase().endsWith(".jdfx")) {
|
|
1765
|
+
const zip = await JSZip.loadAsync(fs.readFileSync(filePath));
|
|
1766
|
+
const docFile = zip.file(JDFX_DOCUMENT_PATH);
|
|
1767
|
+
if (!docFile) throw new Error(`Bundle missing ${JDFX_DOCUMENT_PATH}`);
|
|
1768
|
+
return JSON.parse(await docFile.async("string"));
|
|
1769
|
+
}
|
|
1770
|
+
return JSON.parse(fs.readFileSync(filePath, "utf-8"));
|
|
1771
|
+
}
|
|
1772
|
+
async function chunkFile(inputPath, opts = {}) {
|
|
1773
|
+
const input = path2.resolve(inputPath);
|
|
1774
|
+
if (!fs.existsSync(input)) throw new Error(`File not found: ${input}`);
|
|
1775
|
+
const doc = await loadJdf(input);
|
|
1776
|
+
const strategy = opts.strategy ?? "section";
|
|
1777
|
+
const chunks = chunkDocument(doc, { strategy, maxTokens: opts.maxTokens });
|
|
1778
|
+
const format = opts.format ?? "jsonl";
|
|
1779
|
+
console.log(`Chunking: ${input}`);
|
|
1780
|
+
console.log(`Strategy: ${strategy}${opts.maxTokens ? ` (max ${opts.maxTokens} tokens)` : ""}`);
|
|
1781
|
+
if (format === "inline") {
|
|
1782
|
+
const out = opts.output ? path2.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".jdf");
|
|
1783
|
+
const withIndex = { ...doc, index: { chunker: `jdf-${strategy}-v1`, chunks } };
|
|
1784
|
+
fs.writeFileSync(out, JSON.stringify(withIndex, null, 2));
|
|
1785
|
+
console.log(`Output: ${out} (${chunks.length} chunks in "index" block)`);
|
|
1786
|
+
} else if (format === "json") {
|
|
1787
|
+
const out = opts.output ? path2.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".chunks.json");
|
|
1788
|
+
fs.writeFileSync(out, JSON.stringify(chunks, null, 2));
|
|
1789
|
+
console.log(`Output: ${out} (${chunks.length} chunks)`);
|
|
1790
|
+
} else {
|
|
1791
|
+
const out = opts.output ? path2.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".chunks.jsonl");
|
|
1792
|
+
fs.writeFileSync(out, chunks.map((c) => JSON.stringify(c)).join("\n") + "\n");
|
|
1793
|
+
console.log(`Output: ${out} (${chunks.length} chunks)`);
|
|
1794
|
+
}
|
|
1795
|
+
const totalTokens = chunks.reduce((a, c) => a + c.tokens, 0);
|
|
1796
|
+
console.log(`
|
|
1797
|
+
Done! ${chunks.length} chunks, ~${totalTokens} tokens total.`);
|
|
1798
|
+
return chunks;
|
|
1799
|
+
}
|
|
1800
|
+
var DEFAULT_MODEL = {
|
|
1801
|
+
ollama: "nomic-embed-text",
|
|
1802
|
+
openai: "text-embedding-3-small"
|
|
1803
|
+
};
|
|
1804
|
+
async function embedBatch(provider, model, inputs) {
|
|
1805
|
+
if (inputs.length === 0) return [];
|
|
1806
|
+
switch (provider) {
|
|
1807
|
+
case "ollama":
|
|
1808
|
+
return embedOllama(model, inputs);
|
|
1809
|
+
case "openai":
|
|
1810
|
+
return embedOpenAI(model, inputs);
|
|
1811
|
+
default:
|
|
1812
|
+
throw new Error(`Unknown embedding provider: ${provider}`);
|
|
1813
|
+
}
|
|
1814
|
+
}
|
|
1815
|
+
var OLLAMA_HOST = process.env.OLLAMA_HOST || "http://localhost:11434";
|
|
1816
|
+
var OLLAMA_CONTAINER = "jdf-ollama";
|
|
1817
|
+
async function ollamaUp() {
|
|
1818
|
+
try {
|
|
1819
|
+
const res = await fetch(`${OLLAMA_HOST}/api/tags`, { signal: AbortSignal.timeout(1500) });
|
|
1820
|
+
return res.ok;
|
|
1821
|
+
} catch {
|
|
1822
|
+
return false;
|
|
1823
|
+
}
|
|
1824
|
+
}
|
|
1825
|
+
function dockerReady() {
|
|
1826
|
+
try {
|
|
1827
|
+
execFileSync("docker", ["info"], { stdio: "ignore" });
|
|
1828
|
+
return true;
|
|
1829
|
+
} catch {
|
|
1830
|
+
return false;
|
|
1831
|
+
}
|
|
1832
|
+
}
|
|
1833
|
+
async function ensureOllama(model, autoStart) {
|
|
1834
|
+
if (await ollamaUp()) {
|
|
1835
|
+
await ollamaPull(model);
|
|
1836
|
+
return;
|
|
1837
|
+
}
|
|
1838
|
+
const manualHint = ` Option A \u2014 install Ollama natively (https://ollama.com), then:
|
|
1839
|
+
ollama serve && ollama pull ${model}
|
|
1840
|
+
Option B \u2014 start Docker, then re-run this command (the CLI will launch Ollama for you), or run it yourself:
|
|
1841
|
+
docker run -d --name ${OLLAMA_CONTAINER} -p 11434:11434 -v jdf-ollama:/root/.ollama ollama/ollama
|
|
1842
|
+
docker exec ${OLLAMA_CONTAINER} ollama pull ${model}`;
|
|
1843
|
+
if (!autoStart) {
|
|
1844
|
+
throw new Error(`Ollama isn't running at ${OLLAMA_HOST} and --no-auto-start was given.
|
|
1845
|
+
${manualHint}`);
|
|
1846
|
+
}
|
|
1847
|
+
if (!dockerReady()) {
|
|
1848
|
+
throw new Error(
|
|
1849
|
+
`Ollama isn't running at ${OLLAMA_HOST}, and the Docker daemon isn't available to auto-start it.
|
|
1850
|
+
${manualHint}`
|
|
1851
|
+
);
|
|
1852
|
+
}
|
|
1853
|
+
console.log(`Ollama not reachable \u2014 starting it via Docker (container "${OLLAMA_CONTAINER}")\u2026`);
|
|
1854
|
+
try {
|
|
1855
|
+
execFileSync("docker", ["start", OLLAMA_CONTAINER], { stdio: "ignore" });
|
|
1856
|
+
} catch {
|
|
1857
|
+
try {
|
|
1858
|
+
execFileSync("docker", [
|
|
1859
|
+
"run",
|
|
1860
|
+
"-d",
|
|
1861
|
+
"--name",
|
|
1862
|
+
OLLAMA_CONTAINER,
|
|
1863
|
+
"-p",
|
|
1864
|
+
"11434:11434",
|
|
1865
|
+
"-v",
|
|
1866
|
+
"jdf-ollama:/root/.ollama",
|
|
1867
|
+
"ollama/ollama"
|
|
1868
|
+
], { stdio: "inherit" });
|
|
1869
|
+
} catch (e) {
|
|
1870
|
+
throw new Error(`Failed to start the Ollama container via Docker.
|
|
1871
|
+
${manualHint}`);
|
|
1872
|
+
}
|
|
1873
|
+
}
|
|
1874
|
+
for (let i = 0; i < 30; i++) {
|
|
1875
|
+
if (await ollamaUp()) {
|
|
1876
|
+
console.log(" Ollama is up.");
|
|
1877
|
+
await ollamaPull(model);
|
|
1878
|
+
return;
|
|
1879
|
+
}
|
|
1880
|
+
await new Promise((r) => setTimeout(r, 1e3));
|
|
1881
|
+
}
|
|
1882
|
+
throw new Error("Ollama container started but never became reachable on :11434.");
|
|
1883
|
+
}
|
|
1884
|
+
async function ollamaPull(model) {
|
|
1885
|
+
try {
|
|
1886
|
+
const show = await fetch(`${OLLAMA_HOST}/api/show`, {
|
|
1887
|
+
method: "POST",
|
|
1888
|
+
headers: { "Content-Type": "application/json" },
|
|
1889
|
+
body: JSON.stringify({ name: model })
|
|
1890
|
+
});
|
|
1891
|
+
if (show.ok) return;
|
|
1892
|
+
} catch {
|
|
1893
|
+
}
|
|
1894
|
+
console.log(`Pulling embedding model "${model}" into Ollama (first run only)\u2026`);
|
|
1895
|
+
const res = await fetch(`${OLLAMA_HOST}/api/pull`, {
|
|
1896
|
+
method: "POST",
|
|
1897
|
+
headers: { "Content-Type": "application/json" },
|
|
1898
|
+
body: JSON.stringify({ name: model, stream: false })
|
|
1899
|
+
});
|
|
1900
|
+
if (!res.ok) throw new Error(`Ollama pull failed (${res.status}) for model "${model}".`);
|
|
1901
|
+
console.log(" Model ready.");
|
|
1902
|
+
}
|
|
1903
|
+
async function embedOllama(model, inputs) {
|
|
1904
|
+
const out = [];
|
|
1905
|
+
for (const text of inputs) {
|
|
1906
|
+
const res = await fetch(`${OLLAMA_HOST}/api/embeddings`, {
|
|
1907
|
+
method: "POST",
|
|
1908
|
+
headers: { "Content-Type": "application/json" },
|
|
1909
|
+
body: JSON.stringify({ model, prompt: text })
|
|
1910
|
+
});
|
|
1911
|
+
if (!res.ok) throw new Error(`Ollama embeddings ${res.status} for model "${model}".`);
|
|
1912
|
+
const json = await res.json();
|
|
1913
|
+
if (!Array.isArray(json.embedding)) throw new Error(`Ollama returned no embedding for model "${model}".`);
|
|
1914
|
+
out.push(json.embedding);
|
|
1915
|
+
}
|
|
1916
|
+
return out;
|
|
1917
|
+
}
|
|
1918
|
+
async function prompt(question, hidden = false) {
|
|
1919
|
+
if (!process.stdin.isTTY) return "";
|
|
1920
|
+
const readline = await import('readline');
|
|
1921
|
+
const rl = readline.createInterface({ input: process.stdin, output: process.stdout });
|
|
1922
|
+
if (hidden) {
|
|
1923
|
+
const out = rl;
|
|
1924
|
+
out._writeToOutput = (s) => {
|
|
1925
|
+
if (s.trim().length) out.output.write("*");
|
|
1926
|
+
else out.output.write(s);
|
|
1927
|
+
};
|
|
1928
|
+
}
|
|
1929
|
+
return new Promise((resolve) => rl.question(question, (a) => {
|
|
1930
|
+
rl.close();
|
|
1931
|
+
if (hidden) process.stdout.write("\n");
|
|
1932
|
+
resolve(a.trim());
|
|
1933
|
+
}));
|
|
1934
|
+
}
|
|
1935
|
+
var openaiCreds = null;
|
|
1936
|
+
async function resolveOpenAICreds() {
|
|
1937
|
+
if (openaiCreds) return openaiCreds;
|
|
1938
|
+
let base = process.env.OPENAI_BASE_URL || "";
|
|
1939
|
+
let key = process.env.OPENAI_API_KEY || "";
|
|
1940
|
+
if (!base) base = await prompt("OpenAI API endpoint [https://api.openai.com/v1]: ") || "https://api.openai.com/v1";
|
|
1941
|
+
if (!key) key = await prompt("OpenAI API key: ", true);
|
|
1942
|
+
if (!key) {
|
|
1943
|
+
throw new Error("No OpenAI API key. Set OPENAI_API_KEY (and optionally OPENAI_BASE_URL) or run interactively so the CLI can prompt.");
|
|
1944
|
+
}
|
|
1945
|
+
openaiCreds = { base: base.replace(/\/$/, ""), key };
|
|
1946
|
+
return openaiCreds;
|
|
1947
|
+
}
|
|
1948
|
+
async function embedOpenAI(model, inputs) {
|
|
1949
|
+
const { base, key } = await resolveOpenAICreds();
|
|
1950
|
+
const res = await fetch(`${base}/embeddings`, {
|
|
1951
|
+
method: "POST",
|
|
1952
|
+
headers: { "Content-Type": "application/json", Authorization: `Bearer ${key}` },
|
|
1953
|
+
body: JSON.stringify({ model, input: inputs })
|
|
1954
|
+
});
|
|
1955
|
+
if (!res.ok) {
|
|
1956
|
+
const body = await res.text().catch(() => "");
|
|
1957
|
+
throw new Error(`OpenAI embeddings API ${res.status}: ${body.slice(0, 200)}`);
|
|
1958
|
+
}
|
|
1959
|
+
const json = await res.json();
|
|
1960
|
+
return json.data.sort((a, b) => a.index - b.index).map((d) => d.embedding);
|
|
1961
|
+
}
|
|
1962
|
+
async function loadJdf2(filePath) {
|
|
1963
|
+
if (filePath.toLowerCase().endsWith(".jdfx")) {
|
|
1964
|
+
const zip = await JSZip.loadAsync(fs.readFileSync(filePath));
|
|
1965
|
+
const docFile = zip.file(JDFX_DOCUMENT_PATH);
|
|
1966
|
+
if (!docFile) throw new Error(`Bundle missing ${JDFX_DOCUMENT_PATH}`);
|
|
1967
|
+
return JSON.parse(await docFile.async("string"));
|
|
1968
|
+
}
|
|
1969
|
+
return JSON.parse(fs.readFileSync(filePath, "utf-8"));
|
|
1970
|
+
}
|
|
1971
|
+
function loadCache(cachePath) {
|
|
1972
|
+
try {
|
|
1973
|
+
if (!fs.existsSync(cachePath)) return null;
|
|
1974
|
+
return JSON.parse(fs.readFileSync(cachePath, "utf-8"));
|
|
1975
|
+
} catch {
|
|
1976
|
+
return null;
|
|
1977
|
+
}
|
|
1978
|
+
}
|
|
1979
|
+
function batched(items, size) {
|
|
1980
|
+
const out = [];
|
|
1981
|
+
for (let i = 0; i < items.length; i += size) out.push(items.slice(i, i + size));
|
|
1982
|
+
return out;
|
|
1983
|
+
}
|
|
1984
|
+
async function embedFile(inputPath, opts = {}) {
|
|
1985
|
+
const input = path2.resolve(inputPath);
|
|
1986
|
+
if (!fs.existsSync(input)) throw new Error(`File not found: ${input}`);
|
|
1987
|
+
const provider = opts.provider ?? "ollama";
|
|
1988
|
+
const model = opts.model ?? DEFAULT_MODEL[provider];
|
|
1989
|
+
const strategy = opts.strategy ?? "section";
|
|
1990
|
+
const doc = await loadJdf2(input);
|
|
1991
|
+
const chunks = chunkDocument(doc, { strategy, maxTokens: opts.maxTokens });
|
|
1992
|
+
const output = opts.output ? path2.resolve(opts.output) : input.replace(/\.(jdf|jdfx)$/i, ".embeddings.json");
|
|
1993
|
+
const cachePath = opts.cache ? path2.resolve(opts.cache) : output;
|
|
1994
|
+
console.log(`Embedding: ${input}`);
|
|
1995
|
+
console.log(`Provider: ${provider} / ${model}${provider === "ollama" ? " (local \u2014 no data leaves this machine)" : " (remote API)"}`);
|
|
1996
|
+
console.log(`Strategy: ${strategy} \u2192 ${chunks.length} chunks`);
|
|
1997
|
+
if (provider === "ollama") {
|
|
1998
|
+
await ensureOllama(model, opts.autoStart !== false);
|
|
1999
|
+
}
|
|
2000
|
+
const cache = opts.incremental ? loadCache(cachePath) : null;
|
|
2001
|
+
const cachedVectors = cache?.model === model ? cache.vectors : void 0;
|
|
2002
|
+
if (opts.incremental && cache && cache.model !== model) {
|
|
2003
|
+
console.log(` (cache model ${cache.model} \u2260 ${model} \u2014 re-embedding all)`);
|
|
2004
|
+
}
|
|
2005
|
+
const toEmbed = [];
|
|
2006
|
+
const reused = {};
|
|
2007
|
+
for (const c of chunks) {
|
|
2008
|
+
const hit = cachedVectors?.[c.id];
|
|
2009
|
+
if (hit && hit.hash === c.hash) reused[c.id] = hit;
|
|
2010
|
+
else toEmbed.push(c);
|
|
2011
|
+
}
|
|
2012
|
+
if (opts.incremental) {
|
|
2013
|
+
console.log(` Reused ${Object.keys(reused).length} cached, embedding ${toEmbed.length} changed/new.`);
|
|
2014
|
+
}
|
|
2015
|
+
const vectors = { ...reused };
|
|
2016
|
+
let dims = cache?.dims ?? 0;
|
|
2017
|
+
for (const batch of batched(toEmbed, 96)) {
|
|
2018
|
+
const embs = await embedBatch(provider, model, batch.map((c) => c.text));
|
|
2019
|
+
batch.forEach((c, i) => {
|
|
2020
|
+
vectors[c.id] = { hash: c.hash, vector: embs[i] };
|
|
2021
|
+
if (!dims) dims = embs[i].length;
|
|
2022
|
+
});
|
|
2023
|
+
console.log(` Embedded ${Object.keys(vectors).length}/${chunks.length}\u2026`);
|
|
2024
|
+
}
|
|
2025
|
+
const sidecar = {
|
|
2026
|
+
model,
|
|
2027
|
+
provider,
|
|
2028
|
+
dims,
|
|
2029
|
+
chunker: `jdf-${strategy}-v1`,
|
|
2030
|
+
vectors
|
|
2031
|
+
};
|
|
2032
|
+
fs.writeFileSync(output, JSON.stringify(sidecar));
|
|
2033
|
+
console.log(`
|
|
2034
|
+
Done! ${Object.keys(vectors).length} vectors (${dims}-dim) \u2192 ${output}`);
|
|
2035
|
+
return sidecar;
|
|
2036
|
+
}
|
|
1594
2037
|
|
|
1595
2038
|
// src/index.ts
|
|
1596
2039
|
var HELP = `jdf \u2014 JSON Document Format CLI
|
|
1597
2040
|
|
|
1598
|
-
The CLI exists for
|
|
2041
|
+
The CLI exists for these workflows:
|
|
1599
2042
|
\u2022 PDF \u2192 JDF legacy documents become a structured JSON tree your
|
|
1600
2043
|
RAG / agent / pipeline can read natively.
|
|
1601
2044
|
\u2022 JSON \u2192 JDF LLMs and code emit JSON; this command wraps that JSON
|
|
1602
2045
|
into a validated .jdf (or .jdfx) you can ship.
|
|
2046
|
+
\u2022 JDF \u2192 chunks turn a document into retrieval-ready chunks (RAG).
|
|
2047
|
+
\u2022 JDF \u2192 vectors embed those chunks, incrementally, for a vector store.
|
|
1603
2048
|
|
|
1604
2049
|
Usage:
|
|
1605
2050
|
jdf validate <file.jdf>
|
|
1606
|
-
jdf convert
|
|
2051
|
+
jdf convert <file.{pdf,json,md}> [-o output.{jdf,jdfx}] [--json]
|
|
2052
|
+
jdf chunk <file.{jdf,jdfx}> [--strategy section|element|fixed] [--format jsonl|json|inline] [--max-tokens N] [-o out]
|
|
2053
|
+
jdf embed <file.{jdf,jdfx}> [--provider ollama|openai] [--model NAME] [--strategy \u2026] [--incremental] [-o out]
|
|
1607
2054
|
jdf --help
|
|
1608
2055
|
|
|
1609
2056
|
Commands:
|
|
1610
2057
|
validate Validate a .jdf / .jdfx file against the JDF schema
|
|
1611
|
-
convert Convert a PDF, JSON, or Markdown file into JDF
|
|
1612
|
-
|
|
2058
|
+
convert Convert a PDF, JSON, or Markdown file into JDF (alias: import)
|
|
2059
|
+
chunk Split a JDF document into retrieval-ready chunks (offline, deterministic)
|
|
2060
|
+
embed Compute embeddings for the chunks (local via Ollama by default)
|
|
1613
2061
|
|
|
1614
2062
|
Flags:
|
|
1615
|
-
-o, --output <path> Explicit output path
|
|
1616
|
-
--json
|
|
1617
|
-
|
|
1618
|
-
|
|
1619
|
-
|
|
2063
|
+
-o, --output <path> Explicit output path
|
|
2064
|
+
--json convert: force pure JSON .jdf output (inline base64
|
|
2065
|
+
instead of a .jdfx zip bundle)
|
|
2066
|
+
--strategy <s> chunk/embed: section (default) | element | fixed
|
|
2067
|
+
--format <f> chunk: jsonl (default) | json | inline
|
|
2068
|
+
--max-tokens <n> chunk/embed: soft cap per chunk (default 512)
|
|
2069
|
+
--provider <p> embed: ollama (default, local) | openai (remote API)
|
|
2070
|
+
--model <name> embed: model id (default: nomic-embed-text / text-embedding-3-small)
|
|
2071
|
+
--incremental embed: skip chunks whose content hash is unchanged
|
|
2072
|
+
--no-auto-start embed(ollama): don't auto-launch Ollama via Docker
|
|
2073
|
+
|
|
2074
|
+
Environment (embed):
|
|
2075
|
+
ollama: OLLAMA_HOST (default http://localhost:11434)
|
|
2076
|
+
openai: OPENAI_API_KEY (required), OPENAI_BASE_URL (default api.openai.com)
|
|
1620
2077
|
|
|
1621
2078
|
Examples:
|
|
1622
2079
|
jdf validate spec/examples/hello-world.jdf
|
|
1623
2080
|
jdf convert paper.pdf # PDF \u2192 JDF (or .jdfx for images)
|
|
1624
|
-
jdf convert contract.pdf --json | jq . # PDF \u2192 pure JSON, pipe-friendly
|
|
1625
2081
|
jdf convert response.json -o response.jdf # LLM JSON output \u2192 validated JDF
|
|
1626
|
-
jdf
|
|
2082
|
+
jdf chunk report.jdf # \u2192 report.chunks.jsonl (RAG-ready)
|
|
2083
|
+
jdf chunk report.jdf --format inline # embed the chunk index into the .jdf
|
|
2084
|
+
jdf embed report.jdf # local embeddings via Ollama (auto-setup)
|
|
2085
|
+
jdf embed report.jdf --provider openai --incremental
|
|
1627
2086
|
`;
|
|
1628
|
-
var BOOLEAN_FLAGS = /* @__PURE__ */ new Set(["help", "h", "json", "verbose", "skip-validate"]);
|
|
2087
|
+
var BOOLEAN_FLAGS = /* @__PURE__ */ new Set(["help", "h", "json", "verbose", "skip-validate", "incremental", "no-auto-start"]);
|
|
1629
2088
|
function parseArgs(argv) {
|
|
1630
2089
|
const positional = [];
|
|
1631
2090
|
const flags = {};
|
|
@@ -1715,6 +2174,37 @@ async function main() {
|
|
|
1715
2174
|
process.exit(1);
|
|
1716
2175
|
}
|
|
1717
2176
|
}
|
|
2177
|
+
case "chunk": {
|
|
2178
|
+
const input = positional[0];
|
|
2179
|
+
if (!input) {
|
|
2180
|
+
console.error("Usage: jdf chunk <file.{jdf,jdfx}> [--strategy section|element|fixed] [--format jsonl|json|inline] [--max-tokens N] [-o out]");
|
|
2181
|
+
process.exit(1);
|
|
2182
|
+
}
|
|
2183
|
+
await chunkFile(input, {
|
|
2184
|
+
strategy: typeof flags.strategy === "string" ? flags.strategy : void 0,
|
|
2185
|
+
format: typeof flags.format === "string" ? flags.format : void 0,
|
|
2186
|
+
maxTokens: typeof flags["max-tokens"] === "string" ? parseInt(flags["max-tokens"], 10) : void 0,
|
|
2187
|
+
output: typeof flags.output === "string" ? flags.output : void 0
|
|
2188
|
+
});
|
|
2189
|
+
process.exit(0);
|
|
2190
|
+
}
|
|
2191
|
+
case "embed": {
|
|
2192
|
+
const input = positional[0];
|
|
2193
|
+
if (!input) {
|
|
2194
|
+
console.error("Usage: jdf embed <file.{jdf,jdfx}> [--provider ollama|openai] [--model NAME] [--strategy \u2026] [--incremental] [--no-auto-start] [-o out]");
|
|
2195
|
+
process.exit(1);
|
|
2196
|
+
}
|
|
2197
|
+
await embedFile(input, {
|
|
2198
|
+
provider: typeof flags.provider === "string" ? flags.provider : void 0,
|
|
2199
|
+
model: typeof flags.model === "string" ? flags.model : void 0,
|
|
2200
|
+
strategy: typeof flags.strategy === "string" ? flags.strategy : void 0,
|
|
2201
|
+
maxTokens: typeof flags["max-tokens"] === "string" ? parseInt(flags["max-tokens"], 10) : void 0,
|
|
2202
|
+
incremental: flags.incremental === true,
|
|
2203
|
+
autoStart: flags["no-auto-start"] !== true,
|
|
2204
|
+
output: typeof flags.output === "string" ? flags.output : void 0
|
|
2205
|
+
});
|
|
2206
|
+
process.exit(0);
|
|
2207
|
+
}
|
|
1718
2208
|
default:
|
|
1719
2209
|
console.error(`Unknown command: ${command}`);
|
|
1720
2210
|
console.log(HELP);
|
package/dist/jdf-schema.json
CHANGED
|
@@ -26,9 +26,34 @@
|
|
|
26
26
|
"type": "array",
|
|
27
27
|
"minItems": 1,
|
|
28
28
|
"items": { "$ref": "#/definitions/Page" }
|
|
29
|
-
}
|
|
29
|
+
},
|
|
30
|
+
"index": { "$ref": "#/definitions/DocumentIndex" }
|
|
30
31
|
},
|
|
31
32
|
"definitions": {
|
|
33
|
+
"DocumentIndex": {
|
|
34
|
+
"type": "object",
|
|
35
|
+
"description": "Optional precomputed RAG chunk index (jdf chunk --format inline). Data-only; renderers ignore it.",
|
|
36
|
+
"required": ["chunker", "chunks"],
|
|
37
|
+
"properties": {
|
|
38
|
+
"chunker": { "type": "string" },
|
|
39
|
+
"chunks": {
|
|
40
|
+
"type": "array",
|
|
41
|
+
"items": {
|
|
42
|
+
"type": "object",
|
|
43
|
+
"required": ["id", "text", "hash"],
|
|
44
|
+
"properties": {
|
|
45
|
+
"id": { "type": "string" },
|
|
46
|
+
"text": { "type": "string" },
|
|
47
|
+
"path": { "type": "array", "items": { "type": "string" } },
|
|
48
|
+
"page": { "type": "integer" },
|
|
49
|
+
"types": { "type": "array", "items": { "type": "string" } },
|
|
50
|
+
"tokens": { "type": "integer" },
|
|
51
|
+
"hash": { "type": "string" }
|
|
52
|
+
}
|
|
53
|
+
}
|
|
54
|
+
}
|
|
55
|
+
}
|
|
56
|
+
},
|
|
32
57
|
"Meta": {
|
|
33
58
|
"type": "object",
|
|
34
59
|
"required": ["title"],
|