@equationalapplications/core-llm-wiki 7.4.0 → 7.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +71 -0
- package/dist/{chunk-G5OR2VWD.mjs → chunk-YBWE26XR.mjs} +363 -48
- package/dist/chunk-YBWE26XR.mjs.map +1 -0
- package/dist/index.d.mts +2 -2
- package/dist/index.d.ts +2 -2
- package/dist/index.js +364 -48
- package/dist/index.js.map +1 -1
- package/dist/index.mjs +6 -5
- package/dist/index.mjs.map +1 -1
- package/dist/{testing-Dmh1kfkd.d.mts → testing-d1SrpDvg.d.mts} +117 -3
- package/dist/{testing-Dmh1kfkd.d.ts → testing-d1SrpDvg.d.ts} +117 -3
- package/dist/testing.d.mts +1 -1
- package/dist/testing.d.ts +1 -1
- package/dist/testing.js +360 -45
- package/dist/testing.js.map +1 -1
- package/dist/testing.mjs +1 -1
- package/package.json +2 -2
- package/dist/chunk-G5OR2VWD.mjs.map +0 -1
package/dist/testing.js
CHANGED
|
@@ -593,6 +593,19 @@ function validateTags(tags) {
|
|
|
593
593
|
if (!Array.isArray(tags)) return [];
|
|
594
594
|
return tags.filter((t) => typeof t === "string").map((t) => t.trim().toLowerCase()).filter((t) => t.length > 0 && t.length <= 40).slice(0, 6);
|
|
595
595
|
}
|
|
596
|
+
var MAX_EVIDENCE_QUOTES = 10;
|
|
597
|
+
function normalizeEvidence(raw) {
|
|
598
|
+
if (!Array.isArray(raw)) return void 0;
|
|
599
|
+
const out = [];
|
|
600
|
+
for (const entry of raw) {
|
|
601
|
+
if (typeof entry !== "string") continue;
|
|
602
|
+
const trimmed = entry.trim();
|
|
603
|
+
if (!trimmed) continue;
|
|
604
|
+
out.push(trimmed);
|
|
605
|
+
if (out.length > MAX_EVIDENCE_QUOTES) break;
|
|
606
|
+
}
|
|
607
|
+
return out;
|
|
608
|
+
}
|
|
596
609
|
function validateFact(fact) {
|
|
597
610
|
if (typeof fact?.title !== "string" || typeof fact?.body !== "string") return null;
|
|
598
611
|
const title = clip(fact.title, 80);
|
|
@@ -600,13 +613,17 @@ function validateFact(fact) {
|
|
|
600
613
|
if (!title || !body) return null;
|
|
601
614
|
let confidence = fact.confidence;
|
|
602
615
|
if (confidence !== "certain" && confidence !== "tentative") confidence = "inferred";
|
|
603
|
-
|
|
616
|
+
const valid = {
|
|
604
617
|
...fact,
|
|
605
618
|
title,
|
|
606
619
|
body,
|
|
607
620
|
confidence,
|
|
608
621
|
tags: validateTags(fact.tags)
|
|
609
622
|
};
|
|
623
|
+
const evidence = normalizeEvidence(fact.evidence);
|
|
624
|
+
if (evidence) valid.evidence = evidence;
|
|
625
|
+
else delete valid.evidence;
|
|
626
|
+
return valid;
|
|
610
627
|
}
|
|
611
628
|
function validateTask(task) {
|
|
612
629
|
if (typeof task?.description !== "string") return null;
|
|
@@ -817,8 +834,7 @@ var EmbeddingService = class {
|
|
|
817
834
|
}
|
|
818
835
|
}
|
|
819
836
|
async tryEmbedFact(fact, ctx) {
|
|
820
|
-
|
|
821
|
-
if (typeof embedFn !== "function") return { ok: false, kind: "no_provider" };
|
|
837
|
+
if (typeof this.options.llmProvider.embed !== "function") return { ok: false, kind: "no_provider" };
|
|
822
838
|
let tagsStr;
|
|
823
839
|
if (Array.isArray(fact.tags)) {
|
|
824
840
|
tagsStr = fact.tags.join(" ");
|
|
@@ -835,7 +851,7 @@ var EmbeddingService = class {
|
|
|
835
851
|
const text = clip(`${fact.title} ${fact.body} ${tagsStr}`.trim(), maxEmbedChars);
|
|
836
852
|
let float32Vector;
|
|
837
853
|
try {
|
|
838
|
-
const vector = await
|
|
854
|
+
const vector = await this.options.llmProvider.embed(text);
|
|
839
855
|
if (vector.length === 0 || !vector.every((v) => typeof v === "number" && isFinite(v))) {
|
|
840
856
|
console.warn(`[WikiMemory] embedFact: embed() returned an invalid vector for ${fact.id}; skipping.`);
|
|
841
857
|
this.reportEmbed(ctx, fact, "embedding_failed", "invalid_vector");
|
|
@@ -1510,17 +1526,94 @@ Return ONLY a valid JSON object matching this schema:
|
|
|
1510
1526
|
}
|
|
1511
1527
|
If no manifest type fits a fact, omit that fact from "classifications" entirely \u2014 do not guess.
|
|
1512
1528
|
When echoing an existing fact's title verbatim into "target_title", preserve every JSON escape sequence (\\", \\n, \\\\, \\/) exactly as it appeared in the input body \u2014 do not strip backslashes, do not add unescaped quotes. Do not return markdown, just raw JSON.`;
|
|
1529
|
+
var GROUNDING_SOURCE = {
|
|
1530
|
+
ingest: { key: "facts", source: "the document chunk" },
|
|
1531
|
+
librarian: { key: "facts", source: 'the "summary" text of the events' },
|
|
1532
|
+
heal: { key: "newFacts", source: 'the "summary" text of the recent events or the "body" text of the document anchors' }
|
|
1533
|
+
};
|
|
1534
|
+
function groundingEvidenceBlock(writer, cfg) {
|
|
1535
|
+
const { key, source } = GROUNDING_SOURCE[writer];
|
|
1536
|
+
return `EVIDENCE REQUIREMENT: every object in "${key}" must also carry an "evidence" array of 1 to ${cfg.maxEvidence} quotes. Each quote must be an exact substring copied character-for-character from ${source} (the SOURCE section), at least ${cfg.minEvidenceChars} characters long. Do not paraphrase. Do not quote these instructions, the ontology manifest, or any existing fact. A fact whose quotes cannot be found in the SOURCE section is stored as an unreviewed draft.
|
|
1537
|
+
"evidence": ["exact substring copied from the SOURCE section"]`;
|
|
1538
|
+
}
|
|
1513
1539
|
|
|
1514
1540
|
// src/utils/healConstants.ts
|
|
1515
1541
|
var HEAL_MAX_ANCHORS = 50;
|
|
1516
1542
|
var HEAL_ANCHORS_PER_CANDIDATE = 4;
|
|
1517
1543
|
var HEAL_MAX_FACT_BODY_CHARS_L3 = 4e3;
|
|
1518
1544
|
var HEAL_MAX_TASKS = 50;
|
|
1545
|
+
var HEAL_ANCHOR_BODY_CHARS = 800;
|
|
1546
|
+
|
|
1547
|
+
// src/utils/grounding.ts
|
|
1548
|
+
var GROUNDING_VERIFIER = "process:grounding-check";
|
|
1549
|
+
var WRITERS = ["ingest", "librarian", "heal"];
|
|
1550
|
+
function positiveInt(value, fallback) {
|
|
1551
|
+
return typeof value === "number" && Number.isFinite(value) && value >= 1 ? Math.floor(value) : fallback;
|
|
1552
|
+
}
|
|
1553
|
+
function resolveGrounding(config) {
|
|
1554
|
+
if (!config || config.mode !== "draft") return null;
|
|
1555
|
+
const writers = Array.isArray(config.writers) ? config.writers.filter((w) => WRITERS.includes(w)) : ["ingest"];
|
|
1556
|
+
return {
|
|
1557
|
+
writers: new Set(writers),
|
|
1558
|
+
minEvidenceChars: positiveInt(config.minEvidenceChars, 20),
|
|
1559
|
+
// Never ask for more quotes than checkGrounding accepts (MAX_EVIDENCE_QUOTES).
|
|
1560
|
+
maxEvidence: Math.min(MAX_EVIDENCE_QUOTES, positiveInt(config.maxEvidence, 3)),
|
|
1561
|
+
maxEvidenceChars: positiveInt(config.maxEvidenceChars, 300)
|
|
1562
|
+
};
|
|
1563
|
+
}
|
|
1564
|
+
function normalizeForGrounding(text) {
|
|
1565
|
+
return text.normalize("NFKC").replace(/\s+/g, " ").trim();
|
|
1566
|
+
}
|
|
1567
|
+
function buildGroundingCorpus(parts) {
|
|
1568
|
+
return parts.filter((p) => typeof p === "string").map(normalizeForGrounding);
|
|
1569
|
+
}
|
|
1570
|
+
function checkGrounding(evidence, normalizedCorpus, cfg) {
|
|
1571
|
+
const quotes = evidence ?? [];
|
|
1572
|
+
if (quotes.length > MAX_EVIDENCE_QUOTES) return { status: "failed", reason: "too_many_quotes" };
|
|
1573
|
+
if (quotes.length === 0) return { status: "missing", reason: "no_evidence" };
|
|
1574
|
+
const qualifying = quotes.map(normalizeForGrounding).filter((q) => q.length >= cfg.minEvidenceChars);
|
|
1575
|
+
if (qualifying.length === 0) return { status: "missing", reason: "evidence_too_short" };
|
|
1576
|
+
if (qualifying.some((q) => !normalizedCorpus.some((part) => part.includes(q)))) {
|
|
1577
|
+
return { status: "failed", reason: "quote_not_found" };
|
|
1578
|
+
}
|
|
1579
|
+
return {
|
|
1580
|
+
status: "grounded",
|
|
1581
|
+
retained: qualifying.slice(0, cfg.maxEvidence).map((q) => safeSlice(q, 0, cfg.maxEvidenceChars))
|
|
1582
|
+
};
|
|
1583
|
+
}
|
|
1584
|
+
function groundingOutcome(verdict, now) {
|
|
1585
|
+
if (verdict.status === "grounded") {
|
|
1586
|
+
return {
|
|
1587
|
+
trust: {
|
|
1588
|
+
lifecycle_status: "stable",
|
|
1589
|
+
okf_verified: [{ by: GROUNDING_VERIFIER, at: new Date(now).toISOString() }],
|
|
1590
|
+
last_verified_at: now,
|
|
1591
|
+
last_verified_by: GROUNDING_VERIFIER
|
|
1592
|
+
}
|
|
1593
|
+
};
|
|
1594
|
+
}
|
|
1595
|
+
return {
|
|
1596
|
+
trust: { lifecycle_status: "draft" },
|
|
1597
|
+
diagnostic: { code: verdict.status === "missing" ? "grounding_missing" : "grounding_failed", reason: verdict.reason }
|
|
1598
|
+
};
|
|
1599
|
+
}
|
|
1519
1600
|
|
|
1520
1601
|
// src/services/PromptService.ts
|
|
1521
1602
|
var PromptService = class {
|
|
1522
|
-
constructor(globalOverrides) {
|
|
1603
|
+
constructor(globalOverrides, grounding = null) {
|
|
1523
1604
|
this.globalOverrides = globalOverrides;
|
|
1605
|
+
this.grounding = grounding;
|
|
1606
|
+
}
|
|
1607
|
+
/** The resolved grounding config when `writer` is in `grounding.writers`; otherwise null (spec §6.2). */
|
|
1608
|
+
groundingFor(writer) {
|
|
1609
|
+
return this.grounding?.writers.has(writer) ? this.grounding : null;
|
|
1610
|
+
}
|
|
1611
|
+
/** Appended after any override and after ontology context, so it is always last. */
|
|
1612
|
+
appendGrounding(systemPrompt, writer) {
|
|
1613
|
+
const cfg = this.groundingFor(writer);
|
|
1614
|
+
return cfg ? `${systemPrompt}
|
|
1615
|
+
|
|
1616
|
+
${groundingEvidenceBlock(writer, cfg)}` : systemPrompt;
|
|
1524
1617
|
}
|
|
1525
1618
|
hydrate(template, variables) {
|
|
1526
1619
|
return template.replace(/\{\{\s*(\w+)\s*\}\}/g, (_match, key) => {
|
|
@@ -1550,13 +1643,13 @@ ${ctx.ontologyModeInstructions}`;
|
|
|
1550
1643
|
const hasDocumentChunk = /\{\{\s*documentChunk\s*\}\}/.test(template);
|
|
1551
1644
|
if (hasDocumentChunk || this.hasOntologyPlaceholders(template)) {
|
|
1552
1645
|
return {
|
|
1553
|
-
systemPrompt: this.buildSystemPrompt(template, { documentChunk }, ontologyContext),
|
|
1646
|
+
systemPrompt: this.appendGrounding(this.buildSystemPrompt(template, { documentChunk }, ontologyContext), "ingest"),
|
|
1554
1647
|
userPrompt: hasDocumentChunk ? "Please extract the facts." : `Document Chunk:
|
|
1555
1648
|
${documentChunk}`
|
|
1556
1649
|
};
|
|
1557
1650
|
}
|
|
1558
1651
|
return {
|
|
1559
|
-
systemPrompt: this.appendOntology(template, ontologyContext),
|
|
1652
|
+
systemPrompt: this.appendGrounding(this.appendOntology(template, ontologyContext), "ingest"),
|
|
1560
1653
|
userPrompt: `Document Chunk:
|
|
1561
1654
|
${documentChunk}`
|
|
1562
1655
|
};
|
|
@@ -1565,23 +1658,27 @@ ${documentChunk}`
|
|
|
1565
1658
|
const template = runtimeOverride ?? this.globalOverrides?.librarianSystemPrompt ?? LIBRARIAN_SYSTEM_PROMPT;
|
|
1566
1659
|
const hasEvents = /\{\{\s*events\s*\}\}/.test(template);
|
|
1567
1660
|
const hasCurrentFacts = /\{\{\s*currentFacts\s*\}\}/.test(template);
|
|
1661
|
+
const eventsShown = hasEvents || !hasCurrentFacts;
|
|
1662
|
+
const corpusField = this.groundingFor("librarian") ? { groundingCorpus: buildGroundingCorpus(eventsShown ? events.map(summaryOf) : []) } : {};
|
|
1568
1663
|
if (hasEvents || hasCurrentFacts || this.hasOntologyPlaceholders(template)) {
|
|
1569
1664
|
return {
|
|
1570
|
-
systemPrompt: this.buildSystemPrompt(template, { events, currentFacts }, ontologyContext),
|
|
1665
|
+
systemPrompt: this.appendGrounding(this.buildSystemPrompt(template, { events, currentFacts }, ontologyContext), "librarian"),
|
|
1571
1666
|
userPrompt: hasEvents || hasCurrentFacts ? "Please synthesize the context." : `Events:
|
|
1572
1667
|
${JSON.stringify(events, null, 2)}
|
|
1573
1668
|
|
|
1574
1669
|
Current Facts:
|
|
1575
|
-
${JSON.stringify(currentFacts, null, 2)}
|
|
1670
|
+
${JSON.stringify(currentFacts, null, 2)}`,
|
|
1671
|
+
...corpusField
|
|
1576
1672
|
};
|
|
1577
1673
|
}
|
|
1578
1674
|
return {
|
|
1579
|
-
systemPrompt: this.appendOntology(template, ontologyContext),
|
|
1675
|
+
systemPrompt: this.appendGrounding(this.appendOntology(template, ontologyContext), "librarian"),
|
|
1580
1676
|
userPrompt: `Events:
|
|
1581
1677
|
${JSON.stringify(events, null, 2)}
|
|
1582
1678
|
|
|
1583
1679
|
Current Facts:
|
|
1584
|
-
${JSON.stringify(currentFacts, null, 2)}
|
|
1680
|
+
${JSON.stringify(currentFacts, null, 2)}`,
|
|
1681
|
+
...corpusField
|
|
1585
1682
|
};
|
|
1586
1683
|
}
|
|
1587
1684
|
/**
|
|
@@ -1611,40 +1708,54 @@ ${JSON.stringify(currentFacts, null, 2)}`
|
|
|
1611
1708
|
const effectiveEvents = attemptLevel >= 2 ? [] : recentEvents;
|
|
1612
1709
|
const maxAnchors = Math.max(1, Math.min(HEAL_MAX_ANCHORS, healCandidates.length * HEAL_ANCHORS_PER_CANDIDATE));
|
|
1613
1710
|
const effectiveAnchors = documentAnchors.slice(0, maxAnchors);
|
|
1711
|
+
const template = runtimeOverride ?? this.globalOverrides?.healSystemPrompt ?? HEAL_SYSTEM_PROMPT;
|
|
1712
|
+
const hasRecentEvents = /\{\{\s*recentEvents\s*\}\}/.test(template);
|
|
1713
|
+
const hasDocumentAnchors = /\{\{\s*documentAnchors\s*\}\}/.test(template);
|
|
1714
|
+
const usesPlaceholders = /\{\{\s*healCandidates\s*\}\}/.test(template) || hasDocumentAnchors || /\{\{\s*allTasks\s*\}\}/.test(template) || hasRecentEvents;
|
|
1715
|
+
const healGrounding = this.groundingFor("heal");
|
|
1716
|
+
const promptAnchors = healGrounding ? effectiveAnchors.map(toGroundingAnchor) : effectiveAnchors;
|
|
1717
|
+
const eventsShown = !usesPlaceholders || hasRecentEvents;
|
|
1718
|
+
const anchorsShown = !usesPlaceholders || hasDocumentAnchors;
|
|
1719
|
+
const groundingCorpus = healGrounding ? buildGroundingCorpus([
|
|
1720
|
+
...eventsShown ? effectiveEvents.map(summaryOf) : [],
|
|
1721
|
+
...anchorsShown ? effectiveAnchors.map((a, i) => ({ status: a?.lifecycle_status, shown: promptAnchors[i] })).filter((x) => x.status !== "draft").map((x) => x.shown?.body) : []
|
|
1722
|
+
]) : null;
|
|
1614
1723
|
const { shapedCandidates, degraded } = applyBodyTruncation(
|
|
1615
1724
|
healCandidates,
|
|
1616
1725
|
attemptLevel,
|
|
1617
1726
|
bodyTruncationChars
|
|
1618
1727
|
);
|
|
1619
|
-
const
|
|
1620
|
-
if (
|
|
1728
|
+
const corpusField = groundingCorpus !== null ? { groundingCorpus } : {};
|
|
1729
|
+
if (usesPlaceholders) {
|
|
1621
1730
|
return {
|
|
1622
1731
|
prompts: {
|
|
1623
|
-
systemPrompt: this.hydrate(template, {
|
|
1732
|
+
systemPrompt: this.appendGrounding(this.hydrate(template, {
|
|
1624
1733
|
healCandidates: shapedCandidates,
|
|
1625
|
-
documentAnchors:
|
|
1734
|
+
documentAnchors: promptAnchors,
|
|
1626
1735
|
allTasks: effectiveTasks,
|
|
1627
1736
|
recentEvents: effectiveEvents
|
|
1628
|
-
}),
|
|
1737
|
+
}), "heal"),
|
|
1629
1738
|
userPrompt: "Please heal the memory graph."
|
|
1630
1739
|
},
|
|
1631
|
-
degraded
|
|
1740
|
+
degraded,
|
|
1741
|
+
...corpusField
|
|
1632
1742
|
};
|
|
1633
1743
|
}
|
|
1634
1744
|
return {
|
|
1635
1745
|
prompts: {
|
|
1636
|
-
systemPrompt: template,
|
|
1746
|
+
systemPrompt: this.appendGrounding(template, "heal"),
|
|
1637
1747
|
userPrompt: `Heal Candidates:
|
|
1638
1748
|
${JSON.stringify(shapedCandidates, null, 2)}
|
|
1639
1749
|
Document Anchors (DO NOT MODIFY OR DELETE):
|
|
1640
|
-
${JSON.stringify(
|
|
1750
|
+
${JSON.stringify(promptAnchors, null, 2)}
|
|
1641
1751
|
All Tasks:
|
|
1642
1752
|
${JSON.stringify(effectiveTasks, null, 2)}
|
|
1643
1753
|
Recent Events:
|
|
1644
1754
|
${JSON.stringify(effectiveEvents, null, 2)}
|
|
1645
1755
|
The following document anchors are provided for contradiction detection only. Do not include them in \`downgraded\`, \`deleted\`, or \`newFacts\`.`
|
|
1646
1756
|
},
|
|
1647
|
-
degraded
|
|
1757
|
+
degraded,
|
|
1758
|
+
...corpusField
|
|
1648
1759
|
};
|
|
1649
1760
|
}
|
|
1650
1761
|
buildOntologyBackfillPrompt(facts, runtimeOverride, ontologyContext) {
|
|
@@ -1694,6 +1805,19 @@ function applyBodyTruncation(candidates, attemptLevel, bodyTruncationChars) {
|
|
|
1694
1805
|
}
|
|
1695
1806
|
return { shapedCandidates, degraded };
|
|
1696
1807
|
}
|
|
1808
|
+
function summaryOf(event) {
|
|
1809
|
+
return event?.summary;
|
|
1810
|
+
}
|
|
1811
|
+
function toGroundingAnchor(anchor) {
|
|
1812
|
+
if (typeof anchor !== "object" || anchor === null) return anchor;
|
|
1813
|
+
const a = anchor;
|
|
1814
|
+
return {
|
|
1815
|
+
id: a.id,
|
|
1816
|
+
title: a.title,
|
|
1817
|
+
source_ref: a.source_ref,
|
|
1818
|
+
body: typeof a.body === "string" ? safeSlice(a.body, 0, HEAL_ANCHOR_BODY_CHARS) : ""
|
|
1819
|
+
};
|
|
1820
|
+
}
|
|
1697
1821
|
|
|
1698
1822
|
// src/utils/chunkingDefaults.ts
|
|
1699
1823
|
var DEFAULT_MAX_CHUNK_LENGTH = 12e3;
|
|
@@ -1723,7 +1847,7 @@ var IngestionService = class {
|
|
|
1723
1847
|
this.jobManager = jobManager;
|
|
1724
1848
|
this.embeddingService = embeddingService;
|
|
1725
1849
|
this.ontologyService = ontologyService;
|
|
1726
|
-
this.promptService = promptService ?? new PromptService(this.options.config?.prompts);
|
|
1850
|
+
this.promptService = promptService ?? new PromptService(this.options.config?.prompts, resolveGrounding(this.options.config?.grounding));
|
|
1727
1851
|
}
|
|
1728
1852
|
async ingestDocument(entityId, params, opts) {
|
|
1729
1853
|
const sourceRef = normalizeSourceRef(params.sourceRef);
|
|
@@ -1756,6 +1880,7 @@ var IngestionService = class {
|
|
|
1756
1880
|
const { chunks, truncated } = chunkText(params.documentChunk, maxChunkLength, chunkOverlap);
|
|
1757
1881
|
if (chunks.length === 0) return zeroChunkResult();
|
|
1758
1882
|
const ontologyContext = await this.ontologyService?.buildPromptContext(entityId) ?? null;
|
|
1883
|
+
const ingestGrounding = this.promptService.groundingFor("ingest");
|
|
1759
1884
|
const chunkResults = await withConcurrency(
|
|
1760
1885
|
chunks.map((chunk, chunkIndex) => async () => {
|
|
1761
1886
|
const { systemPrompt, userPrompt } = this.promptService.buildIngestPrompt(
|
|
@@ -1779,11 +1904,16 @@ var IngestionService = class {
|
|
|
1779
1904
|
rejected.push({ itemIndex, reason: factRejectionReason(raw) });
|
|
1780
1905
|
}
|
|
1781
1906
|
});
|
|
1907
|
+
const verdicts = ingestGrounding ? (() => {
|
|
1908
|
+
const corpus = buildGroundingCorpus([chunk]);
|
|
1909
|
+
return facts.map((f) => checkGrounding(f.evidence, corpus, ingestGrounding));
|
|
1910
|
+
})() : [];
|
|
1782
1911
|
return {
|
|
1783
1912
|
status: "ok",
|
|
1784
1913
|
facts,
|
|
1785
1914
|
itemIndexes,
|
|
1786
1915
|
rejected,
|
|
1916
|
+
verdicts,
|
|
1787
1917
|
ontology_updates: result2.ontology_updates
|
|
1788
1918
|
};
|
|
1789
1919
|
} catch (e) {
|
|
@@ -1828,6 +1958,7 @@ var IngestionService = class {
|
|
|
1828
1958
|
const failures = [];
|
|
1829
1959
|
const seen = /* @__PURE__ */ new Set();
|
|
1830
1960
|
const orderedChunkFacts = [];
|
|
1961
|
+
const groundingLedger = /* @__PURE__ */ new Map();
|
|
1831
1962
|
const diagBuffer = new DiagnosticBuffer();
|
|
1832
1963
|
const diagBase = { entityId, operation: "ingest", trigger: "call" };
|
|
1833
1964
|
for (const [chunkIndex, slot] of chunkResults.entries()) {
|
|
@@ -1846,6 +1977,7 @@ var IngestionService = class {
|
|
|
1846
1977
|
if (!seen.has(normalizedTitle)) {
|
|
1847
1978
|
seen.add(normalizedTitle);
|
|
1848
1979
|
dedupedFacts.push(fact);
|
|
1980
|
+
if (ingestGrounding) groundingLedger.set(fact, { verdict: slot.verdicts[k], chunkIndex, itemIndex: slot.itemIndexes[k] });
|
|
1849
1981
|
} else {
|
|
1850
1982
|
diagBuffer.push({
|
|
1851
1983
|
...diagBase,
|
|
@@ -1882,7 +2014,7 @@ var IngestionService = class {
|
|
|
1882
2014
|
try {
|
|
1883
2015
|
if (failedChunks === 0) {
|
|
1884
2016
|
const fullResult = await this.db.withTransactionAsync(async (tx) => {
|
|
1885
|
-
return await this.runFullUpsertGraph(entityId, sourceRef, sourceHash, orderedChunkFacts, tx, diagBuffer);
|
|
2017
|
+
return await this.runFullUpsertGraph(entityId, sourceRef, sourceHash, orderedChunkFacts, tx, diagBuffer, groundingLedger);
|
|
1886
2018
|
});
|
|
1887
2019
|
deletedSourceFactIds.push(...fullResult.deletedSourceFactIds);
|
|
1888
2020
|
insertedFacts.push(...fullResult.insertedFacts);
|
|
@@ -1890,7 +2022,7 @@ var IngestionService = class {
|
|
|
1890
2022
|
const partialResult = await this.db.withTransactionAsync(async (tx) => {
|
|
1891
2023
|
const flat = [];
|
|
1892
2024
|
for (const slot of orderedChunkFacts) flat.push(...slot.facts);
|
|
1893
|
-
return await this.appendPartialFacts(entityId, sourceRef, flat, tx, diagBuffer);
|
|
2025
|
+
return await this.appendPartialFacts(entityId, sourceRef, flat, tx, diagBuffer, groundingLedger);
|
|
1894
2026
|
});
|
|
1895
2027
|
insertedFacts.push(...partialResult.insertedDescriptors);
|
|
1896
2028
|
}
|
|
@@ -2032,7 +2164,8 @@ var IngestionService = class {
|
|
|
2032
2164
|
last_accessed_at: null,
|
|
2033
2165
|
access_count: 0,
|
|
2034
2166
|
deleted_at: null,
|
|
2035
|
-
okf_type: normalized.okf_type
|
|
2167
|
+
okf_type: normalized.okf_type,
|
|
2168
|
+
...opts?.nodeTrust?.get(node.id)
|
|
2036
2169
|
};
|
|
2037
2170
|
wikiFacts.push(wikiFact);
|
|
2038
2171
|
}
|
|
@@ -2123,9 +2256,10 @@ var IngestionService = class {
|
|
|
2123
2256
|
* On the partial path the caller never invokes this method; the empty
|
|
2124
2257
|
* arrays stay empty.
|
|
2125
2258
|
*/
|
|
2126
|
-
async runFullUpsertGraph(entityId, sourceRef, sourceHash, orderedChunkFacts, tx, diagBuffer) {
|
|
2259
|
+
async runFullUpsertGraph(entityId, sourceRef, sourceHash, orderedChunkFacts, tx, diagBuffer, groundingLedger = /* @__PURE__ */ new Map()) {
|
|
2127
2260
|
const deletedSourceFactIds = [];
|
|
2128
2261
|
const insertedFacts = [];
|
|
2262
|
+
const nodeTrust = /* @__PURE__ */ new Map();
|
|
2129
2263
|
deletedSourceFactIds.push(...await this.entryRepo.findIdsBySource(entityId, sourceRef, null, tx, false));
|
|
2130
2264
|
const titleIndex = /* @__PURE__ */ new Map();
|
|
2131
2265
|
const existingFacts = await this.entryRepo.findRecentByEntityId(entityId, 500, tx, sourceRef);
|
|
@@ -2169,6 +2303,20 @@ var IngestionService = class {
|
|
|
2169
2303
|
tags: fact.tags,
|
|
2170
2304
|
confidence: fact.confidence
|
|
2171
2305
|
});
|
|
2306
|
+
const grounding = groundingLedger.get(fact);
|
|
2307
|
+
if (grounding) {
|
|
2308
|
+
const outcome = groundingOutcome(grounding.verdict, now);
|
|
2309
|
+
nodeTrust.set(id, outcome.trust);
|
|
2310
|
+
if (outcome.diagnostic) {
|
|
2311
|
+
diagBuffer.push({
|
|
2312
|
+
entityId,
|
|
2313
|
+
operation: "ingest",
|
|
2314
|
+
trigger: "call",
|
|
2315
|
+
code: outcome.diagnostic.code,
|
|
2316
|
+
detail: { factId: id, sourceRef, chunkIndex: grounding.chunkIndex, itemIndex: grounding.itemIndex, reason: outcome.diagnostic.reason }
|
|
2317
|
+
});
|
|
2318
|
+
}
|
|
2319
|
+
}
|
|
2172
2320
|
insertedFacts.push({ id, entity_id: entityId, title: fact.title, body: fact.body, tags: JSON.stringify(fact.tags) });
|
|
2173
2321
|
titleIndex.set(normalizeTitleKey(fact.title), { id, okf_type: normalized.okf_type });
|
|
2174
2322
|
if (normalized.edges.length > 0) {
|
|
@@ -2200,7 +2348,7 @@ var IngestionService = class {
|
|
|
2200
2348
|
entityId,
|
|
2201
2349
|
{ sourceRef, sourceHash, nodes: hostNodes, edges: hostEdges },
|
|
2202
2350
|
tx,
|
|
2203
|
-
{ strict: false, diag: { buffer: diagBuffer, operation: "ingest" } }
|
|
2351
|
+
{ strict: false, diag: { buffer: diagBuffer, operation: "ingest" }, ...nodeTrust.size > 0 ? { nodeTrust } : {} }
|
|
2204
2352
|
);
|
|
2205
2353
|
return { deletedSourceFactIds, insertedFacts };
|
|
2206
2354
|
}
|
|
@@ -2224,7 +2372,7 @@ var IngestionService = class {
|
|
|
2224
2372
|
* Runs INSIDE the caller's `tx`. Does not open a nested transaction.
|
|
2225
2373
|
* Returns `{ inserted, skippedDuplicate }` for observability.
|
|
2226
2374
|
*/
|
|
2227
|
-
async appendPartialFacts(entityId, sourceRef, dedupedFacts, tx, diagBuffer) {
|
|
2375
|
+
async appendPartialFacts(entityId, sourceRef, dedupedFacts, tx, diagBuffer, groundingLedger = /* @__PURE__ */ new Map()) {
|
|
2228
2376
|
const liveIds = await this.entryRepo.findIdsBySource(entityId, sourceRef, null, tx, false);
|
|
2229
2377
|
const liveFacts = liveIds.length === 0 ? [] : await this.entryRepo.findByIds(liveIds, void 0, tx);
|
|
2230
2378
|
const liveTitles = new Set(liveFacts.map((f) => normalizeTitleKey(f.title)));
|
|
@@ -2247,6 +2395,17 @@ var IngestionService = class {
|
|
|
2247
2395
|
}
|
|
2248
2396
|
liveTitles.add(normalizedTitle);
|
|
2249
2397
|
const id = generateId("fact_");
|
|
2398
|
+
const grounding = groundingLedger.get(fact);
|
|
2399
|
+
const outcome = grounding ? groundingOutcome(grounding.verdict, now) : null;
|
|
2400
|
+
if (grounding && outcome?.diagnostic) {
|
|
2401
|
+
diagBuffer.push({
|
|
2402
|
+
entityId,
|
|
2403
|
+
operation: "ingest",
|
|
2404
|
+
trigger: "call",
|
|
2405
|
+
code: outcome.diagnostic.code,
|
|
2406
|
+
detail: { factId: id, sourceRef, chunkIndex: grounding.chunkIndex, itemIndex: grounding.itemIndex, reason: outcome.diagnostic.reason }
|
|
2407
|
+
});
|
|
2408
|
+
}
|
|
2250
2409
|
const wikiFact = {
|
|
2251
2410
|
id,
|
|
2252
2411
|
entity_id: entityId,
|
|
@@ -2265,7 +2424,8 @@ var IngestionService = class {
|
|
|
2265
2424
|
last_accessed_at: null,
|
|
2266
2425
|
access_count: 0,
|
|
2267
2426
|
deleted_at: null,
|
|
2268
|
-
okf_type: null
|
|
2427
|
+
okf_type: null,
|
|
2428
|
+
...outcome?.trust
|
|
2269
2429
|
};
|
|
2270
2430
|
await this.entryRepo.upsert(wikiFact, tx);
|
|
2271
2431
|
insertedDescriptors.push({
|
|
@@ -2281,6 +2441,51 @@ var IngestionService = class {
|
|
|
2281
2441
|
}
|
|
2282
2442
|
};
|
|
2283
2443
|
|
|
2444
|
+
// src/utils/classifier.ts
|
|
2445
|
+
var isUnit = (v) => typeof v === "number" && Number.isFinite(v) && v >= 0 && v <= 1;
|
|
2446
|
+
function validateClassifierAnswer(response, key, question) {
|
|
2447
|
+
try {
|
|
2448
|
+
if (response === null || typeof response !== "object") return { ok: false, reason: "malformed" };
|
|
2449
|
+
const answers = response.answers;
|
|
2450
|
+
if (answers === null || typeof answers !== "object") return { ok: false, reason: "missing_answer" };
|
|
2451
|
+
if (!Object.prototype.hasOwnProperty.call(answers, key)) return { ok: false, reason: "missing_answer" };
|
|
2452
|
+
const raw = answers[key];
|
|
2453
|
+
if (raw === null || typeof raw !== "object") return { ok: false, reason: "missing_answer" };
|
|
2454
|
+
const a = raw;
|
|
2455
|
+
if (a.kind !== question.kind) return { ok: false, reason: "kind_mismatch" };
|
|
2456
|
+
if (question.kind === "choice") {
|
|
2457
|
+
if (typeof a.choice !== "string" || !question.options.includes(a.choice)) return { ok: false, reason: "choice_not_offered" };
|
|
2458
|
+
if (!isUnit(a.confidence)) return { ok: false, reason: "invalid_probability" };
|
|
2459
|
+
const probs = a.probabilities;
|
|
2460
|
+
if (probs === null || typeof probs !== "object" || Array.isArray(probs)) return { ok: false, reason: "invalid_probability" };
|
|
2461
|
+
const probabilities = {};
|
|
2462
|
+
for (const [k, v] of Object.entries(probs)) {
|
|
2463
|
+
if (!isUnit(v)) return { ok: false, reason: "invalid_probability" };
|
|
2464
|
+
probabilities[k] = v;
|
|
2465
|
+
}
|
|
2466
|
+
return { ok: true, answer: { kind: "choice", choice: a.choice, confidence: a.confidence, probabilities } };
|
|
2467
|
+
}
|
|
2468
|
+
if (question.kind === "binary") {
|
|
2469
|
+
if (!isUnit(a.probability)) return { ok: false, reason: "invalid_probability" };
|
|
2470
|
+
return { ok: true, answer: { kind: "binary", probability: a.probability } };
|
|
2471
|
+
}
|
|
2472
|
+
const maxScore = question.levels.length - 1;
|
|
2473
|
+
if (typeof a.score !== "number" || !Number.isFinite(a.score) || a.score < 0 || a.score > maxScore) {
|
|
2474
|
+
return { ok: false, reason: "score_out_of_range" };
|
|
2475
|
+
}
|
|
2476
|
+
if (!isUnit(a.confidence)) return { ok: false, reason: "invalid_probability" };
|
|
2477
|
+
if (!Array.isArray(a.probabilities) || !a.probabilities.every(isUnit)) return { ok: false, reason: "invalid_probability" };
|
|
2478
|
+
return { ok: true, answer: { kind: "score", score: a.score, confidence: a.confidence, probabilities: a.probabilities.slice() } };
|
|
2479
|
+
} catch {
|
|
2480
|
+
return { ok: false, reason: "malformed" };
|
|
2481
|
+
}
|
|
2482
|
+
}
|
|
2483
|
+
function classifierStateForFact(fact) {
|
|
2484
|
+
const parts = [fact.title, fact.body];
|
|
2485
|
+
if (Array.isArray(fact.tags) && fact.tags.length > 0) parts.push(`Tags: ${fact.tags.join(", ")}`);
|
|
2486
|
+
return parts.join("\n\n");
|
|
2487
|
+
}
|
|
2488
|
+
|
|
2284
2489
|
// src/utils/embedding.ts
|
|
2285
2490
|
function parseEmbedding(blob, text) {
|
|
2286
2491
|
if (blob && blob.byteLength > 0) {
|
|
@@ -2416,7 +2621,7 @@ async function runBatched(args) {
|
|
|
2416
2621
|
}
|
|
2417
2622
|
let result;
|
|
2418
2623
|
try {
|
|
2419
|
-
result = parse(responseText, batch);
|
|
2624
|
+
result = parse(responseText, batch, prompts);
|
|
2420
2625
|
} catch (err) {
|
|
2421
2626
|
await onFailure(batch, err, attemptLevel, false);
|
|
2422
2627
|
return;
|
|
@@ -2497,7 +2702,7 @@ var MaintenanceService = class {
|
|
|
2497
2702
|
this.jobManager = jobManager;
|
|
2498
2703
|
this.embeddingService = embeddingService;
|
|
2499
2704
|
this.ontologyService = ontologyService;
|
|
2500
|
-
this.promptService = promptService ?? new PromptService(this.options.config?.prompts);
|
|
2705
|
+
this.promptService = promptService ?? new PromptService(this.options.config?.prompts, resolveGrounding(this.options.config?.grounding));
|
|
2501
2706
|
}
|
|
2502
2707
|
async runPrune(entityId, options) {
|
|
2503
2708
|
this.jobManager.acquireLock("prune", entityId);
|
|
@@ -2802,8 +3007,10 @@ var MaintenanceService = class {
|
|
|
2802
3007
|
};
|
|
2803
3008
|
});
|
|
2804
3009
|
const ontologyContext = await this.ontologyService?.buildPromptContext(entityId) ?? null;
|
|
2805
|
-
const
|
|
2806
|
-
|
|
3010
|
+
const promptEvents = events.reverse();
|
|
3011
|
+
const librarianGrounding = this.promptService.groundingFor("librarian");
|
|
3012
|
+
const { systemPrompt, userPrompt, groundingCorpus: librarianCorpus = [] } = this.promptService.buildLibrarianPrompt(
|
|
3013
|
+
promptEvents,
|
|
2807
3014
|
currentFacts,
|
|
2808
3015
|
promptOverride,
|
|
2809
3016
|
ontologyContext
|
|
@@ -2872,6 +3079,10 @@ var MaintenanceService = class {
|
|
|
2872
3079
|
const validationDrops = [];
|
|
2873
3080
|
const normalized = this.ontologyService?.validateAndNormalizeFact(ontologyFact, manifest, { strict: false, drops: validationDrops }) ?? { okf_type: null, edges: [] };
|
|
2874
3081
|
for (const drop of validationDrops) diagBuffer.push(edgeDropDiagnostic(drop, { ...diagBase, factId: id }));
|
|
3082
|
+
const grounding = librarianGrounding ? groundingOutcome(checkGrounding(fact.evidence, librarianCorpus, librarianGrounding), now) : null;
|
|
3083
|
+
if (grounding?.diagnostic) {
|
|
3084
|
+
diagBuffer.push({ ...diagBase, code: grounding.diagnostic.code, detail: { factId: id, itemIndex: validFactItemIndexes[k], reason: grounding.diagnostic.reason } });
|
|
3085
|
+
}
|
|
2875
3086
|
const factObj = {
|
|
2876
3087
|
id,
|
|
2877
3088
|
entity_id: entityId,
|
|
@@ -2887,7 +3098,8 @@ var MaintenanceService = class {
|
|
|
2887
3098
|
last_accessed_at: null,
|
|
2888
3099
|
access_count: 0,
|
|
2889
3100
|
deleted_at: null,
|
|
2890
|
-
okf_type: normalized.okf_type
|
|
3101
|
+
okf_type: normalized.okf_type,
|
|
3102
|
+
...grounding?.trust
|
|
2891
3103
|
};
|
|
2892
3104
|
await this.entryRepo.upsert(factObj, tx);
|
|
2893
3105
|
insertedFacts.push({ id, entity_id: entityId, title: fact.title, body: fact.body, tags: JSON.stringify(fact.tags) });
|
|
@@ -3003,11 +3215,14 @@ var MaintenanceService = class {
|
|
|
3003
3215
|
}
|
|
3004
3216
|
const allTasks = await this.taskRepo.findAllPending([entityId], HEAL_MAX_TASKS);
|
|
3005
3217
|
const recentEvents = await this.eventRepo.getRecent(entityId, 20);
|
|
3218
|
+
const healGrounding = this.promptService.groundingFor("heal");
|
|
3006
3219
|
const toPromptShape = (f) => {
|
|
3007
3220
|
const { embedding: _embedding, embedding_blob: _blob, ...rest } = f;
|
|
3008
|
-
|
|
3221
|
+
const shown = healGrounding ? (({ lifecycle_status: _l, okf_verified: _v, last_verified_at: _a, last_verified_by: _b, ...r }) => r)(rest) : rest;
|
|
3222
|
+
return { ...shown, tags: typeof shown.tags === "string" ? JSON.parse(shown.tags) : shown.tags };
|
|
3009
3223
|
};
|
|
3010
3224
|
const anchorCache = /* @__PURE__ */ new Map();
|
|
3225
|
+
const corpusByPrompt = /* @__PURE__ */ new WeakMap();
|
|
3011
3226
|
const degraded = [];
|
|
3012
3227
|
const outcome = await runBatched({
|
|
3013
3228
|
items: healCandidates,
|
|
@@ -3020,9 +3235,10 @@ var MaintenanceService = class {
|
|
|
3020
3235
|
// not overfetch the same 200 keyword hits a 25-fact batch once did.
|
|
3021
3236
|
// buildHealPrompt applies the matching cap on its side — values must match.
|
|
3022
3237
|
Math.min(HEAL_MAX_ANCHORS, batch.length * HEAL_ANCHORS_PER_CANDIDATE),
|
|
3023
|
-
anchorCache
|
|
3238
|
+
anchorCache,
|
|
3239
|
+
healGrounding !== null
|
|
3024
3240
|
);
|
|
3025
|
-
const { prompts, degraded: batchDegraded } = await this.promptService.buildHealPrompt(
|
|
3241
|
+
const { prompts, degraded: batchDegraded, groundingCorpus } = await this.promptService.buildHealPrompt(
|
|
3026
3242
|
batch.map(toPromptShape),
|
|
3027
3243
|
documentAnchors,
|
|
3028
3244
|
allTasks,
|
|
@@ -3032,16 +3248,18 @@ var MaintenanceService = class {
|
|
|
3032
3248
|
bodyTruncationChars
|
|
3033
3249
|
);
|
|
3034
3250
|
degraded.push(...batchDegraded);
|
|
3251
|
+
if (groundingCorpus !== void 0) corpusByPrompt.set(prompts, groundingCorpus);
|
|
3035
3252
|
return prompts;
|
|
3036
3253
|
},
|
|
3037
3254
|
call: (prompts) => this.options.llmProvider.generateText(prompts),
|
|
3038
|
-
parse: (responseText, batch) => {
|
|
3255
|
+
parse: (responseText, batch, prompts) => {
|
|
3039
3256
|
const result = parseJsonResponse(responseText);
|
|
3040
3257
|
return {
|
|
3041
3258
|
batch,
|
|
3042
3259
|
downgraded: Array.isArray(result.downgraded) ? result.downgraded : [],
|
|
3043
3260
|
deleted: Array.isArray(result.deleted) ? result.deleted : [],
|
|
3044
|
-
newFacts: Array.isArray(result.newFacts) ? result.newFacts : []
|
|
3261
|
+
newFacts: Array.isArray(result.newFacts) ? result.newFacts : [],
|
|
3262
|
+
corpus: corpusByPrompt.get(prompts) ?? null
|
|
3045
3263
|
};
|
|
3046
3264
|
},
|
|
3047
3265
|
maxOutputTokens: this.options.llmProvider.maxOutputTokens,
|
|
@@ -3056,6 +3274,7 @@ var MaintenanceService = class {
|
|
|
3056
3274
|
const safeDeletedSet = /* @__PURE__ */ new Set();
|
|
3057
3275
|
const newFacts = [];
|
|
3058
3276
|
const newFactItemIndexes = [];
|
|
3277
|
+
const newFactCorpora = [];
|
|
3059
3278
|
for (const batchResult of outcome.results) {
|
|
3060
3279
|
const mutableIds = new Set(batchResult.batch.map((f) => f.id));
|
|
3061
3280
|
for (const id of batchResult.downgraded) if (mutableIds.has(id)) safeDowngradedSet.add(id);
|
|
@@ -3063,6 +3282,7 @@ var MaintenanceService = class {
|
|
|
3063
3282
|
batchResult.newFacts.forEach((raw, itemIndex) => {
|
|
3064
3283
|
newFacts.push(raw);
|
|
3065
3284
|
newFactItemIndexes.push(itemIndex);
|
|
3285
|
+
newFactCorpora.push(batchResult.corpus);
|
|
3066
3286
|
});
|
|
3067
3287
|
}
|
|
3068
3288
|
const safeDowngraded = Array.from(safeDowngradedSet);
|
|
@@ -3070,12 +3290,14 @@ var MaintenanceService = class {
|
|
|
3070
3290
|
const diagBuffer = new DiagnosticBuffer();
|
|
3071
3291
|
const validNewFacts = [];
|
|
3072
3292
|
const validNewFactItemIndexes = [];
|
|
3293
|
+
const validNewFactCorpora = [];
|
|
3073
3294
|
newFacts.forEach((raw, k) => {
|
|
3074
3295
|
const itemIndex = newFactItemIndexes[k];
|
|
3075
3296
|
const valid = validateFact(raw);
|
|
3076
3297
|
if (valid) {
|
|
3077
3298
|
validNewFacts.push(valid);
|
|
3078
3299
|
validNewFactItemIndexes.push(itemIndex);
|
|
3300
|
+
validNewFactCorpora.push(newFactCorpora[k]);
|
|
3079
3301
|
} else {
|
|
3080
3302
|
diagBuffer.push({ ...diagBase, code: "fact_rejected", detail: { itemIndex, reason: factRejectionReason(raw) } });
|
|
3081
3303
|
}
|
|
@@ -3105,6 +3327,10 @@ var MaintenanceService = class {
|
|
|
3105
3327
|
continue;
|
|
3106
3328
|
}
|
|
3107
3329
|
const id = generateId("fact_");
|
|
3330
|
+
const grounding = healGrounding ? groundingOutcome(checkGrounding(fact.evidence, validNewFactCorpora[k] ?? [], healGrounding), now) : null;
|
|
3331
|
+
if (grounding?.diagnostic) {
|
|
3332
|
+
diagBuffer.push({ ...diagBase, code: grounding.diagnostic.code, detail: { factId: id, itemIndex: validNewFactItemIndexes[k], reason: grounding.diagnostic.reason } });
|
|
3333
|
+
}
|
|
3108
3334
|
const factObj = {
|
|
3109
3335
|
id,
|
|
3110
3336
|
entity_id: entityId,
|
|
@@ -3119,7 +3345,8 @@ var MaintenanceService = class {
|
|
|
3119
3345
|
updated_at: now,
|
|
3120
3346
|
last_accessed_at: null,
|
|
3121
3347
|
access_count: 0,
|
|
3122
|
-
deleted_at: null
|
|
3348
|
+
deleted_at: null,
|
|
3349
|
+
...grounding?.trust
|
|
3123
3350
|
};
|
|
3124
3351
|
await this.entryRepo.upsert(factObj, tx);
|
|
3125
3352
|
insertedFacts.push({ id, entity_id: entityId, title: fact.title, body: fact.body, tags: JSON.stringify(fact.tags) });
|
|
@@ -3191,7 +3418,7 @@ var MaintenanceService = class {
|
|
|
3191
3418
|
if (!ontologyService) {
|
|
3192
3419
|
return { ...zeroed, remaining: 0, deferred: 0 };
|
|
3193
3420
|
}
|
|
3194
|
-
const { mode } = await ontologyService.getEffectiveState(entityId);
|
|
3421
|
+
const { mode, manifest: effectiveManifest } = await ontologyService.getEffectiveState(entityId);
|
|
3195
3422
|
if (mode === "off") {
|
|
3196
3423
|
const counts2 = await this.entryRepo.countUntypedByEntityId(entityId, recheckCutoff);
|
|
3197
3424
|
return { ...zeroed, remaining: 0, deferred: counts2.deferred };
|
|
@@ -3201,6 +3428,14 @@ var MaintenanceService = class {
|
|
|
3201
3428
|
const counts2 = await this.entryRepo.countUntypedByEntityId(entityId, recheckCutoff);
|
|
3202
3429
|
return { ...zeroed, remaining: counts2.eligible, deferred: counts2.deferred };
|
|
3203
3430
|
}
|
|
3431
|
+
const classifierMode = options?.classifier ?? this.options.config?.ontology?.backfillClassifier ?? "llm";
|
|
3432
|
+
const classify = this.options.llmProvider.classify;
|
|
3433
|
+
if (classifierMode === "auto" && typeof classify === "function") {
|
|
3434
|
+
const slugs = effectiveManifest.node_types.map((n) => n.type);
|
|
3435
|
+
if (slugs.length > 0 && slugs.length <= 255) {
|
|
3436
|
+
return this._runClassifierBackfill(entityId, candidates, effectiveManifest, now, recheckCutoff);
|
|
3437
|
+
}
|
|
3438
|
+
}
|
|
3204
3439
|
const ontologyContext = await ontologyService.buildPromptContext(entityId);
|
|
3205
3440
|
const toPromptShape = (f) => ({ id: f.id, title: f.title, body: f.body, tags: f.tags });
|
|
3206
3441
|
const buildPrompt = (facts) => this.promptService.buildOntologyBackfillPrompt(
|
|
@@ -3279,6 +3514,87 @@ var MaintenanceService = class {
|
|
|
3279
3514
|
deferred: counts.deferred
|
|
3280
3515
|
};
|
|
3281
3516
|
}
|
|
3517
|
+
/**
|
|
3518
|
+
* Classifier-mode backfill (spec §7.3): one `choice` question per untyped
|
|
3519
|
+
* fact over the effective manifest's node-type slugs. Accepted answers go
|
|
3520
|
+
* through `_applyOntologyBackfillBatch` exactly like LLM classifications.
|
|
3521
|
+
* No edges are proposed.
|
|
3522
|
+
*/
|
|
3523
|
+
async _runClassifierBackfill(entityId, candidates, manifest, now, recheckCutoff) {
|
|
3524
|
+
const provider = this.options.llmProvider;
|
|
3525
|
+
const rawMin = this.options.config?.ontology?.classifyMinConfidence;
|
|
3526
|
+
const minConfidence = typeof rawMin === "number" && Number.isFinite(rawMin) && rawMin >= 0 && rawMin <= 1 ? rawMin : 0.5;
|
|
3527
|
+
const rawConcurrency = this.options.config?.chunkConcurrency ?? 1;
|
|
3528
|
+
const concurrency = Number.isFinite(rawConcurrency) && rawConcurrency >= 1 ? Math.floor(rawConcurrency) : 1;
|
|
3529
|
+
const question = {
|
|
3530
|
+
kind: "choice",
|
|
3531
|
+
options: manifest.node_types.map((n) => n.type),
|
|
3532
|
+
instructions: "Choose the ontology node type that best describes this fact.\n" + manifest.node_types.map((n) => `- ${n.type}: ${n.description}`).join("\n")
|
|
3533
|
+
};
|
|
3534
|
+
const outcomes = await withConcurrency(
|
|
3535
|
+
candidates.map((fact) => async () => {
|
|
3536
|
+
let response;
|
|
3537
|
+
try {
|
|
3538
|
+
response = await provider.classify({ state: classifierStateForFact(fact), questions: { okf_type: question } });
|
|
3539
|
+
} catch {
|
|
3540
|
+
return { fact, kind: "threw" };
|
|
3541
|
+
}
|
|
3542
|
+
const checked = validateClassifierAnswer(response, "okf_type", question);
|
|
3543
|
+
if (!checked.ok) return { fact, kind: "invalid", reason: checked.reason };
|
|
3544
|
+
if (checked.answer.kind !== "choice") return { fact, kind: "invalid", reason: "kind_mismatch" };
|
|
3545
|
+
if (checked.answer.confidence < minConfidence) return { fact, kind: "low_confidence" };
|
|
3546
|
+
return { fact, kind: "accepted", okfType: checked.answer.choice };
|
|
3547
|
+
}),
|
|
3548
|
+
concurrency
|
|
3549
|
+
);
|
|
3550
|
+
const attempted = outcomes.filter((o) => o.kind !== "threw").map((o) => o.fact);
|
|
3551
|
+
const skipped = outcomes.length - attempted.length;
|
|
3552
|
+
const classifications = outcomes.filter((o) => o.kind === "accepted").map((o) => ({ id: o.fact.id, okf_type: o.okfType }));
|
|
3553
|
+
const invalidCount = outcomes.filter((o) => o.kind === "invalid").length;
|
|
3554
|
+
let typed = 0;
|
|
3555
|
+
let failedValidation = 0;
|
|
3556
|
+
let edgesAdded = 0;
|
|
3557
|
+
let aborted = false;
|
|
3558
|
+
if (attempted.length > 0) {
|
|
3559
|
+
const applied = await this._applyOntologyBackfillBatch(
|
|
3560
|
+
entityId,
|
|
3561
|
+
{ batch: attempted, classifications, ontologyUpdates: void 0 },
|
|
3562
|
+
now
|
|
3563
|
+
);
|
|
3564
|
+
aborted = applied.abortedOntologyOff;
|
|
3565
|
+
if (!aborted) {
|
|
3566
|
+
typed = applied.typed;
|
|
3567
|
+
failedValidation = invalidCount + applied.failedValidation;
|
|
3568
|
+
edgesAdded = applied.edgesAdded;
|
|
3569
|
+
}
|
|
3570
|
+
}
|
|
3571
|
+
const diagBuffer = new DiagnosticBuffer();
|
|
3572
|
+
const diagBase = { entityId, operation: "ontologyBackfill", trigger: "call" };
|
|
3573
|
+
for (const o of outcomes) {
|
|
3574
|
+
if (o.kind === "low_confidence") {
|
|
3575
|
+
diagBuffer.push({ ...diagBase, code: "classification_low_confidence", detail: { factId: o.fact.id, reason: "below_threshold" } });
|
|
3576
|
+
} else if (o.kind === "invalid") {
|
|
3577
|
+
diagBuffer.push({ ...diagBase, code: "classification_invalid", detail: { factId: o.fact.id, reason: o.reason } });
|
|
3578
|
+
} else if (o.kind === "threw") {
|
|
3579
|
+
diagBuffer.push({ ...diagBase, code: "classification_invalid", detail: { factId: o.fact.id, reason: "classify_threw" } });
|
|
3580
|
+
}
|
|
3581
|
+
}
|
|
3582
|
+
diagBuffer.flush(this.options);
|
|
3583
|
+
const counts = await this.entryRepo.countUntypedByEntityId(entityId, recheckCutoff);
|
|
3584
|
+
if (aborted) {
|
|
3585
|
+
return { scanned: 0, typed: 0, failedValidation: 0, edgesAdded: 0, skipped, remaining: 0, deferred: counts.deferred };
|
|
3586
|
+
}
|
|
3587
|
+
this.searchService.evictCache(entityId);
|
|
3588
|
+
return {
|
|
3589
|
+
scanned: candidates.length,
|
|
3590
|
+
typed,
|
|
3591
|
+
failedValidation,
|
|
3592
|
+
edgesAdded,
|
|
3593
|
+
skipped,
|
|
3594
|
+
remaining: counts.eligible,
|
|
3595
|
+
deferred: counts.deferred
|
|
3596
|
+
};
|
|
3597
|
+
}
|
|
3282
3598
|
/**
|
|
3283
3599
|
* Applies one parsed backfill batch in its own transaction. Per-batch rather
|
|
3284
3600
|
* than one transaction for the pass, so mergeEmergentUpdates semantics and
|
|
@@ -3394,7 +3710,7 @@ var MaintenanceService = class {
|
|
|
3394
3710
|
* batches that reduce to the same query share one lookup. Caller-owned and
|
|
3395
3711
|
* per-pass — see the call site in doRunHeal.
|
|
3396
3712
|
*/
|
|
3397
|
-
async _selectHealAnchors(entityId, batch, cap = HEAL_MAX_ANCHORS, cache) {
|
|
3713
|
+
async _selectHealAnchors(entityId, batch, cap = HEAL_MAX_ANCHORS, cache, withBodies = false) {
|
|
3398
3714
|
const query = batch.map((f) => f.title).join(" ").trim();
|
|
3399
3715
|
if (!query) return [];
|
|
3400
3716
|
const cached = cache?.get(query);
|
|
@@ -3407,7 +3723,7 @@ var MaintenanceService = class {
|
|
|
3407
3723
|
const hitIds = hits.map((h) => h.id);
|
|
3408
3724
|
const anchors = [];
|
|
3409
3725
|
if (hitIds.length > 0) {
|
|
3410
|
-
const rows = await this.entryRepo.findAnchorRowsByIds(entityId, hitIds);
|
|
3726
|
+
const rows = withBodies ? await this.entryRepo.findAnchorRowsByIds(entityId, hitIds, void 0, { withBody: true }) : await this.entryRepo.findAnchorRowsByIds(entityId, hitIds);
|
|
3411
3727
|
const byId = new Map(rows.map((r) => [r.id, r]));
|
|
3412
3728
|
for (const id of hitIds) {
|
|
3413
3729
|
const row = byId.get(id);
|
|
@@ -3563,7 +3879,6 @@ var RetrievalService = class {
|
|
|
3563
3879
|
const hybridWeight = options?.hybridWeight ?? config?.hybridWeight;
|
|
3564
3880
|
const weight = hybridWeight !== void 0 && !Number.isNaN(hybridWeight) ? Math.max(0, Math.min(1, hybridWeight)) : void 0;
|
|
3565
3881
|
const skipEmbed = weight === 0;
|
|
3566
|
-
const embedFn = this.options.llmProvider.embed;
|
|
3567
3882
|
let facts = [];
|
|
3568
3883
|
let scoreByFactId;
|
|
3569
3884
|
if (maxResults === 0) ; else if (trimmedQuery) {
|
|
@@ -3574,11 +3889,11 @@ var RetrievalService = class {
|
|
|
3574
3889
|
const padLimit = (n) => n >= Number.MAX_SAFE_INTEGER ? n : n + draftPad;
|
|
3575
3890
|
if (scoredEntityIds.length === 0) {
|
|
3576
3891
|
usedEmbed = true;
|
|
3577
|
-
} else if (!skipEmbed &&
|
|
3892
|
+
} else if (!skipEmbed && typeof this.options.llmProvider.embed === "function") {
|
|
3578
3893
|
let rankerShouldRethrow = false;
|
|
3579
3894
|
let pendingRankerFallbackError;
|
|
3580
3895
|
try {
|
|
3581
|
-
const queryVec = await
|
|
3896
|
+
const queryVec = await this.options.llmProvider.embed(trimmedQuery);
|
|
3582
3897
|
if (queryVec.length === 0 || !queryVec.every((v) => typeof v === "number" && isFinite(v))) {
|
|
3583
3898
|
throw new Error(
|
|
3584
3899
|
"embed() returned an empty or non-finite vector. Falling back to keyword search."
|