@equationalapplications/core-llm-wiki 4.22.0 → 4.23.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +10 -6
- package/dist/{chunk-LCAU6NYC.mjs → chunk-MYZJLVX4.mjs} +311 -45
- package/dist/chunk-MYZJLVX4.mjs.map +1 -0
- package/dist/index.d.mts +2 -2
- package/dist/index.d.ts +2 -2
- package/dist/index.js +348 -43
- package/dist/index.js.map +1 -1
- package/dist/index.mjs +41 -2
- package/dist/index.mjs.map +1 -1
- package/dist/{testing-Bk8J6QLX.d.mts → testing-CBjAuTSl.d.mts} +72 -0
- package/dist/{testing-Bk8J6QLX.d.ts → testing-CBjAuTSl.d.ts} +72 -0
- package/dist/testing.d.mts +1 -1
- package/dist/testing.d.ts +1 -1
- package/dist/testing.js +309 -43
- package/dist/testing.js.map +1 -1
- package/dist/testing.mjs +1 -1
- package/package.json +6 -6
- package/dist/chunk-LCAU6NYC.mjs.map +0 -1
package/README.md
CHANGED
|
@@ -42,6 +42,7 @@ const wikiMemory = new WikiMemory(db, {
|
|
|
42
42
|
// Your LLM call for extracting facts, tasks
|
|
43
43
|
return 'Model output';
|
|
44
44
|
},
|
|
45
|
+
maxOutputTokens: 4096, // optional — lets runHeal/runOntologyBackfill size their first LLM call correctly instead of discovering the ceiling via a truncated response and retrying smaller
|
|
45
46
|
embed: async (text: string) => {
|
|
46
47
|
// Your embedding service (e.g., OpenAI, Cohere, local)
|
|
47
48
|
const response = await fetch('https://your-app.example.com/api/embed', {
|
|
@@ -550,18 +551,21 @@ have `okf_type = NULL` and no edges, so `traverseGraph` cannot reach them.
|
|
|
550
551
|
|
|
551
552
|
```ts
|
|
552
553
|
const result = await wiki.runOntologyBackfill(entityId);
|
|
553
|
-
// { scanned, typed, failedValidation, edgesAdded, remaining, deferred }
|
|
554
|
+
// { scanned, typed, failedValidation, edgesAdded, remaining, deferred, skipped }
|
|
554
555
|
```
|
|
555
556
|
|
|
556
557
|
- **When to call it:** you own the trigger — the library provides the operation,
|
|
557
558
|
not a scheduler (same as `runLibrarian`/`runHeal`). A good default is after
|
|
558
559
|
each sync/import completes. `WikiBusyError` under concurrency is safe to
|
|
559
560
|
swallow; the next trigger retries.
|
|
560
|
-
- **Cost model:** one LLM
|
|
561
|
-
one SELECT. Each run scans at most 25 facts (override via
|
|
562
|
-
`options.batchSize`), oldest first
|
|
563
|
-
|
|
564
|
-
|
|
561
|
+
- **Cost model:** one or more LLM calls only when eligible untyped facts exist;
|
|
562
|
+
otherwise one SELECT. Each run scans at most 25 facts (override via
|
|
563
|
+
`options.batchSize`), oldest first. Facts are sent to the LLM in bounded
|
|
564
|
+
sub-batches — sized from `llmProvider.maxOutputTokens` when supplied,
|
|
565
|
+
otherwise a conservative default — with the full serialized prompt
|
|
566
|
+
(system + user) capped at 40k chars. A sub-batch whose response is truncated
|
|
567
|
+
or fails to parse is halved and retried automatically; a single fact that
|
|
568
|
+
still fails alone is counted in `skipped` rather than aborting the run.
|
|
565
569
|
- **Convergence:** loop `while (result.remaining > 0)` to drain a backlog.
|
|
566
570
|
`remaining` counts only *eligible* untyped facts, so the loop terminates even
|
|
567
571
|
when unclassifiable facts exist; those are cooldown-stamped and retried after
|
|
@@ -256,9 +256,23 @@ var _SearchService = class _SearchService {
|
|
|
256
256
|
this.entryRepo = entryRepo;
|
|
257
257
|
this.miniSearchEntryIdsByEntity = /* @__PURE__ */ new Map();
|
|
258
258
|
this.vectorCache = /* @__PURE__ */ new Map();
|
|
259
|
+
/**
|
|
260
|
+
* Serializes rebuilds. `rebuildIndex` awaits a repository read between
|
|
261
|
+
* snapshotting the previous id set and discarding it, so two concurrent
|
|
262
|
+
* sync() calls for one entity can interleave: a slow, stale read lands last
|
|
263
|
+
* and discards documents the fresh read just added. Chaining also keeps
|
|
264
|
+
* discard()/addAll() out of each other's way, which is what accrued the
|
|
265
|
+
* auto-vacuum debt behind the TypeError in #64.
|
|
266
|
+
*/
|
|
267
|
+
this.syncChain = Promise.resolve();
|
|
259
268
|
this.miniSearch = new MiniSearch({
|
|
260
269
|
fields: ["title", "body", "tags"],
|
|
261
270
|
storeFields: ["entity_id"],
|
|
271
|
+
// Vacuuming is driven explicitly at the end of each serialized rebuild
|
|
272
|
+
// (see sync). Auto-vacuum fires on its own schedule, asynchronously with
|
|
273
|
+
// respect to the caller, and traversing the tree mid-rebuild is what
|
|
274
|
+
// threw the uncaught TypeError in MiniSearch.performVacuuming (#64).
|
|
275
|
+
autoVacuum: false,
|
|
262
276
|
searchOptions: {
|
|
263
277
|
boost: { title: 2 },
|
|
264
278
|
fuzzy: 0.2,
|
|
@@ -269,10 +283,26 @@ var _SearchService = class _SearchService {
|
|
|
269
283
|
/**
|
|
270
284
|
* Rebuilds the search index and clears the vector cache for a given entity.
|
|
271
285
|
* A direct replacement for manually syncing state after a DB transaction.
|
|
286
|
+
*
|
|
287
|
+
* Rebuilds are serialized per instance and never reject: the MiniSearch index
|
|
288
|
+
* is a rebuildable cache over SQLite, so degraded keyword search is the
|
|
289
|
+
* correct failure mode and killing the host process is not.
|
|
272
290
|
*/
|
|
273
291
|
async sync(entityId) {
|
|
274
|
-
|
|
275
|
-
|
|
292
|
+
const work = this.syncChain.then(async () => {
|
|
293
|
+
try {
|
|
294
|
+
try {
|
|
295
|
+
await this.rebuildIndex(entityId);
|
|
296
|
+
await this.miniSearch.vacuum();
|
|
297
|
+
} finally {
|
|
298
|
+
this.evictCache(entityId);
|
|
299
|
+
}
|
|
300
|
+
} catch (err) {
|
|
301
|
+
console.warn(`[WikiMemory] search index rebuild failed for ${entityId ?? "*"}:`, err);
|
|
302
|
+
}
|
|
303
|
+
});
|
|
304
|
+
this.syncChain = work;
|
|
305
|
+
return work;
|
|
276
306
|
}
|
|
277
307
|
/**
|
|
278
308
|
* Clears the parsed vector cache. Useful for mid-loop flush guarantees
|
|
@@ -1423,12 +1453,122 @@ var MetadataRepository = class extends BaseRepository {
|
|
|
1423
1453
|
}
|
|
1424
1454
|
};
|
|
1425
1455
|
|
|
1456
|
+
// src/services/BoundedLlmCall.ts
|
|
1457
|
+
var DEFAULT_BATCH_SIZE = 10;
|
|
1458
|
+
var ESTIMATED_OUTPUT_TOKENS_PER_ITEM = 150;
|
|
1459
|
+
var OUTPUT_BUDGET_FRACTION = 0.8;
|
|
1460
|
+
var TRUNCATION_PATTERNS = [
|
|
1461
|
+
/truncat/i,
|
|
1462
|
+
/token limit/i,
|
|
1463
|
+
/max(imum)?[ _-]?tokens?/i,
|
|
1464
|
+
/output limit/i,
|
|
1465
|
+
/length limit/i,
|
|
1466
|
+
/finish[_ ]?reason/i
|
|
1467
|
+
];
|
|
1468
|
+
var EXCEEDS_LIMIT_PATTERN = /exceed[a-z]*[^.]{0,40}\b(model|context)?[ _-]?limit/i;
|
|
1469
|
+
function isTruncationError(err) {
|
|
1470
|
+
const message = err instanceof Error ? err.message : String(err ?? "");
|
|
1471
|
+
if (EXCEEDS_LIMIT_PATTERN.test(message)) return false;
|
|
1472
|
+
return TRUNCATION_PATTERNS.some((pattern) => pattern.test(message));
|
|
1473
|
+
}
|
|
1474
|
+
function initialBatchSize(maxOutputTokens) {
|
|
1475
|
+
if (!maxOutputTokens || !Number.isFinite(maxOutputTokens) || maxOutputTokens <= 0) {
|
|
1476
|
+
return DEFAULT_BATCH_SIZE;
|
|
1477
|
+
}
|
|
1478
|
+
const estimate = Math.floor(
|
|
1479
|
+
maxOutputTokens * OUTPUT_BUDGET_FRACTION / ESTIMATED_OUTPUT_TOKENS_PER_ITEM
|
|
1480
|
+
);
|
|
1481
|
+
return Math.max(DEFAULT_BATCH_SIZE, estimate);
|
|
1482
|
+
}
|
|
1483
|
+
var promptLength = (prompts) => prompts.systemPrompt.length + prompts.userPrompt.length;
|
|
1484
|
+
async function runBatched(args) {
|
|
1485
|
+
const { items, buildPrompt, call, parse, maxPromptChars, maxOutputTokens, onSkip } = args;
|
|
1486
|
+
const results = [];
|
|
1487
|
+
const skipped = [];
|
|
1488
|
+
let batches = 0;
|
|
1489
|
+
let batchSize = initialBatchSize(maxOutputTokens);
|
|
1490
|
+
const trim = async (candidate) => {
|
|
1491
|
+
const whole = await buildPrompt(candidate);
|
|
1492
|
+
if (candidate.length <= 1 || promptLength(whole) <= maxPromptChars) {
|
|
1493
|
+
return { batch: candidate, prompts: whole };
|
|
1494
|
+
}
|
|
1495
|
+
let low = 2;
|
|
1496
|
+
let high = candidate.length - 1;
|
|
1497
|
+
let best;
|
|
1498
|
+
let bestPrompts;
|
|
1499
|
+
while (low <= high) {
|
|
1500
|
+
const mid = Math.floor((low + high) / 2);
|
|
1501
|
+
const batch = candidate.slice(0, mid);
|
|
1502
|
+
const prompts = await buildPrompt(batch);
|
|
1503
|
+
if (promptLength(prompts) <= maxPromptChars) {
|
|
1504
|
+
best = batch;
|
|
1505
|
+
bestPrompts = prompts;
|
|
1506
|
+
low = mid + 1;
|
|
1507
|
+
} else {
|
|
1508
|
+
high = mid - 1;
|
|
1509
|
+
}
|
|
1510
|
+
}
|
|
1511
|
+
if (best && bestPrompts) return { batch: best, prompts: bestPrompts };
|
|
1512
|
+
const single = candidate.slice(0, 1);
|
|
1513
|
+
return { batch: single, prompts: await buildPrompt(single) };
|
|
1514
|
+
};
|
|
1515
|
+
const onFailure = async (batch, err) => {
|
|
1516
|
+
if (batch.length <= 1) {
|
|
1517
|
+
if (batch.length === 1) {
|
|
1518
|
+
skipped.push(batch[0]);
|
|
1519
|
+
onSkip?.(batch[0], err);
|
|
1520
|
+
}
|
|
1521
|
+
return;
|
|
1522
|
+
}
|
|
1523
|
+
const mid = Math.ceil(batch.length / 2);
|
|
1524
|
+
if (mid < batchSize) batchSize = mid;
|
|
1525
|
+
let i = 0;
|
|
1526
|
+
while (i < batch.length) {
|
|
1527
|
+
const size = Math.min(batchSize, batch.length - i);
|
|
1528
|
+
const trimmed = await trim(batch.slice(i, i + size));
|
|
1529
|
+
await attempt(trimmed.batch, trimmed.prompts);
|
|
1530
|
+
i += trimmed.batch.length;
|
|
1531
|
+
}
|
|
1532
|
+
};
|
|
1533
|
+
const attempt = async (batch, prebuilt) => {
|
|
1534
|
+
if (batch.length === 0) return;
|
|
1535
|
+
const prompts = prebuilt ?? await buildPrompt(batch);
|
|
1536
|
+
batches++;
|
|
1537
|
+
let responseText;
|
|
1538
|
+
try {
|
|
1539
|
+
responseText = await call(prompts);
|
|
1540
|
+
} catch (err) {
|
|
1541
|
+
if (!isTruncationError(err)) throw err;
|
|
1542
|
+
await onFailure(batch, err);
|
|
1543
|
+
return;
|
|
1544
|
+
}
|
|
1545
|
+
let result;
|
|
1546
|
+
try {
|
|
1547
|
+
result = parse(responseText, batch);
|
|
1548
|
+
} catch (err) {
|
|
1549
|
+
await onFailure(batch, err);
|
|
1550
|
+
return;
|
|
1551
|
+
}
|
|
1552
|
+
results.push(result);
|
|
1553
|
+
};
|
|
1554
|
+
let index = 0;
|
|
1555
|
+
while (index < items.length) {
|
|
1556
|
+
const { batch, prompts } = await trim(items.slice(index, index + batchSize));
|
|
1557
|
+
index += batch.length;
|
|
1558
|
+
await attempt(batch, prompts);
|
|
1559
|
+
}
|
|
1560
|
+
return { results, skipped, batches };
|
|
1561
|
+
}
|
|
1562
|
+
|
|
1426
1563
|
// src/services/MaintenanceService.ts
|
|
1427
1564
|
var FUZZY_THRESHOLD = 0.5;
|
|
1428
1565
|
var MIN_TOKENS_TO_QUALIFY = 3;
|
|
1429
1566
|
var ONTOLOGY_BACKFILL_BATCH_SIZE = 25;
|
|
1430
1567
|
var ONTOLOGY_BACKFILL_MAX_PROMPT_CHARS = 4e4;
|
|
1431
1568
|
var ONTOLOGY_BACKFILL_RECHECK_MS = 7 * 24 * 60 * 60 * 1e3;
|
|
1569
|
+
var HEAL_MAX_ANCHORS = 50;
|
|
1570
|
+
var HEAL_ANCHOR_SEARCH_OVERFETCH = 4;
|
|
1571
|
+
var HEAL_MAX_PROMPT_CHARS = 4e4;
|
|
1432
1572
|
var MaintenanceService = class {
|
|
1433
1573
|
constructor(db, prefix, options, entryRepo, taskRepo, eventRepo, metadataRepo, searchService, jobManager, embeddingService, promptService, ontologyService) {
|
|
1434
1574
|
this.db = db;
|
|
@@ -1811,30 +1951,56 @@ var MaintenanceService = class {
|
|
|
1811
1951
|
console.warn(`[WikiMemory] onEmbeddingPersisted hook failed during heal orphan pass for ${factId}:`, hookErr);
|
|
1812
1952
|
}
|
|
1813
1953
|
}
|
|
1814
|
-
const
|
|
1954
|
+
const healCandidates = await this.entryRepo.findHealCandidatesByEntityId(entityId);
|
|
1815
1955
|
const allTasks = await this.taskRepo.findAllPending([entityId]);
|
|
1816
1956
|
const recentEvents = await this.eventRepo.getRecent(entityId, 20);
|
|
1817
|
-
const
|
|
1818
|
-
const documentAnchors = allFactsRows.filter((f) => f.source_type === "immutable_document").map(({ id, title, source_ref }) => ({ id, title, source_ref }));
|
|
1819
|
-
const healCandidatesForPrompt = healCandidates.map((f) => {
|
|
1957
|
+
const toPromptShape = (f) => {
|
|
1820
1958
|
const { embedding: _embedding, embedding_blob: _blob, ...rest } = f;
|
|
1821
1959
|
return { ...rest, tags: typeof rest.tags === "string" ? JSON.parse(rest.tags) : rest.tags };
|
|
1960
|
+
};
|
|
1961
|
+
const anchorCache = /* @__PURE__ */ new Map();
|
|
1962
|
+
const outcome = await runBatched({
|
|
1963
|
+
items: healCandidates,
|
|
1964
|
+
buildPrompt: async (batch) => {
|
|
1965
|
+
const documentAnchors = await this._selectHealAnchors(entityId, batch, anchorCache);
|
|
1966
|
+
return this.promptService.buildHealPrompt(
|
|
1967
|
+
batch.map(toPromptShape),
|
|
1968
|
+
documentAnchors,
|
|
1969
|
+
allTasks,
|
|
1970
|
+
recentEvents,
|
|
1971
|
+
promptOverride
|
|
1972
|
+
);
|
|
1973
|
+
},
|
|
1974
|
+
call: (prompts) => this.options.llmProvider.generateText(prompts),
|
|
1975
|
+
parse: (responseText, batch) => {
|
|
1976
|
+
const result = parseJsonResponse(responseText);
|
|
1977
|
+
return {
|
|
1978
|
+
batch,
|
|
1979
|
+
downgraded: Array.isArray(result.downgraded) ? result.downgraded : [],
|
|
1980
|
+
deleted: Array.isArray(result.deleted) ? result.deleted : [],
|
|
1981
|
+
newFacts: Array.isArray(result.newFacts) ? result.newFacts : []
|
|
1982
|
+
};
|
|
1983
|
+
},
|
|
1984
|
+
maxOutputTokens: this.options.llmProvider.maxOutputTokens,
|
|
1985
|
+
maxPromptChars: HEAL_MAX_PROMPT_CHARS,
|
|
1986
|
+
onSkip: (fact, err) => {
|
|
1987
|
+
console.warn(
|
|
1988
|
+
`[WikiMemory] heal skipped ${entityId}/${fact.id}: response could not be bounded`,
|
|
1989
|
+
err
|
|
1990
|
+
);
|
|
1991
|
+
}
|
|
1822
1992
|
});
|
|
1823
|
-
const
|
|
1824
|
-
|
|
1825
|
-
|
|
1826
|
-
|
|
1827
|
-
|
|
1828
|
-
|
|
1829
|
-
|
|
1830
|
-
|
|
1831
|
-
|
|
1832
|
-
const
|
|
1833
|
-
const
|
|
1834
|
-
const deleted = Array.isArray(result.deleted) ? result.deleted : [];
|
|
1835
|
-
const newFacts = Array.isArray(result.newFacts) ? result.newFacts : [];
|
|
1836
|
-
const safeDowngraded = Array.from(new Set(downgraded.filter((id) => mutableIds.has(id))));
|
|
1837
|
-
const safeDeleted = Array.from(new Set(deleted.filter((id) => mutableIds.has(id))));
|
|
1993
|
+
const safeDowngradedSet = /* @__PURE__ */ new Set();
|
|
1994
|
+
const safeDeletedSet = /* @__PURE__ */ new Set();
|
|
1995
|
+
const newFacts = [];
|
|
1996
|
+
for (const batchResult of outcome.results) {
|
|
1997
|
+
const mutableIds = new Set(batchResult.batch.map((f) => f.id));
|
|
1998
|
+
for (const id of batchResult.downgraded) if (mutableIds.has(id)) safeDowngradedSet.add(id);
|
|
1999
|
+
for (const id of batchResult.deleted) if (mutableIds.has(id)) safeDeletedSet.add(id);
|
|
2000
|
+
newFacts.push(...batchResult.newFacts);
|
|
2001
|
+
}
|
|
2002
|
+
const safeDowngraded = Array.from(safeDowngradedSet);
|
|
2003
|
+
const safeDeleted = Array.from(safeDeletedSet);
|
|
1838
2004
|
const validNewFacts = newFacts.map(validateFact).filter((f) => f !== null);
|
|
1839
2005
|
const insertedFacts = [];
|
|
1840
2006
|
const uniqueDeletedFactIds = Array.from(new Set(safeDeleted));
|
|
@@ -1901,7 +2067,7 @@ var MaintenanceService = class {
|
|
|
1901
2067
|
}
|
|
1902
2068
|
const now = Date.now();
|
|
1903
2069
|
const recheckCutoff = now - ONTOLOGY_BACKFILL_RECHECK_MS;
|
|
1904
|
-
const zeroed = { scanned: 0, typed: 0, failedValidation: 0, edgesAdded: 0 };
|
|
2070
|
+
const zeroed = { scanned: 0, typed: 0, failedValidation: 0, edgesAdded: 0, skipped: 0 };
|
|
1905
2071
|
const ontologyService = this.ontologyService;
|
|
1906
2072
|
if (!ontologyService) {
|
|
1907
2073
|
return { ...zeroed, remaining: 0, deferred: 0 };
|
|
@@ -1923,18 +2089,79 @@ var MaintenanceService = class {
|
|
|
1923
2089
|
options?.promptOverride,
|
|
1924
2090
|
ontologyContext
|
|
1925
2091
|
);
|
|
1926
|
-
const
|
|
1927
|
-
|
|
1928
|
-
|
|
1929
|
-
|
|
1930
|
-
|
|
1931
|
-
|
|
1932
|
-
|
|
1933
|
-
|
|
1934
|
-
|
|
1935
|
-
|
|
1936
|
-
|
|
1937
|
-
|
|
2092
|
+
const outcome = await runBatched({
|
|
2093
|
+
items: candidates,
|
|
2094
|
+
buildPrompt,
|
|
2095
|
+
call: (prompts) => this.options.llmProvider.generateText(prompts),
|
|
2096
|
+
parse: (responseText, batch) => {
|
|
2097
|
+
const parsed = parseJsonResponse(responseText);
|
|
2098
|
+
return {
|
|
2099
|
+
batch,
|
|
2100
|
+
classifications: Array.isArray(parsed.classifications) ? parsed.classifications : [],
|
|
2101
|
+
ontologyUpdates: parsed.ontology_updates
|
|
2102
|
+
};
|
|
2103
|
+
},
|
|
2104
|
+
maxOutputTokens: this.options.llmProvider.maxOutputTokens,
|
|
2105
|
+
maxPromptChars: ONTOLOGY_BACKFILL_MAX_PROMPT_CHARS,
|
|
2106
|
+
onSkip: (fact, err) => {
|
|
2107
|
+
console.warn(
|
|
2108
|
+
`[WikiMemory] ontology backfill skipped ${entityId}/${fact.id}: response could not be bounded`,
|
|
2109
|
+
err
|
|
2110
|
+
);
|
|
2111
|
+
}
|
|
2112
|
+
});
|
|
2113
|
+
let typed = 0;
|
|
2114
|
+
let failedValidation = 0;
|
|
2115
|
+
let edgesAdded = 0;
|
|
2116
|
+
let scanned = 0;
|
|
2117
|
+
let abortedOntologyOff = false;
|
|
2118
|
+
for (const batchResult of outcome.results) {
|
|
2119
|
+
const applied = await this._applyOntologyBackfillBatch(entityId, batchResult, now);
|
|
2120
|
+
if (applied.abortedOntologyOff) {
|
|
2121
|
+
abortedOntologyOff = true;
|
|
2122
|
+
break;
|
|
2123
|
+
}
|
|
2124
|
+
typed += applied.typed;
|
|
2125
|
+
failedValidation += applied.failedValidation;
|
|
2126
|
+
edgesAdded += applied.edgesAdded;
|
|
2127
|
+
scanned += batchResult.batch.length;
|
|
2128
|
+
}
|
|
2129
|
+
if (abortedOntologyOff) {
|
|
2130
|
+
const counts2 = await this.entryRepo.countUntypedByEntityId(entityId, recheckCutoff);
|
|
2131
|
+
return {
|
|
2132
|
+
scanned,
|
|
2133
|
+
typed,
|
|
2134
|
+
failedValidation,
|
|
2135
|
+
edgesAdded,
|
|
2136
|
+
skipped: outcome.skipped.length,
|
|
2137
|
+
remaining: 0,
|
|
2138
|
+
deferred: counts2.deferred
|
|
2139
|
+
};
|
|
2140
|
+
}
|
|
2141
|
+
if (outcome.skipped.length > 0) {
|
|
2142
|
+
await this.entryRepo.markOntologyChecked(outcome.skipped.map((f) => f.id), entityId, now, this.db);
|
|
2143
|
+
}
|
|
2144
|
+
this.searchService.evictCache(entityId);
|
|
2145
|
+
const counts = await this.entryRepo.countUntypedByEntityId(entityId, recheckCutoff);
|
|
2146
|
+
return {
|
|
2147
|
+
scanned,
|
|
2148
|
+
typed,
|
|
2149
|
+
failedValidation,
|
|
2150
|
+
edgesAdded,
|
|
2151
|
+
skipped: outcome.skipped.length,
|
|
2152
|
+
remaining: counts.eligible,
|
|
2153
|
+
deferred: counts.deferred
|
|
2154
|
+
};
|
|
2155
|
+
}
|
|
2156
|
+
/**
|
|
2157
|
+
* Applies one parsed backfill batch in its own transaction. Per-batch rather
|
|
2158
|
+
* than one transaction for the pass, so mergeEmergentUpdates semantics and
|
|
2159
|
+
* the mid-flight `mode === 'off'` abort check keep the shape they had when a
|
|
2160
|
+
* pass was a single call.
|
|
2161
|
+
*/
|
|
2162
|
+
async _applyOntologyBackfillBatch(entityId, batchResult, now) {
|
|
2163
|
+
const ontologyService = this.ontologyService;
|
|
2164
|
+
const { batch, classifications, ontologyUpdates } = batchResult;
|
|
1938
2165
|
let typed = 0;
|
|
1939
2166
|
let failedValidation = 0;
|
|
1940
2167
|
let edgesAdded = 0;
|
|
@@ -1945,8 +2172,8 @@ var MaintenanceService = class {
|
|
|
1945
2172
|
abortedOntologyOff = true;
|
|
1946
2173
|
return;
|
|
1947
2174
|
}
|
|
1948
|
-
if (txMode === "emergent" &&
|
|
1949
|
-
manifest = await ontologyService.mergeEmergentUpdates(entityId,
|
|
2175
|
+
if (txMode === "emergent" && ontologyUpdates) {
|
|
2176
|
+
manifest = await ontologyService.mergeEmergentUpdates(entityId, ontologyUpdates, tx);
|
|
1950
2177
|
}
|
|
1951
2178
|
const titleRows = await this.entryRepo.findTitleIndexByEntityId(entityId, tx);
|
|
1952
2179
|
const titleIndex = /* @__PURE__ */ new Map();
|
|
@@ -2000,19 +2227,58 @@ var MaintenanceService = class {
|
|
|
2000
2227
|
}
|
|
2001
2228
|
await this.entryRepo.markOntologyChecked(batch.map((f) => f.id), entityId, now, tx);
|
|
2002
2229
|
});
|
|
2003
|
-
|
|
2004
|
-
const counts2 = await this.entryRepo.countUntypedByEntityId(entityId, recheckCutoff);
|
|
2005
|
-
return { ...zeroed, remaining: 0, deferred: counts2.deferred };
|
|
2006
|
-
}
|
|
2007
|
-
this.searchService.evictCache(entityId);
|
|
2008
|
-
const counts = await this.entryRepo.countUntypedByEntityId(entityId, recheckCutoff);
|
|
2009
|
-
return { scanned: batch.length, typed, failedValidation, edgesAdded, remaining: counts.eligible, deferred: counts.deferred };
|
|
2230
|
+
return { typed, failedValidation, edgesAdded, abortedOntologyOff };
|
|
2010
2231
|
}
|
|
2011
2232
|
_validatePruneDuration(value, name) {
|
|
2012
2233
|
if (value !== null && value !== void 0 && (typeof value !== "number" || !isFinite(value) || value < 0)) {
|
|
2013
2234
|
throw new Error(`Invalid ${name}: must be a non-negative finite number or null`);
|
|
2014
2235
|
}
|
|
2015
2236
|
}
|
|
2237
|
+
/**
|
|
2238
|
+
* Anchors relevant to one batch of heal candidates.
|
|
2239
|
+
*
|
|
2240
|
+
* Heal used to pass every immutable_document fact for the entity — 2560 rows
|
|
2241
|
+
* against 31 candidates on the corpus behind #63 — which is what blew the
|
|
2242
|
+
* output ceiling. Anchors are now retrieved by keyword relevance to the batch
|
|
2243
|
+
* and capped.
|
|
2244
|
+
*
|
|
2245
|
+
* The MiniSearch index holds all facts, not only anchors, so hits are
|
|
2246
|
+
* overfetched and the source_type restriction is applied after retrieval, in
|
|
2247
|
+
* SQL. Search rank order is preserved through the filter.
|
|
2248
|
+
*
|
|
2249
|
+
* Accepted tradeoff: an anchor that contradicts a candidate while sharing no
|
|
2250
|
+
* vocabulary with it is now missed. Exhaustive-but-broken traded for
|
|
2251
|
+
* relevance-scoped-and-working.
|
|
2252
|
+
*
|
|
2253
|
+
* `cache` is keyed by the derived query rather than by the batch, so two
|
|
2254
|
+
* batches that reduce to the same query share one lookup. Caller-owned and
|
|
2255
|
+
* per-pass — see the call site in doRunHeal.
|
|
2256
|
+
*/
|
|
2257
|
+
async _selectHealAnchors(entityId, batch, cache) {
|
|
2258
|
+
const query = batch.map((f) => f.title).join(" ").trim();
|
|
2259
|
+
if (!query) return [];
|
|
2260
|
+
const cached = cache?.get(query);
|
|
2261
|
+
if (cached) return cached;
|
|
2262
|
+
const hits = this.searchService.searchKeyword(
|
|
2263
|
+
query,
|
|
2264
|
+
[entityId],
|
|
2265
|
+
HEAL_MAX_ANCHORS * HEAL_ANCHOR_SEARCH_OVERFETCH
|
|
2266
|
+
);
|
|
2267
|
+
const hitIds = hits.map((h) => h.id);
|
|
2268
|
+
const anchors = [];
|
|
2269
|
+
if (hitIds.length > 0) {
|
|
2270
|
+
const rows = await this.entryRepo.findAnchorRowsByIds(entityId, hitIds);
|
|
2271
|
+
const byId = new Map(rows.map((r) => [r.id, r]));
|
|
2272
|
+
for (const id of hitIds) {
|
|
2273
|
+
const row = byId.get(id);
|
|
2274
|
+
if (!row) continue;
|
|
2275
|
+
anchors.push(row);
|
|
2276
|
+
if (anchors.length >= HEAL_MAX_ANCHORS) break;
|
|
2277
|
+
}
|
|
2278
|
+
}
|
|
2279
|
+
cache?.set(query, anchors);
|
|
2280
|
+
return anchors;
|
|
2281
|
+
}
|
|
2016
2282
|
_sanitizeRankerError(err) {
|
|
2017
2283
|
return sanitizeRankerError(err, this.options.sanitizeRankerErrors);
|
|
2018
2284
|
}
|
|
@@ -3222,5 +3488,5 @@ var WriteService = class {
|
|
|
3222
3488
|
};
|
|
3223
3489
|
|
|
3224
3490
|
export { BaseRepository, EmbeddingService, HOOK_TIMEOUT_MARKER, ImportExportService, IngestionService, JobManager, MaintenanceService, MetadataRepository, ONTOLOGY_BACKFILL_BATCH_SIZE, ONTOLOGY_BACKFILL_MAX_PROMPT_CHARS, ONTOLOGY_BACKFILL_RECHECK_MS, ONTOLOGY_BACKFILL_SYSTEM_PROMPT, PromptService, PrunePartialFailureError, RetrievalService, SearchService, WikiBusyError, WikiTransactionError, WriteService, __privateAdd, __privateGet, __privateSet, configureRandomSource, emptyManifest, entitySummaryMetaKey, extractSqliteCode, generateId, normalizeSourceHash, normalizeSourceRef, normalizeTitleKey, parseEmbedding, resolveEdgeDefinitions, resolveNodeType, validateInlineEdges, validateManifest };
|
|
3225
|
-
//# sourceMappingURL=chunk-
|
|
3226
|
-
//# sourceMappingURL=chunk-
|
|
3491
|
+
//# sourceMappingURL=chunk-MYZJLVX4.mjs.map
|
|
3492
|
+
//# sourceMappingURL=chunk-MYZJLVX4.mjs.map
|