@equationalapplications/core-llm-wiki 4.21.0 → 4.23.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +11 -6
- package/dist/{chunk-UYLYN5N4.mjs → chunk-MYZJLVX4.mjs} +412 -131
- package/dist/chunk-MYZJLVX4.mjs.map +1 -0
- package/dist/index.d.mts +5 -3
- package/dist/index.d.ts +5 -3
- package/dist/index.js +389 -64
- package/dist/index.js.map +1 -1
- package/dist/index.mjs +49 -6
- package/dist/index.mjs.map +1 -1
- package/dist/{testing-i91HR1TG.d.mts → testing-CBjAuTSl.d.mts} +73 -1
- package/dist/{testing-i91HR1TG.d.ts → testing-CBjAuTSl.d.ts} +73 -1
- package/dist/testing.d.mts +1 -1
- package/dist/testing.d.ts +1 -1
- package/dist/testing.js +309 -43
- package/dist/testing.js.map +1 -1
- package/dist/testing.mjs +1 -1
- package/package.json +6 -6
- package/dist/chunk-UYLYN5N4.mjs.map +0 -1
package/README.md
CHANGED
|
@@ -42,6 +42,7 @@ const wikiMemory = new WikiMemory(db, {
|
|
|
42
42
|
// Your LLM call for extracting facts, tasks
|
|
43
43
|
return 'Model output';
|
|
44
44
|
},
|
|
45
|
+
maxOutputTokens: 4096, // optional — lets runHeal/runOntologyBackfill size their first LLM call correctly instead of discovering the ceiling via a truncated response and retrying smaller
|
|
45
46
|
embed: async (text: string) => {
|
|
46
47
|
// Your embedding service (e.g., OpenAI, Cohere, local)
|
|
47
48
|
const response = await fetch('https://your-app.example.com/api/embed', {
|
|
@@ -550,18 +551,21 @@ have `okf_type = NULL` and no edges, so `traverseGraph` cannot reach them.
|
|
|
550
551
|
|
|
551
552
|
```ts
|
|
552
553
|
const result = await wiki.runOntologyBackfill(entityId);
|
|
553
|
-
// { scanned, typed, failedValidation, edgesAdded, remaining, deferred }
|
|
554
|
+
// { scanned, typed, failedValidation, edgesAdded, remaining, deferred, skipped }
|
|
554
555
|
```
|
|
555
556
|
|
|
556
557
|
- **When to call it:** you own the trigger — the library provides the operation,
|
|
557
558
|
not a scheduler (same as `runLibrarian`/`runHeal`). A good default is after
|
|
558
559
|
each sync/import completes. `WikiBusyError` under concurrency is safe to
|
|
559
560
|
swallow; the next trigger retries.
|
|
560
|
-
- **Cost model:** one LLM
|
|
561
|
-
one SELECT. Each run scans at most 25 facts (override via
|
|
562
|
-
`options.batchSize`), oldest first
|
|
563
|
-
|
|
564
|
-
|
|
561
|
+
- **Cost model:** one or more LLM calls only when eligible untyped facts exist;
|
|
562
|
+
otherwise one SELECT. Each run scans at most 25 facts (override via
|
|
563
|
+
`options.batchSize`), oldest first. Facts are sent to the LLM in bounded
|
|
564
|
+
sub-batches — sized from `llmProvider.maxOutputTokens` when supplied,
|
|
565
|
+
otherwise a conservative default — with the full serialized prompt
|
|
566
|
+
(system + user) capped at 40k chars. A sub-batch whose response is truncated
|
|
567
|
+
or fails to parse is halved and retried automatically; a single fact that
|
|
568
|
+
still fails alone is counted in `skipped` rather than aborting the run.
|
|
565
569
|
- **Convergence:** loop `while (result.remaining > 0)` to drain a backlog.
|
|
566
570
|
`remaining` counts only *eligible* untyped facts, so the loop terminates even
|
|
567
571
|
when unclassifiable facts exist; those are cooldown-stamped and retried after
|
|
@@ -951,6 +955,7 @@ The flowchart shows:
|
|
|
951
955
|
| [@equationalapplications/prisma-outbox](https://github.com/equationalapplications/expo-llm-wiki/blob/main/packages/prisma-outbox/README.md) | Sync SQLite outbox events to Prisma |
|
|
952
956
|
| [@equationalapplications/core-llm-tools](https://github.com/equationalapplications/expo-llm-wiki/blob/main/packages/core-llm-tools/README.md) | Gemini tool schemas and capability injector |
|
|
953
957
|
| [@equationalapplications/core-okf](https://github.com/equationalapplications/expo-llm-wiki/blob/main/packages/okf/README.md) | Zero-dependency Open Knowledge Format (OKF) v0.1 primitives — parse and produce interoperable knowledge bundles. |
|
|
958
|
+
| [@equationalapplications/schema-org-llm-wiki](https://github.com/equationalapplications/expo-llm-wiki/blob/main/packages/schema-org/README.md) | Curated schema.org warm-agent ontology manifest |
|
|
954
959
|
|
|
955
960
|
## License
|
|
956
961
|
|
|
@@ -109,6 +109,106 @@ function generateId(prefix = "") {
|
|
|
109
109
|
);
|
|
110
110
|
}
|
|
111
111
|
|
|
112
|
+
// src/utils/ontology.ts
|
|
113
|
+
function emptyManifest() {
|
|
114
|
+
return { node_types: [], edge_types: [] };
|
|
115
|
+
}
|
|
116
|
+
function normalizeTitleKey(title) {
|
|
117
|
+
return title.trim().toLowerCase().replace(/\s+/g, " ");
|
|
118
|
+
}
|
|
119
|
+
function resolveNodeType(raw, manifest) {
|
|
120
|
+
const slug = raw.trim();
|
|
121
|
+
if (!slug) return null;
|
|
122
|
+
const hit = manifest.node_types.find((n) => n.type.toLowerCase() === slug.toLowerCase());
|
|
123
|
+
return hit?.type ?? null;
|
|
124
|
+
}
|
|
125
|
+
function resolveEdgeDefinitions(rawEdgeType, manifest) {
|
|
126
|
+
const slug = rawEdgeType.trim();
|
|
127
|
+
if (!slug) return [];
|
|
128
|
+
return manifest.edge_types.filter((e) => e.type.toLowerCase() === slug.toLowerCase());
|
|
129
|
+
}
|
|
130
|
+
function edgeTripleKey(type, sourceType, targetType) {
|
|
131
|
+
return `${type.trim().toLowerCase()}|${sourceType.trim().toLowerCase()}|${targetType.trim().toLowerCase()}`;
|
|
132
|
+
}
|
|
133
|
+
function validateManifest(manifest) {
|
|
134
|
+
const nodeSlugs = /* @__PURE__ */ new Set();
|
|
135
|
+
for (const node of manifest.node_types ?? []) {
|
|
136
|
+
const type = node.type?.trim();
|
|
137
|
+
if (!type) throw new Error("Ontology node type slug must be non-empty");
|
|
138
|
+
const key = type.toLowerCase();
|
|
139
|
+
if (nodeSlugs.has(key)) throw new Error(`Duplicate node type: ${type}`);
|
|
140
|
+
nodeSlugs.add(key);
|
|
141
|
+
}
|
|
142
|
+
const edgeKeys = /* @__PURE__ */ new Set();
|
|
143
|
+
const edgeNames = /* @__PURE__ */ new Map();
|
|
144
|
+
for (const edge of manifest.edge_types ?? []) {
|
|
145
|
+
const edgeType = edge.type?.trim();
|
|
146
|
+
const sourceType = edge.source_type?.trim();
|
|
147
|
+
const targetType = edge.target_type?.trim();
|
|
148
|
+
if (!edgeType) throw new Error("Ontology edge type slug must be non-empty");
|
|
149
|
+
if (!sourceType || !targetType || !nodeSlugs.has(sourceType.toLowerCase()) || !nodeSlugs.has(targetType.toLowerCase())) {
|
|
150
|
+
throw new Error(`Edge type ${edgeType} references unknown node type`);
|
|
151
|
+
}
|
|
152
|
+
const edgeKey = edgeTripleKey(edgeType, sourceType, targetType);
|
|
153
|
+
if (edgeKeys.has(edgeKey)) {
|
|
154
|
+
throw new Error(`Duplicate edge definition: ${edgeType} (${sourceType} \u2192 ${targetType})`);
|
|
155
|
+
}
|
|
156
|
+
edgeKeys.add(edgeKey);
|
|
157
|
+
const canonical = edgeNames.get(edgeType.toLowerCase());
|
|
158
|
+
if (canonical === void 0) {
|
|
159
|
+
edgeNames.set(edgeType.toLowerCase(), edgeType);
|
|
160
|
+
} else if (canonical !== edgeType) {
|
|
161
|
+
throw new Error(`Inconsistent casing for edge type: ${edgeType} conflicts with ${canonical}`);
|
|
162
|
+
}
|
|
163
|
+
}
|
|
164
|
+
}
|
|
165
|
+
function mergeOntologyUpdates(current, updates) {
|
|
166
|
+
const node_types = [...current.node_types];
|
|
167
|
+
const edge_types = [...current.edge_types];
|
|
168
|
+
const nodeSlugs = new Set(node_types.map((n) => n.type.trim().toLowerCase()));
|
|
169
|
+
const edgeKeys = new Set(edge_types.map((e) => edgeTripleKey(e.type, e.source_type, e.target_type)));
|
|
170
|
+
const edgeNames = new Map(edge_types.map((e) => [e.type.trim().toLowerCase(), e.type.trim()]));
|
|
171
|
+
for (const node of updates.node_types ?? []) {
|
|
172
|
+
const type = node?.type?.trim();
|
|
173
|
+
if (!type) continue;
|
|
174
|
+
const key = type.toLowerCase();
|
|
175
|
+
if (nodeSlugs.has(key)) continue;
|
|
176
|
+
node_types.push({ type, description: String(node.description ?? "") });
|
|
177
|
+
nodeSlugs.add(key);
|
|
178
|
+
}
|
|
179
|
+
for (const edge of updates.edge_types ?? []) {
|
|
180
|
+
const rawEdgeType = edge?.type?.trim();
|
|
181
|
+
const sourceType = edge?.source_type?.trim();
|
|
182
|
+
const targetType = edge?.target_type?.trim();
|
|
183
|
+
if (!rawEdgeType || !sourceType || !targetType) continue;
|
|
184
|
+
const edgeType = edgeNames.get(rawEdgeType.toLowerCase()) ?? rawEdgeType;
|
|
185
|
+
const edgeKey = edgeTripleKey(edgeType, sourceType, targetType);
|
|
186
|
+
if (edgeKeys.has(edgeKey)) continue;
|
|
187
|
+
if (!nodeSlugs.has(sourceType.toLowerCase()) || !nodeSlugs.has(targetType.toLowerCase())) continue;
|
|
188
|
+
edgeNames.set(edgeType.toLowerCase(), edgeType);
|
|
189
|
+
edge_types.push({
|
|
190
|
+
type: edgeType,
|
|
191
|
+
source_type: sourceType,
|
|
192
|
+
target_type: targetType,
|
|
193
|
+
description: String(edge.description ?? "")
|
|
194
|
+
});
|
|
195
|
+
edgeKeys.add(edgeKey);
|
|
196
|
+
}
|
|
197
|
+
return { node_types, edge_types };
|
|
198
|
+
}
|
|
199
|
+
function validateInlineEdges(sourceType, _targetType, edges, manifest) {
|
|
200
|
+
if (!Array.isArray(edges)) return [];
|
|
201
|
+
const valid = [];
|
|
202
|
+
for (const edge of edges) {
|
|
203
|
+
if (typeof edge?.edge_type !== "string" || typeof edge?.target_title !== "string") continue;
|
|
204
|
+
const defs = resolveEdgeDefinitions(edge.edge_type, manifest);
|
|
205
|
+
const match = defs.find((d) => d.source_type.toLowerCase() === sourceType.toLowerCase());
|
|
206
|
+
if (!match) continue;
|
|
207
|
+
valid.push({ edge_type: match.type, target_title: edge.target_title });
|
|
208
|
+
}
|
|
209
|
+
return valid;
|
|
210
|
+
}
|
|
211
|
+
|
|
112
212
|
// src/utils/embedding.ts
|
|
113
213
|
function parseEmbedding(blob, text) {
|
|
114
214
|
if (blob && blob.byteLength > 0) {
|
|
@@ -156,9 +256,23 @@ var _SearchService = class _SearchService {
|
|
|
156
256
|
this.entryRepo = entryRepo;
|
|
157
257
|
this.miniSearchEntryIdsByEntity = /* @__PURE__ */ new Map();
|
|
158
258
|
this.vectorCache = /* @__PURE__ */ new Map();
|
|
259
|
+
/**
|
|
260
|
+
* Serializes rebuilds. `rebuildIndex` awaits a repository read between
|
|
261
|
+
* snapshotting the previous id set and discarding it, so two concurrent
|
|
262
|
+
* sync() calls for one entity can interleave: a slow, stale read lands last
|
|
263
|
+
* and discards documents the fresh read just added. Chaining also keeps
|
|
264
|
+
* discard()/addAll() out of each other's way, which is what accrued the
|
|
265
|
+
* auto-vacuum debt behind the TypeError in #64.
|
|
266
|
+
*/
|
|
267
|
+
this.syncChain = Promise.resolve();
|
|
159
268
|
this.miniSearch = new MiniSearch({
|
|
160
269
|
fields: ["title", "body", "tags"],
|
|
161
270
|
storeFields: ["entity_id"],
|
|
271
|
+
// Vacuuming is driven explicitly at the end of each serialized rebuild
|
|
272
|
+
// (see sync). Auto-vacuum fires on its own schedule, asynchronously with
|
|
273
|
+
// respect to the caller, and traversing the tree mid-rebuild is what
|
|
274
|
+
// threw the uncaught TypeError in MiniSearch.performVacuuming (#64).
|
|
275
|
+
autoVacuum: false,
|
|
162
276
|
searchOptions: {
|
|
163
277
|
boost: { title: 2 },
|
|
164
278
|
fuzzy: 0.2,
|
|
@@ -169,10 +283,26 @@ var _SearchService = class _SearchService {
|
|
|
169
283
|
/**
|
|
170
284
|
* Rebuilds the search index and clears the vector cache for a given entity.
|
|
171
285
|
* A direct replacement for manually syncing state after a DB transaction.
|
|
286
|
+
*
|
|
287
|
+
* Rebuilds are serialized per instance and never reject: the MiniSearch index
|
|
288
|
+
* is a rebuildable cache over SQLite, so degraded keyword search is the
|
|
289
|
+
* correct failure mode and killing the host process is not.
|
|
172
290
|
*/
|
|
173
291
|
async sync(entityId) {
|
|
174
|
-
|
|
175
|
-
|
|
292
|
+
const work = this.syncChain.then(async () => {
|
|
293
|
+
try {
|
|
294
|
+
try {
|
|
295
|
+
await this.rebuildIndex(entityId);
|
|
296
|
+
await this.miniSearch.vacuum();
|
|
297
|
+
} finally {
|
|
298
|
+
this.evictCache(entityId);
|
|
299
|
+
}
|
|
300
|
+
} catch (err) {
|
|
301
|
+
console.warn(`[WikiMemory] search index rebuild failed for ${entityId ?? "*"}:`, err);
|
|
302
|
+
}
|
|
303
|
+
});
|
|
304
|
+
this.syncChain = work;
|
|
305
|
+
return work;
|
|
176
306
|
}
|
|
177
307
|
/**
|
|
178
308
|
* Clears the parsed vector cache. Useful for mid-loop flush guarantees
|
|
@@ -1012,91 +1142,6 @@ function jaccardScore(a, b) {
|
|
|
1012
1142
|
return intersection.size / union.size;
|
|
1013
1143
|
}
|
|
1014
1144
|
|
|
1015
|
-
// src/utils/ontology.ts
|
|
1016
|
-
function emptyManifest() {
|
|
1017
|
-
return { node_types: [], edge_types: [] };
|
|
1018
|
-
}
|
|
1019
|
-
function normalizeTitleKey(title) {
|
|
1020
|
-
return title.trim().toLowerCase().replace(/\s+/g, " ");
|
|
1021
|
-
}
|
|
1022
|
-
function resolveNodeType(raw, manifest) {
|
|
1023
|
-
const slug = raw.trim();
|
|
1024
|
-
if (!slug) return null;
|
|
1025
|
-
const hit = manifest.node_types.find((n) => n.type.toLowerCase() === slug.toLowerCase());
|
|
1026
|
-
return hit?.type ?? null;
|
|
1027
|
-
}
|
|
1028
|
-
function resolveEdgeDefinition(rawEdgeType, manifest) {
|
|
1029
|
-
const slug = rawEdgeType.trim();
|
|
1030
|
-
if (!slug) return null;
|
|
1031
|
-
return manifest.edge_types.find((e) => e.type.toLowerCase() === slug.toLowerCase()) ?? null;
|
|
1032
|
-
}
|
|
1033
|
-
function validateManifest(manifest) {
|
|
1034
|
-
const nodeSlugs = /* @__PURE__ */ new Set();
|
|
1035
|
-
for (const node of manifest.node_types ?? []) {
|
|
1036
|
-
const type = node.type?.trim();
|
|
1037
|
-
if (!type) throw new Error("Ontology node type slug must be non-empty");
|
|
1038
|
-
const key = type.toLowerCase();
|
|
1039
|
-
if (nodeSlugs.has(key)) throw new Error(`Duplicate node type: ${type}`);
|
|
1040
|
-
nodeSlugs.add(key);
|
|
1041
|
-
}
|
|
1042
|
-
const edgeSlugs = /* @__PURE__ */ new Set();
|
|
1043
|
-
for (const edge of manifest.edge_types ?? []) {
|
|
1044
|
-
const edgeType = edge.type?.trim();
|
|
1045
|
-
const sourceType = edge.source_type?.trim();
|
|
1046
|
-
const targetType = edge.target_type?.trim();
|
|
1047
|
-
if (!edgeType) throw new Error("Ontology edge type slug must be non-empty");
|
|
1048
|
-
const edgeKey = edgeType.toLowerCase();
|
|
1049
|
-
if (edgeSlugs.has(edgeKey)) throw new Error(`Duplicate edge type: ${edgeType}`);
|
|
1050
|
-
edgeSlugs.add(edgeKey);
|
|
1051
|
-
if (!sourceType || !targetType || !nodeSlugs.has(sourceType.toLowerCase()) || !nodeSlugs.has(targetType.toLowerCase())) {
|
|
1052
|
-
throw new Error(`Edge type ${edgeType} references unknown node type`);
|
|
1053
|
-
}
|
|
1054
|
-
}
|
|
1055
|
-
}
|
|
1056
|
-
function mergeOntologyUpdates(current, updates) {
|
|
1057
|
-
const node_types = [...current.node_types];
|
|
1058
|
-
const edge_types = [...current.edge_types];
|
|
1059
|
-
const nodeSlugs = new Set(node_types.map((n) => n.type.trim().toLowerCase()));
|
|
1060
|
-
const edgeSlugs = new Set(edge_types.map((e) => e.type.trim().toLowerCase()));
|
|
1061
|
-
for (const node of updates.node_types ?? []) {
|
|
1062
|
-
const type = node?.type?.trim();
|
|
1063
|
-
if (!type) continue;
|
|
1064
|
-
const key = type.toLowerCase();
|
|
1065
|
-
if (nodeSlugs.has(key)) continue;
|
|
1066
|
-
node_types.push({ type, description: String(node.description ?? "") });
|
|
1067
|
-
nodeSlugs.add(key);
|
|
1068
|
-
}
|
|
1069
|
-
for (const edge of updates.edge_types ?? []) {
|
|
1070
|
-
const edgeType = edge?.type?.trim();
|
|
1071
|
-
const sourceType = edge?.source_type?.trim();
|
|
1072
|
-
const targetType = edge?.target_type?.trim();
|
|
1073
|
-
if (!edgeType || !sourceType || !targetType) continue;
|
|
1074
|
-
const edgeKey = edgeType.toLowerCase();
|
|
1075
|
-
if (edgeSlugs.has(edgeKey)) continue;
|
|
1076
|
-
if (!nodeSlugs.has(sourceType.toLowerCase()) || !nodeSlugs.has(targetType.toLowerCase())) continue;
|
|
1077
|
-
edge_types.push({
|
|
1078
|
-
type: edgeType,
|
|
1079
|
-
source_type: sourceType,
|
|
1080
|
-
target_type: targetType,
|
|
1081
|
-
description: String(edge.description ?? "")
|
|
1082
|
-
});
|
|
1083
|
-
edgeSlugs.add(edgeKey);
|
|
1084
|
-
}
|
|
1085
|
-
return { node_types, edge_types };
|
|
1086
|
-
}
|
|
1087
|
-
function validateInlineEdges(sourceType, _targetType, edges, manifest) {
|
|
1088
|
-
if (!Array.isArray(edges)) return [];
|
|
1089
|
-
const valid = [];
|
|
1090
|
-
for (const edge of edges) {
|
|
1091
|
-
if (typeof edge?.edge_type !== "string" || typeof edge?.target_title !== "string") continue;
|
|
1092
|
-
const def = resolveEdgeDefinition(edge.edge_type, manifest);
|
|
1093
|
-
if (!def) continue;
|
|
1094
|
-
if (def.source_type.toLowerCase() !== sourceType.toLowerCase()) continue;
|
|
1095
|
-
valid.push({ edge_type: def.type, target_title: edge.target_title });
|
|
1096
|
-
}
|
|
1097
|
-
return valid;
|
|
1098
|
-
}
|
|
1099
|
-
|
|
1100
1145
|
// src/services/IngestionService.ts
|
|
1101
1146
|
var IngestionService = class {
|
|
1102
1147
|
constructor(db, prefix, options, entryRepo, searchService, jobManager, embeddingService, promptService, ontologyService) {
|
|
@@ -1408,12 +1453,122 @@ var MetadataRepository = class extends BaseRepository {
|
|
|
1408
1453
|
}
|
|
1409
1454
|
};
|
|
1410
1455
|
|
|
1456
|
+
// src/services/BoundedLlmCall.ts
|
|
1457
|
+
var DEFAULT_BATCH_SIZE = 10;
|
|
1458
|
+
var ESTIMATED_OUTPUT_TOKENS_PER_ITEM = 150;
|
|
1459
|
+
var OUTPUT_BUDGET_FRACTION = 0.8;
|
|
1460
|
+
var TRUNCATION_PATTERNS = [
|
|
1461
|
+
/truncat/i,
|
|
1462
|
+
/token limit/i,
|
|
1463
|
+
/max(imum)?[ _-]?tokens?/i,
|
|
1464
|
+
/output limit/i,
|
|
1465
|
+
/length limit/i,
|
|
1466
|
+
/finish[_ ]?reason/i
|
|
1467
|
+
];
|
|
1468
|
+
var EXCEEDS_LIMIT_PATTERN = /exceed[a-z]*[^.]{0,40}\b(model|context)?[ _-]?limit/i;
|
|
1469
|
+
function isTruncationError(err) {
|
|
1470
|
+
const message = err instanceof Error ? err.message : String(err ?? "");
|
|
1471
|
+
if (EXCEEDS_LIMIT_PATTERN.test(message)) return false;
|
|
1472
|
+
return TRUNCATION_PATTERNS.some((pattern) => pattern.test(message));
|
|
1473
|
+
}
|
|
1474
|
+
function initialBatchSize(maxOutputTokens) {
|
|
1475
|
+
if (!maxOutputTokens || !Number.isFinite(maxOutputTokens) || maxOutputTokens <= 0) {
|
|
1476
|
+
return DEFAULT_BATCH_SIZE;
|
|
1477
|
+
}
|
|
1478
|
+
const estimate = Math.floor(
|
|
1479
|
+
maxOutputTokens * OUTPUT_BUDGET_FRACTION / ESTIMATED_OUTPUT_TOKENS_PER_ITEM
|
|
1480
|
+
);
|
|
1481
|
+
return Math.max(DEFAULT_BATCH_SIZE, estimate);
|
|
1482
|
+
}
|
|
1483
|
+
var promptLength = (prompts) => prompts.systemPrompt.length + prompts.userPrompt.length;
|
|
1484
|
+
async function runBatched(args) {
|
|
1485
|
+
const { items, buildPrompt, call, parse, maxPromptChars, maxOutputTokens, onSkip } = args;
|
|
1486
|
+
const results = [];
|
|
1487
|
+
const skipped = [];
|
|
1488
|
+
let batches = 0;
|
|
1489
|
+
let batchSize = initialBatchSize(maxOutputTokens);
|
|
1490
|
+
const trim = async (candidate) => {
|
|
1491
|
+
const whole = await buildPrompt(candidate);
|
|
1492
|
+
if (candidate.length <= 1 || promptLength(whole) <= maxPromptChars) {
|
|
1493
|
+
return { batch: candidate, prompts: whole };
|
|
1494
|
+
}
|
|
1495
|
+
let low = 2;
|
|
1496
|
+
let high = candidate.length - 1;
|
|
1497
|
+
let best;
|
|
1498
|
+
let bestPrompts;
|
|
1499
|
+
while (low <= high) {
|
|
1500
|
+
const mid = Math.floor((low + high) / 2);
|
|
1501
|
+
const batch = candidate.slice(0, mid);
|
|
1502
|
+
const prompts = await buildPrompt(batch);
|
|
1503
|
+
if (promptLength(prompts) <= maxPromptChars) {
|
|
1504
|
+
best = batch;
|
|
1505
|
+
bestPrompts = prompts;
|
|
1506
|
+
low = mid + 1;
|
|
1507
|
+
} else {
|
|
1508
|
+
high = mid - 1;
|
|
1509
|
+
}
|
|
1510
|
+
}
|
|
1511
|
+
if (best && bestPrompts) return { batch: best, prompts: bestPrompts };
|
|
1512
|
+
const single = candidate.slice(0, 1);
|
|
1513
|
+
return { batch: single, prompts: await buildPrompt(single) };
|
|
1514
|
+
};
|
|
1515
|
+
const onFailure = async (batch, err) => {
|
|
1516
|
+
if (batch.length <= 1) {
|
|
1517
|
+
if (batch.length === 1) {
|
|
1518
|
+
skipped.push(batch[0]);
|
|
1519
|
+
onSkip?.(batch[0], err);
|
|
1520
|
+
}
|
|
1521
|
+
return;
|
|
1522
|
+
}
|
|
1523
|
+
const mid = Math.ceil(batch.length / 2);
|
|
1524
|
+
if (mid < batchSize) batchSize = mid;
|
|
1525
|
+
let i = 0;
|
|
1526
|
+
while (i < batch.length) {
|
|
1527
|
+
const size = Math.min(batchSize, batch.length - i);
|
|
1528
|
+
const trimmed = await trim(batch.slice(i, i + size));
|
|
1529
|
+
await attempt(trimmed.batch, trimmed.prompts);
|
|
1530
|
+
i += trimmed.batch.length;
|
|
1531
|
+
}
|
|
1532
|
+
};
|
|
1533
|
+
const attempt = async (batch, prebuilt) => {
|
|
1534
|
+
if (batch.length === 0) return;
|
|
1535
|
+
const prompts = prebuilt ?? await buildPrompt(batch);
|
|
1536
|
+
batches++;
|
|
1537
|
+
let responseText;
|
|
1538
|
+
try {
|
|
1539
|
+
responseText = await call(prompts);
|
|
1540
|
+
} catch (err) {
|
|
1541
|
+
if (!isTruncationError(err)) throw err;
|
|
1542
|
+
await onFailure(batch, err);
|
|
1543
|
+
return;
|
|
1544
|
+
}
|
|
1545
|
+
let result;
|
|
1546
|
+
try {
|
|
1547
|
+
result = parse(responseText, batch);
|
|
1548
|
+
} catch (err) {
|
|
1549
|
+
await onFailure(batch, err);
|
|
1550
|
+
return;
|
|
1551
|
+
}
|
|
1552
|
+
results.push(result);
|
|
1553
|
+
};
|
|
1554
|
+
let index = 0;
|
|
1555
|
+
while (index < items.length) {
|
|
1556
|
+
const { batch, prompts } = await trim(items.slice(index, index + batchSize));
|
|
1557
|
+
index += batch.length;
|
|
1558
|
+
await attempt(batch, prompts);
|
|
1559
|
+
}
|
|
1560
|
+
return { results, skipped, batches };
|
|
1561
|
+
}
|
|
1562
|
+
|
|
1411
1563
|
// src/services/MaintenanceService.ts
|
|
1412
1564
|
var FUZZY_THRESHOLD = 0.5;
|
|
1413
1565
|
var MIN_TOKENS_TO_QUALIFY = 3;
|
|
1414
1566
|
var ONTOLOGY_BACKFILL_BATCH_SIZE = 25;
|
|
1415
1567
|
var ONTOLOGY_BACKFILL_MAX_PROMPT_CHARS = 4e4;
|
|
1416
1568
|
var ONTOLOGY_BACKFILL_RECHECK_MS = 7 * 24 * 60 * 60 * 1e3;
|
|
1569
|
+
var HEAL_MAX_ANCHORS = 50;
|
|
1570
|
+
var HEAL_ANCHOR_SEARCH_OVERFETCH = 4;
|
|
1571
|
+
var HEAL_MAX_PROMPT_CHARS = 4e4;
|
|
1417
1572
|
var MaintenanceService = class {
|
|
1418
1573
|
constructor(db, prefix, options, entryRepo, taskRepo, eventRepo, metadataRepo, searchService, jobManager, embeddingService, promptService, ontologyService) {
|
|
1419
1574
|
this.db = db;
|
|
@@ -1796,30 +1951,56 @@ var MaintenanceService = class {
|
|
|
1796
1951
|
console.warn(`[WikiMemory] onEmbeddingPersisted hook failed during heal orphan pass for ${factId}:`, hookErr);
|
|
1797
1952
|
}
|
|
1798
1953
|
}
|
|
1799
|
-
const
|
|
1954
|
+
const healCandidates = await this.entryRepo.findHealCandidatesByEntityId(entityId);
|
|
1800
1955
|
const allTasks = await this.taskRepo.findAllPending([entityId]);
|
|
1801
1956
|
const recentEvents = await this.eventRepo.getRecent(entityId, 20);
|
|
1802
|
-
const
|
|
1803
|
-
const documentAnchors = allFactsRows.filter((f) => f.source_type === "immutable_document").map(({ id, title, source_ref }) => ({ id, title, source_ref }));
|
|
1804
|
-
const healCandidatesForPrompt = healCandidates.map((f) => {
|
|
1957
|
+
const toPromptShape = (f) => {
|
|
1805
1958
|
const { embedding: _embedding, embedding_blob: _blob, ...rest } = f;
|
|
1806
1959
|
return { ...rest, tags: typeof rest.tags === "string" ? JSON.parse(rest.tags) : rest.tags };
|
|
1960
|
+
};
|
|
1961
|
+
const anchorCache = /* @__PURE__ */ new Map();
|
|
1962
|
+
const outcome = await runBatched({
|
|
1963
|
+
items: healCandidates,
|
|
1964
|
+
buildPrompt: async (batch) => {
|
|
1965
|
+
const documentAnchors = await this._selectHealAnchors(entityId, batch, anchorCache);
|
|
1966
|
+
return this.promptService.buildHealPrompt(
|
|
1967
|
+
batch.map(toPromptShape),
|
|
1968
|
+
documentAnchors,
|
|
1969
|
+
allTasks,
|
|
1970
|
+
recentEvents,
|
|
1971
|
+
promptOverride
|
|
1972
|
+
);
|
|
1973
|
+
},
|
|
1974
|
+
call: (prompts) => this.options.llmProvider.generateText(prompts),
|
|
1975
|
+
parse: (responseText, batch) => {
|
|
1976
|
+
const result = parseJsonResponse(responseText);
|
|
1977
|
+
return {
|
|
1978
|
+
batch,
|
|
1979
|
+
downgraded: Array.isArray(result.downgraded) ? result.downgraded : [],
|
|
1980
|
+
deleted: Array.isArray(result.deleted) ? result.deleted : [],
|
|
1981
|
+
newFacts: Array.isArray(result.newFacts) ? result.newFacts : []
|
|
1982
|
+
};
|
|
1983
|
+
},
|
|
1984
|
+
maxOutputTokens: this.options.llmProvider.maxOutputTokens,
|
|
1985
|
+
maxPromptChars: HEAL_MAX_PROMPT_CHARS,
|
|
1986
|
+
onSkip: (fact, err) => {
|
|
1987
|
+
console.warn(
|
|
1988
|
+
`[WikiMemory] heal skipped ${entityId}/${fact.id}: response could not be bounded`,
|
|
1989
|
+
err
|
|
1990
|
+
);
|
|
1991
|
+
}
|
|
1807
1992
|
});
|
|
1808
|
-
const
|
|
1809
|
-
|
|
1810
|
-
|
|
1811
|
-
|
|
1812
|
-
|
|
1813
|
-
|
|
1814
|
-
|
|
1815
|
-
|
|
1816
|
-
|
|
1817
|
-
const
|
|
1818
|
-
const
|
|
1819
|
-
const deleted = Array.isArray(result.deleted) ? result.deleted : [];
|
|
1820
|
-
const newFacts = Array.isArray(result.newFacts) ? result.newFacts : [];
|
|
1821
|
-
const safeDowngraded = Array.from(new Set(downgraded.filter((id) => mutableIds.has(id))));
|
|
1822
|
-
const safeDeleted = Array.from(new Set(deleted.filter((id) => mutableIds.has(id))));
|
|
1993
|
+
const safeDowngradedSet = /* @__PURE__ */ new Set();
|
|
1994
|
+
const safeDeletedSet = /* @__PURE__ */ new Set();
|
|
1995
|
+
const newFacts = [];
|
|
1996
|
+
for (const batchResult of outcome.results) {
|
|
1997
|
+
const mutableIds = new Set(batchResult.batch.map((f) => f.id));
|
|
1998
|
+
for (const id of batchResult.downgraded) if (mutableIds.has(id)) safeDowngradedSet.add(id);
|
|
1999
|
+
for (const id of batchResult.deleted) if (mutableIds.has(id)) safeDeletedSet.add(id);
|
|
2000
|
+
newFacts.push(...batchResult.newFacts);
|
|
2001
|
+
}
|
|
2002
|
+
const safeDowngraded = Array.from(safeDowngradedSet);
|
|
2003
|
+
const safeDeleted = Array.from(safeDeletedSet);
|
|
1823
2004
|
const validNewFacts = newFacts.map(validateFact).filter((f) => f !== null);
|
|
1824
2005
|
const insertedFacts = [];
|
|
1825
2006
|
const uniqueDeletedFactIds = Array.from(new Set(safeDeleted));
|
|
@@ -1886,7 +2067,7 @@ var MaintenanceService = class {
|
|
|
1886
2067
|
}
|
|
1887
2068
|
const now = Date.now();
|
|
1888
2069
|
const recheckCutoff = now - ONTOLOGY_BACKFILL_RECHECK_MS;
|
|
1889
|
-
const zeroed = { scanned: 0, typed: 0, failedValidation: 0, edgesAdded: 0 };
|
|
2070
|
+
const zeroed = { scanned: 0, typed: 0, failedValidation: 0, edgesAdded: 0, skipped: 0 };
|
|
1890
2071
|
const ontologyService = this.ontologyService;
|
|
1891
2072
|
if (!ontologyService) {
|
|
1892
2073
|
return { ...zeroed, remaining: 0, deferred: 0 };
|
|
@@ -1908,18 +2089,79 @@ var MaintenanceService = class {
|
|
|
1908
2089
|
options?.promptOverride,
|
|
1909
2090
|
ontologyContext
|
|
1910
2091
|
);
|
|
1911
|
-
const
|
|
1912
|
-
|
|
1913
|
-
|
|
1914
|
-
|
|
1915
|
-
|
|
1916
|
-
|
|
1917
|
-
|
|
1918
|
-
|
|
1919
|
-
|
|
1920
|
-
|
|
1921
|
-
|
|
1922
|
-
|
|
2092
|
+
const outcome = await runBatched({
|
|
2093
|
+
items: candidates,
|
|
2094
|
+
buildPrompt,
|
|
2095
|
+
call: (prompts) => this.options.llmProvider.generateText(prompts),
|
|
2096
|
+
parse: (responseText, batch) => {
|
|
2097
|
+
const parsed = parseJsonResponse(responseText);
|
|
2098
|
+
return {
|
|
2099
|
+
batch,
|
|
2100
|
+
classifications: Array.isArray(parsed.classifications) ? parsed.classifications : [],
|
|
2101
|
+
ontologyUpdates: parsed.ontology_updates
|
|
2102
|
+
};
|
|
2103
|
+
},
|
|
2104
|
+
maxOutputTokens: this.options.llmProvider.maxOutputTokens,
|
|
2105
|
+
maxPromptChars: ONTOLOGY_BACKFILL_MAX_PROMPT_CHARS,
|
|
2106
|
+
onSkip: (fact, err) => {
|
|
2107
|
+
console.warn(
|
|
2108
|
+
`[WikiMemory] ontology backfill skipped ${entityId}/${fact.id}: response could not be bounded`,
|
|
2109
|
+
err
|
|
2110
|
+
);
|
|
2111
|
+
}
|
|
2112
|
+
});
|
|
2113
|
+
let typed = 0;
|
|
2114
|
+
let failedValidation = 0;
|
|
2115
|
+
let edgesAdded = 0;
|
|
2116
|
+
let scanned = 0;
|
|
2117
|
+
let abortedOntologyOff = false;
|
|
2118
|
+
for (const batchResult of outcome.results) {
|
|
2119
|
+
const applied = await this._applyOntologyBackfillBatch(entityId, batchResult, now);
|
|
2120
|
+
if (applied.abortedOntologyOff) {
|
|
2121
|
+
abortedOntologyOff = true;
|
|
2122
|
+
break;
|
|
2123
|
+
}
|
|
2124
|
+
typed += applied.typed;
|
|
2125
|
+
failedValidation += applied.failedValidation;
|
|
2126
|
+
edgesAdded += applied.edgesAdded;
|
|
2127
|
+
scanned += batchResult.batch.length;
|
|
2128
|
+
}
|
|
2129
|
+
if (abortedOntologyOff) {
|
|
2130
|
+
const counts2 = await this.entryRepo.countUntypedByEntityId(entityId, recheckCutoff);
|
|
2131
|
+
return {
|
|
2132
|
+
scanned,
|
|
2133
|
+
typed,
|
|
2134
|
+
failedValidation,
|
|
2135
|
+
edgesAdded,
|
|
2136
|
+
skipped: outcome.skipped.length,
|
|
2137
|
+
remaining: 0,
|
|
2138
|
+
deferred: counts2.deferred
|
|
2139
|
+
};
|
|
2140
|
+
}
|
|
2141
|
+
if (outcome.skipped.length > 0) {
|
|
2142
|
+
await this.entryRepo.markOntologyChecked(outcome.skipped.map((f) => f.id), entityId, now, this.db);
|
|
2143
|
+
}
|
|
2144
|
+
this.searchService.evictCache(entityId);
|
|
2145
|
+
const counts = await this.entryRepo.countUntypedByEntityId(entityId, recheckCutoff);
|
|
2146
|
+
return {
|
|
2147
|
+
scanned,
|
|
2148
|
+
typed,
|
|
2149
|
+
failedValidation,
|
|
2150
|
+
edgesAdded,
|
|
2151
|
+
skipped: outcome.skipped.length,
|
|
2152
|
+
remaining: counts.eligible,
|
|
2153
|
+
deferred: counts.deferred
|
|
2154
|
+
};
|
|
2155
|
+
}
|
|
2156
|
+
/**
|
|
2157
|
+
* Applies one parsed backfill batch in its own transaction. Per-batch rather
|
|
2158
|
+
* than one transaction for the pass, so mergeEmergentUpdates semantics and
|
|
2159
|
+
* the mid-flight `mode === 'off'` abort check keep the shape they had when a
|
|
2160
|
+
* pass was a single call.
|
|
2161
|
+
*/
|
|
2162
|
+
async _applyOntologyBackfillBatch(entityId, batchResult, now) {
|
|
2163
|
+
const ontologyService = this.ontologyService;
|
|
2164
|
+
const { batch, classifications, ontologyUpdates } = batchResult;
|
|
1923
2165
|
let typed = 0;
|
|
1924
2166
|
let failedValidation = 0;
|
|
1925
2167
|
let edgesAdded = 0;
|
|
@@ -1930,8 +2172,8 @@ var MaintenanceService = class {
|
|
|
1930
2172
|
abortedOntologyOff = true;
|
|
1931
2173
|
return;
|
|
1932
2174
|
}
|
|
1933
|
-
if (txMode === "emergent" &&
|
|
1934
|
-
manifest = await ontologyService.mergeEmergentUpdates(entityId,
|
|
2175
|
+
if (txMode === "emergent" && ontologyUpdates) {
|
|
2176
|
+
manifest = await ontologyService.mergeEmergentUpdates(entityId, ontologyUpdates, tx);
|
|
1935
2177
|
}
|
|
1936
2178
|
const titleRows = await this.entryRepo.findTitleIndexByEntityId(entityId, tx);
|
|
1937
2179
|
const titleIndex = /* @__PURE__ */ new Map();
|
|
@@ -1985,19 +2227,58 @@ var MaintenanceService = class {
|
|
|
1985
2227
|
}
|
|
1986
2228
|
await this.entryRepo.markOntologyChecked(batch.map((f) => f.id), entityId, now, tx);
|
|
1987
2229
|
});
|
|
1988
|
-
|
|
1989
|
-
const counts2 = await this.entryRepo.countUntypedByEntityId(entityId, recheckCutoff);
|
|
1990
|
-
return { ...zeroed, remaining: 0, deferred: counts2.deferred };
|
|
1991
|
-
}
|
|
1992
|
-
this.searchService.evictCache(entityId);
|
|
1993
|
-
const counts = await this.entryRepo.countUntypedByEntityId(entityId, recheckCutoff);
|
|
1994
|
-
return { scanned: batch.length, typed, failedValidation, edgesAdded, remaining: counts.eligible, deferred: counts.deferred };
|
|
2230
|
+
return { typed, failedValidation, edgesAdded, abortedOntologyOff };
|
|
1995
2231
|
}
|
|
1996
2232
|
_validatePruneDuration(value, name) {
|
|
1997
2233
|
if (value !== null && value !== void 0 && (typeof value !== "number" || !isFinite(value) || value < 0)) {
|
|
1998
2234
|
throw new Error(`Invalid ${name}: must be a non-negative finite number or null`);
|
|
1999
2235
|
}
|
|
2000
2236
|
}
|
|
2237
|
+
/**
|
|
2238
|
+
* Anchors relevant to one batch of heal candidates.
|
|
2239
|
+
*
|
|
2240
|
+
* Heal used to pass every immutable_document fact for the entity — 2560 rows
|
|
2241
|
+
* against 31 candidates on the corpus behind #63 — which is what blew the
|
|
2242
|
+
* output ceiling. Anchors are now retrieved by keyword relevance to the batch
|
|
2243
|
+
* and capped.
|
|
2244
|
+
*
|
|
2245
|
+
* The MiniSearch index holds all facts, not only anchors, so hits are
|
|
2246
|
+
* overfetched and the source_type restriction is applied after retrieval, in
|
|
2247
|
+
* SQL. Search rank order is preserved through the filter.
|
|
2248
|
+
*
|
|
2249
|
+
* Accepted tradeoff: an anchor that contradicts a candidate while sharing no
|
|
2250
|
+
* vocabulary with it is now missed. Exhaustive-but-broken traded for
|
|
2251
|
+
* relevance-scoped-and-working.
|
|
2252
|
+
*
|
|
2253
|
+
* `cache` is keyed by the derived query rather than by the batch, so two
|
|
2254
|
+
* batches that reduce to the same query share one lookup. Caller-owned and
|
|
2255
|
+
* per-pass — see the call site in doRunHeal.
|
|
2256
|
+
*/
|
|
2257
|
+
async _selectHealAnchors(entityId, batch, cache) {
|
|
2258
|
+
const query = batch.map((f) => f.title).join(" ").trim();
|
|
2259
|
+
if (!query) return [];
|
|
2260
|
+
const cached = cache?.get(query);
|
|
2261
|
+
if (cached) return cached;
|
|
2262
|
+
const hits = this.searchService.searchKeyword(
|
|
2263
|
+
query,
|
|
2264
|
+
[entityId],
|
|
2265
|
+
HEAL_MAX_ANCHORS * HEAL_ANCHOR_SEARCH_OVERFETCH
|
|
2266
|
+
);
|
|
2267
|
+
const hitIds = hits.map((h) => h.id);
|
|
2268
|
+
const anchors = [];
|
|
2269
|
+
if (hitIds.length > 0) {
|
|
2270
|
+
const rows = await this.entryRepo.findAnchorRowsByIds(entityId, hitIds);
|
|
2271
|
+
const byId = new Map(rows.map((r) => [r.id, r]));
|
|
2272
|
+
for (const id of hitIds) {
|
|
2273
|
+
const row = byId.get(id);
|
|
2274
|
+
if (!row) continue;
|
|
2275
|
+
anchors.push(row);
|
|
2276
|
+
if (anchors.length >= HEAL_MAX_ANCHORS) break;
|
|
2277
|
+
}
|
|
2278
|
+
}
|
|
2279
|
+
cache?.set(query, anchors);
|
|
2280
|
+
return anchors;
|
|
2281
|
+
}
|
|
2001
2282
|
_sanitizeRankerError(err) {
|
|
2002
2283
|
return sanitizeRankerError(err, this.options.sanitizeRankerErrors);
|
|
2003
2284
|
}
|
|
@@ -3206,6 +3487,6 @@ var WriteService = class {
|
|
|
3206
3487
|
}
|
|
3207
3488
|
};
|
|
3208
3489
|
|
|
3209
|
-
export { BaseRepository, EmbeddingService, HOOK_TIMEOUT_MARKER, ImportExportService, IngestionService, JobManager, MaintenanceService, MetadataRepository, ONTOLOGY_BACKFILL_BATCH_SIZE, ONTOLOGY_BACKFILL_MAX_PROMPT_CHARS, ONTOLOGY_BACKFILL_RECHECK_MS, ONTOLOGY_BACKFILL_SYSTEM_PROMPT, PromptService, PrunePartialFailureError, RetrievalService, SearchService, WikiBusyError, WikiTransactionError, WriteService, __privateAdd, __privateGet, __privateSet, configureRandomSource, emptyManifest, entitySummaryMetaKey, extractSqliteCode, generateId, normalizeSourceHash, normalizeSourceRef, normalizeTitleKey, parseEmbedding,
|
|
3210
|
-
//# sourceMappingURL=chunk-
|
|
3211
|
-
//# sourceMappingURL=chunk-
|
|
3490
|
+
export { BaseRepository, EmbeddingService, HOOK_TIMEOUT_MARKER, ImportExportService, IngestionService, JobManager, MaintenanceService, MetadataRepository, ONTOLOGY_BACKFILL_BATCH_SIZE, ONTOLOGY_BACKFILL_MAX_PROMPT_CHARS, ONTOLOGY_BACKFILL_RECHECK_MS, ONTOLOGY_BACKFILL_SYSTEM_PROMPT, PromptService, PrunePartialFailureError, RetrievalService, SearchService, WikiBusyError, WikiTransactionError, WriteService, __privateAdd, __privateGet, __privateSet, configureRandomSource, emptyManifest, entitySummaryMetaKey, extractSqliteCode, generateId, normalizeSourceHash, normalizeSourceRef, normalizeTitleKey, parseEmbedding, resolveEdgeDefinitions, resolveNodeType, validateInlineEdges, validateManifest };
|
|
3491
|
+
//# sourceMappingURL=chunk-MYZJLVX4.mjs.map
|
|
3492
|
+
//# sourceMappingURL=chunk-MYZJLVX4.mjs.map
|