@equationalapplications/core-llm-wiki 5.4.0 → 5.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +29 -0
- package/dist/{chunk-N2NXN6PO.mjs → chunk-QVY5DFJV.mjs} +535 -105
- package/dist/chunk-QVY5DFJV.mjs.map +1 -0
- package/dist/index.d.mts +3 -3
- package/dist/index.d.ts +3 -3
- package/dist/index.js +534 -102
- package/dist/index.js.map +1 -1
- package/dist/index.mjs +2 -2
- package/dist/index.mjs.map +1 -1
- package/dist/{testing-C71mnMxq.d.mts → testing-D0RnZjyW.d.mts} +108 -11
- package/dist/{testing-C71mnMxq.d.ts → testing-D0RnZjyW.d.ts} +108 -11
- package/dist/testing.d.mts +1 -1
- package/dist/testing.d.ts +1 -1
- package/dist/testing.js +532 -102
- package/dist/testing.js.map +1 -1
- package/dist/testing.mjs +1 -1
- package/package.json +2 -2
- package/dist/chunk-N2NXN6PO.mjs.map +0 -1
package/dist/testing.js
CHANGED
|
@@ -110,26 +110,88 @@ var PrunePartialFailureError = class extends Error {
|
|
|
110
110
|
}
|
|
111
111
|
};
|
|
112
112
|
var HOOK_TIMEOUT_MARKER = /* @__PURE__ */ Symbol("WikiMemoryHookTimeout");
|
|
113
|
+
var WikiParseError = class extends Error {
|
|
114
|
+
constructor(message, opts) {
|
|
115
|
+
super(message);
|
|
116
|
+
this.name = "WikiParseError";
|
|
117
|
+
this.tier = opts.tier;
|
|
118
|
+
this.position = opts.position ?? null;
|
|
119
|
+
this.slice = opts.slice ?? "";
|
|
120
|
+
}
|
|
121
|
+
};
|
|
122
|
+
var WikiIngestEmptyError = class extends Error {
|
|
123
|
+
constructor(params) {
|
|
124
|
+
const summary = `All ${params.chunks} chunks failed for sourceRef "${params.sourceRef}"; see parseFailures for per-chunk detail`;
|
|
125
|
+
super(summary);
|
|
126
|
+
this.name = "WikiIngestEmptyError";
|
|
127
|
+
this.parseFailures = params.parseFailures;
|
|
128
|
+
this.sourceRef = params.sourceRef;
|
|
129
|
+
this.chunks = params.chunks;
|
|
130
|
+
}
|
|
131
|
+
};
|
|
113
132
|
|
|
114
133
|
// src/utils/pure.ts
|
|
134
|
+
var MAX_REPAIR_CANDIDATES = 5;
|
|
135
|
+
function throwNoJsonFound(text, start) {
|
|
136
|
+
throw new WikiParseError(
|
|
137
|
+
"No JSON object/array found in LLM response",
|
|
138
|
+
{ tier: "strict", position: start, slice: text }
|
|
139
|
+
);
|
|
140
|
+
}
|
|
115
141
|
function parseJsonResponse(text) {
|
|
116
142
|
const firstBrace = text.indexOf("{");
|
|
117
143
|
const firstBracket = text.indexOf("[");
|
|
118
144
|
let start;
|
|
119
145
|
let openChar;
|
|
120
|
-
let closeChar;
|
|
121
146
|
const useBrace = firstBrace !== -1 && (firstBracket === -1 || firstBrace < firstBracket);
|
|
122
147
|
if (useBrace) {
|
|
123
148
|
start = firstBrace;
|
|
124
149
|
openChar = "{";
|
|
125
|
-
closeChar = "}";
|
|
126
150
|
} else if (firstBracket !== -1) {
|
|
127
151
|
start = firstBracket;
|
|
128
152
|
openChar = "[";
|
|
129
|
-
closeChar = "]";
|
|
130
153
|
} else {
|
|
131
|
-
|
|
154
|
+
throwNoJsonFound(text, null);
|
|
155
|
+
}
|
|
156
|
+
const slice = scanJsonSlice(text, start, openChar);
|
|
157
|
+
if (slice !== null) {
|
|
158
|
+
try {
|
|
159
|
+
return JSON.parse(slice);
|
|
160
|
+
} catch {
|
|
161
|
+
}
|
|
162
|
+
}
|
|
163
|
+
const repairOdd = containerAwareRepair(text, start, false);
|
|
164
|
+
if (repairOdd.success !== null && !repairOdd.ambiguous) {
|
|
165
|
+
return repairOdd.success;
|
|
166
|
+
}
|
|
167
|
+
const repairEven = containerAwareRepair(text, start, true);
|
|
168
|
+
if (repairOdd.success !== null && repairEven.success !== null) {
|
|
169
|
+
const pick = repairEven.escapes < repairOdd.escapes ? repairEven : repairOdd;
|
|
170
|
+
return pick.success;
|
|
171
|
+
}
|
|
172
|
+
if (repairOdd.success !== null) {
|
|
173
|
+
return repairOdd.success;
|
|
174
|
+
}
|
|
175
|
+
if (repairEven.success !== null) {
|
|
176
|
+
return repairEven.success;
|
|
132
177
|
}
|
|
178
|
+
const repair = repairEven.failed !== null ? repairEven : repairOdd;
|
|
179
|
+
if (repair.failed !== null) {
|
|
180
|
+
throw new WikiParseError(
|
|
181
|
+
`Repair produced a candidate but JSON.parse rejected it: ${repair.failed.message}`,
|
|
182
|
+
{ tier: "repair", position: repair.failed.position, slice: repair.failed.candidate }
|
|
183
|
+
);
|
|
184
|
+
}
|
|
185
|
+
if (slice === null) {
|
|
186
|
+
throwNoJsonFound(text, start);
|
|
187
|
+
}
|
|
188
|
+
throw new WikiParseError(
|
|
189
|
+
"No parsable JSON candidate found",
|
|
190
|
+
{ tier: "all", position: start, slice: text }
|
|
191
|
+
);
|
|
192
|
+
}
|
|
193
|
+
function scanJsonSlice(text, start, openChar) {
|
|
194
|
+
const closeChar = openChar === "{" ? "}" : "]";
|
|
133
195
|
let depth = 0;
|
|
134
196
|
let inString = false;
|
|
135
197
|
let escape = false;
|
|
@@ -161,8 +223,193 @@ function parseJsonResponse(text) {
|
|
|
161
223
|
}
|
|
162
224
|
}
|
|
163
225
|
}
|
|
164
|
-
if (end === -1)
|
|
165
|
-
return
|
|
226
|
+
if (end === -1) return null;
|
|
227
|
+
return safeSlice(text, start, end + 1);
|
|
228
|
+
}
|
|
229
|
+
function containerAwareRepair(text, start, closeOnEvenParity) {
|
|
230
|
+
const stack = [];
|
|
231
|
+
let inString = false;
|
|
232
|
+
let stringRole = null;
|
|
233
|
+
let escape = false;
|
|
234
|
+
let i = start;
|
|
235
|
+
let out = "";
|
|
236
|
+
let bareQuoteCount = 0;
|
|
237
|
+
let bestFailed = null;
|
|
238
|
+
let escapes = 0;
|
|
239
|
+
let ambiguous = false;
|
|
240
|
+
let bestSuccess = null;
|
|
241
|
+
let attempts = 0;
|
|
242
|
+
function tryEmitCandidate(candidate) {
|
|
243
|
+
attempts++;
|
|
244
|
+
try {
|
|
245
|
+
bestSuccess = JSON.parse(candidate);
|
|
246
|
+
return true;
|
|
247
|
+
} catch (err) {
|
|
248
|
+
if (bestFailed === null) {
|
|
249
|
+
bestFailed = {
|
|
250
|
+
candidate,
|
|
251
|
+
position: extractParsePosition(err),
|
|
252
|
+
message: safeErrorToString(err)
|
|
253
|
+
};
|
|
254
|
+
}
|
|
255
|
+
return false;
|
|
256
|
+
}
|
|
257
|
+
}
|
|
258
|
+
while (i < text.length) {
|
|
259
|
+
const ch = text[i];
|
|
260
|
+
if (stack.length === 0 && !inString) {
|
|
261
|
+
if (ch === "{" || ch === "[") {
|
|
262
|
+
stack.push({ container: ch === "{" ? "object" : "array", expectKey: true });
|
|
263
|
+
out += ch;
|
|
264
|
+
}
|
|
265
|
+
i++;
|
|
266
|
+
continue;
|
|
267
|
+
}
|
|
268
|
+
if (escape) {
|
|
269
|
+
out += ch;
|
|
270
|
+
escape = false;
|
|
271
|
+
i++;
|
|
272
|
+
continue;
|
|
273
|
+
}
|
|
274
|
+
if (inString && ch === "\\") {
|
|
275
|
+
out += ch;
|
|
276
|
+
escape = true;
|
|
277
|
+
i++;
|
|
278
|
+
continue;
|
|
279
|
+
}
|
|
280
|
+
if (inString) {
|
|
281
|
+
const code = ch.charCodeAt(0);
|
|
282
|
+
if (code < 32) {
|
|
283
|
+
out += controlCharEscape(code);
|
|
284
|
+
i++;
|
|
285
|
+
continue;
|
|
286
|
+
}
|
|
287
|
+
}
|
|
288
|
+
if (ch === '"') {
|
|
289
|
+
let j = i + 1;
|
|
290
|
+
while (j < text.length && (text[j] === " " || text[j] === " " || text[j] === "\n" || text[j] === "\r")) {
|
|
291
|
+
j++;
|
|
292
|
+
}
|
|
293
|
+
const next = j < text.length ? text[j] : "";
|
|
294
|
+
if (!inString) {
|
|
295
|
+
if (next === "}" || next === "]") {
|
|
296
|
+
if (j > i + 1) out += text.slice(i + 1, j);
|
|
297
|
+
out += next;
|
|
298
|
+
i = j + 1;
|
|
299
|
+
stack.pop();
|
|
300
|
+
if (stack.length === 0) {
|
|
301
|
+
if (tryEmitCandidate(out)) return { success: bestSuccess, failed: bestFailed, escapes, ambiguous };
|
|
302
|
+
if (attempts >= MAX_REPAIR_CANDIDATES) break;
|
|
303
|
+
out = "";
|
|
304
|
+
}
|
|
305
|
+
continue;
|
|
306
|
+
}
|
|
307
|
+
const top = stack[stack.length - 1];
|
|
308
|
+
stringRole = top?.container === "object" && top.expectKey ? "key" : "value";
|
|
309
|
+
out += ch;
|
|
310
|
+
inString = true;
|
|
311
|
+
i++;
|
|
312
|
+
continue;
|
|
313
|
+
}
|
|
314
|
+
const commaOnly = next === ",";
|
|
315
|
+
const isKeyClose = stringRole === "key" && (next === ":" || next === '"' && j + 1 < text.length && text[j + 1] === ":");
|
|
316
|
+
const isValueClose = !commaOnly && (next === "," || next === "}" || next === "]");
|
|
317
|
+
if (commaOnly && bareQuoteCount > 0) ambiguous = true;
|
|
318
|
+
const shouldCommaClose = commaOnly && (bareQuoteCount === 0 || (closeOnEvenParity ? bareQuoteCount % 2 === 0 : bareQuoteCount % 2 === 1));
|
|
319
|
+
const isClosing = isKeyClose || isValueClose || shouldCommaClose;
|
|
320
|
+
if (isClosing) {
|
|
321
|
+
out += ch;
|
|
322
|
+
inString = false;
|
|
323
|
+
const topFrame = stack[stack.length - 1];
|
|
324
|
+
if (topFrame && topFrame.container === "object") {
|
|
325
|
+
topFrame.expectKey = stringRole === "value";
|
|
326
|
+
}
|
|
327
|
+
stringRole = null;
|
|
328
|
+
bareQuoteCount = 0;
|
|
329
|
+
i++;
|
|
330
|
+
continue;
|
|
331
|
+
}
|
|
332
|
+
out += "\\" + ch;
|
|
333
|
+
bareQuoteCount++;
|
|
334
|
+
escapes++;
|
|
335
|
+
i++;
|
|
336
|
+
continue;
|
|
337
|
+
}
|
|
338
|
+
if (!inString && (ch === "{" || ch === "[")) {
|
|
339
|
+
stack.push({ container: ch === "{" ? "object" : "array", expectKey: true });
|
|
340
|
+
out += ch;
|
|
341
|
+
i++;
|
|
342
|
+
continue;
|
|
343
|
+
}
|
|
344
|
+
if (!inString && ch === ",") {
|
|
345
|
+
const topFrame = stack[stack.length - 1];
|
|
346
|
+
if (topFrame && topFrame.container === "object") {
|
|
347
|
+
topFrame.expectKey = true;
|
|
348
|
+
}
|
|
349
|
+
out += ch;
|
|
350
|
+
i++;
|
|
351
|
+
continue;
|
|
352
|
+
}
|
|
353
|
+
if (!inString && (ch === "}" || ch === "]")) {
|
|
354
|
+
out += ch;
|
|
355
|
+
stack.pop();
|
|
356
|
+
if (stack.length === 0) {
|
|
357
|
+
if (tryEmitCandidate(out)) return { success: bestSuccess, failed: bestFailed, escapes, ambiguous };
|
|
358
|
+
if (attempts >= MAX_REPAIR_CANDIDATES) break;
|
|
359
|
+
out = "";
|
|
360
|
+
}
|
|
361
|
+
i++;
|
|
362
|
+
continue;
|
|
363
|
+
}
|
|
364
|
+
out += ch;
|
|
365
|
+
i++;
|
|
366
|
+
}
|
|
367
|
+
return { success: null, failed: bestFailed, escapes, ambiguous };
|
|
368
|
+
}
|
|
369
|
+
function controlCharEscape(code) {
|
|
370
|
+
switch (code) {
|
|
371
|
+
case 10:
|
|
372
|
+
return "\\n";
|
|
373
|
+
case 9:
|
|
374
|
+
return "\\t";
|
|
375
|
+
case 13:
|
|
376
|
+
return "\\r";
|
|
377
|
+
case 8:
|
|
378
|
+
return "\\b";
|
|
379
|
+
case 12:
|
|
380
|
+
return "\\f";
|
|
381
|
+
default:
|
|
382
|
+
return "\\u" + code.toString(16).padStart(4, "0");
|
|
383
|
+
}
|
|
384
|
+
}
|
|
385
|
+
function extractParsePosition(err) {
|
|
386
|
+
const match = /position\s+(\d+)/.exec(safeErrorToString(err));
|
|
387
|
+
return match ? Number(match[1]) : null;
|
|
388
|
+
}
|
|
389
|
+
function safeErrorToString(e) {
|
|
390
|
+
if (e instanceof Error) {
|
|
391
|
+
const msg = readErrorField(e, "message");
|
|
392
|
+
if (typeof msg === "string" && msg.length > 0) return msg;
|
|
393
|
+
const name = readErrorField(e, "name");
|
|
394
|
+
if (typeof name === "string" && name.length > 0) return name;
|
|
395
|
+
return "[Error]";
|
|
396
|
+
}
|
|
397
|
+
try {
|
|
398
|
+
return String(e);
|
|
399
|
+
} catch {
|
|
400
|
+
try {
|
|
401
|
+
return Object.prototype.toString.call(e);
|
|
402
|
+
} catch {
|
|
403
|
+
return "[unstringifiable error]";
|
|
404
|
+
}
|
|
405
|
+
}
|
|
406
|
+
}
|
|
407
|
+
function readErrorField(e, key) {
|
|
408
|
+
try {
|
|
409
|
+
return e[key];
|
|
410
|
+
} catch {
|
|
411
|
+
return void 0;
|
|
412
|
+
}
|
|
166
413
|
}
|
|
167
414
|
function sanitizeRankerError(err, sanitizeRankerErrors) {
|
|
168
415
|
if (sanitizeRankerErrors === false) {
|
|
@@ -969,7 +1216,7 @@ Return ONLY a valid JSON object matching this schema:
|
|
|
969
1216
|
{
|
|
970
1217
|
"facts": [{ "title": "string (max 80 chars)", "body": "string (max 800 chars)", "tags": ["string"], "confidence": "certain|inferred|tentative" }]
|
|
971
1218
|
}
|
|
972
|
-
Extract verbatim factual content. Do not return markdown, just raw JSON.`;
|
|
1219
|
+
Extract verbatim factual content. JSON escaping rules: every literal " character in the source must be escaped as \\" inside any JSON string body, and every literal newline as \\n. Source prose containing quotes (e.g. a worked example with "...") must still be reproduced exactly \u2014 re-escape, do not omit. Do not return markdown, just raw JSON.`;
|
|
973
1220
|
var ONTOLOGY_BACKFILL_SYSTEM_PROMPT = `You are a knowledge classification agent. You will receive existing memory facts that currently have no ontology type. For each input fact { "id", "title", "body", "tags" }, assign the best matching okf_type from the ontology manifest and optionally propose edges to related facts by title.
|
|
974
1221
|
Return ONLY a valid JSON object matching this schema:
|
|
975
1222
|
{
|
|
@@ -978,7 +1225,7 @@ Return ONLY a valid JSON object matching this schema:
|
|
|
978
1225
|
]
|
|
979
1226
|
}
|
|
980
1227
|
If no manifest type fits a fact, omit that fact from "classifications" entirely \u2014 do not guess.
|
|
981
|
-
Do not return markdown, just raw JSON.`;
|
|
1228
|
+
When echoing an existing fact's title verbatim into "target_title", preserve every JSON escape sequence (\\", \\n, \\\\, \\/) exactly as it appeared in the input body \u2014 do not strip backslashes, do not add unescaped quotes. Do not return markdown, just raw JSON.`;
|
|
982
1229
|
|
|
983
1230
|
// src/services/PromptService.ts
|
|
984
1231
|
var PromptService = class {
|
|
@@ -1091,6 +1338,16 @@ var DEFAULT_MAX_CHUNK_LENGTH = 12e3;
|
|
|
1091
1338
|
var DEFAULT_CHUNK_OVERLAP = 400;
|
|
1092
1339
|
|
|
1093
1340
|
// src/services/IngestionService.ts
|
|
1341
|
+
function zeroChunkResult(duplicateOf) {
|
|
1342
|
+
const result = {
|
|
1343
|
+
truncated: false,
|
|
1344
|
+
chunks: 0,
|
|
1345
|
+
ingestedChunks: 0,
|
|
1346
|
+
failedChunks: 0
|
|
1347
|
+
};
|
|
1348
|
+
if (duplicateOf !== void 0) result.duplicateOf = duplicateOf;
|
|
1349
|
+
return result;
|
|
1350
|
+
}
|
|
1094
1351
|
var IngestionService = class {
|
|
1095
1352
|
constructor(db, prefix, options, entryRepo, sourceRefIndexRepo, metadataRepo, edgeRepo, searchService, jobManager, embeddingService, promptService, ontologyService) {
|
|
1096
1353
|
this.db = db;
|
|
@@ -1131,113 +1388,110 @@ var IngestionService = class {
|
|
|
1131
1388
|
if (onDuplicateHash === "throw") {
|
|
1132
1389
|
throw new WikiDuplicateHashError({ canonical, sourceHash, entityId });
|
|
1133
1390
|
}
|
|
1134
|
-
return
|
|
1391
|
+
return zeroChunkResult(canonical);
|
|
1135
1392
|
}
|
|
1136
1393
|
}
|
|
1137
1394
|
const { chunks, truncated } = chunkText(params.documentChunk, maxChunkLength, chunkOverlap);
|
|
1138
|
-
if (chunks.length === 0) return
|
|
1395
|
+
if (chunks.length === 0) return zeroChunkResult();
|
|
1396
|
+
const ontologyContext = await this.ontologyService?.buildPromptContext(entityId) ?? null;
|
|
1139
1397
|
const chunkResults = await withConcurrency(
|
|
1140
|
-
chunks.map((chunk) => async () => {
|
|
1141
|
-
const ontologyContext = await this.ontologyService?.buildPromptContext(entityId) ?? null;
|
|
1398
|
+
chunks.map((chunk, chunkIndex) => async () => {
|
|
1142
1399
|
const { systemPrompt, userPrompt } = this.promptService.buildIngestPrompt(
|
|
1143
1400
|
chunk,
|
|
1144
1401
|
params.promptOverride,
|
|
1145
1402
|
ontologyContext
|
|
1146
1403
|
);
|
|
1147
|
-
|
|
1148
|
-
|
|
1149
|
-
|
|
1150
|
-
|
|
1151
|
-
|
|
1152
|
-
|
|
1404
|
+
try {
|
|
1405
|
+
const responseText = await this.options.llmProvider.generateText({ systemPrompt, userPrompt });
|
|
1406
|
+
const result2 = parseJsonResponse(responseText);
|
|
1407
|
+
return {
|
|
1408
|
+
status: "ok",
|
|
1409
|
+
facts: (Array.isArray(result2.facts) ? result2.facts : []).map(validateFact).filter((f) => f !== null),
|
|
1410
|
+
ontology_updates: result2.ontology_updates
|
|
1411
|
+
};
|
|
1412
|
+
} catch (e) {
|
|
1413
|
+
const failure = e instanceof WikiParseError ? {
|
|
1414
|
+
chunkIndex,
|
|
1415
|
+
sourceRef,
|
|
1416
|
+
source: "parse",
|
|
1417
|
+
tier: e.tier,
|
|
1418
|
+
position: e.position,
|
|
1419
|
+
message: e.message
|
|
1420
|
+
} : {
|
|
1421
|
+
chunkIndex,
|
|
1422
|
+
sourceRef,
|
|
1423
|
+
source: "llm",
|
|
1424
|
+
// `safeErrorToString` is a non-throwing coercion: an LLM
|
|
1425
|
+
// provider may reject with a non-Error value, and even
|
|
1426
|
+
// `String(e)` itself can throw when `e.toString()` throws.
|
|
1427
|
+
// Letting such an exception escape this catch would cause
|
|
1428
|
+
// `withConcurrency` to reject, discarding every sibling
|
|
1429
|
+
// chunk's results — the exact failure mode this per-chunk
|
|
1430
|
+
// try/catch exists to prevent. Keeping the value in the
|
|
1431
|
+
// `ChunkFailure` preserves the typed diagnostic for
|
|
1432
|
+
// callers (see parseFailures) without unwinding the loop.
|
|
1433
|
+
position: null,
|
|
1434
|
+
message: safeErrorToString(e)
|
|
1435
|
+
};
|
|
1436
|
+
const total = chunks.length;
|
|
1437
|
+
const tags = [];
|
|
1438
|
+
if (failure.tier) tags.push(`tier=${failure.tier}`);
|
|
1439
|
+
if (failure.position !== null) tags.push(`position=${failure.position}`);
|
|
1440
|
+
const tagsSuffix = tags.length > 0 ? `; ${tags.join(" ")}` : "";
|
|
1441
|
+
console.warn(
|
|
1442
|
+
`[WikiMemory] ingest chunk ${chunkIndex + 1}/${total} ${failure.source} failed (sourceRef=${sourceRef}${tagsSuffix})`
|
|
1443
|
+
);
|
|
1444
|
+
return { status: "failed", error: failure };
|
|
1445
|
+
}
|
|
1153
1446
|
}),
|
|
1154
1447
|
chunkConcurrency
|
|
1155
1448
|
);
|
|
1449
|
+
let ingestedChunks = 0;
|
|
1450
|
+
let failedChunks = 0;
|
|
1451
|
+
const failures = [];
|
|
1156
1452
|
const seen = /* @__PURE__ */ new Set();
|
|
1157
1453
|
const orderedChunkFacts = [];
|
|
1158
|
-
for (const
|
|
1454
|
+
for (const slot of chunkResults) {
|
|
1455
|
+
if (slot.status === "failed") {
|
|
1456
|
+
failedChunks++;
|
|
1457
|
+
failures.push(slot.error);
|
|
1458
|
+
continue;
|
|
1459
|
+
}
|
|
1460
|
+
ingestedChunks++;
|
|
1159
1461
|
const dedupedFacts = [];
|
|
1160
|
-
for (const fact of
|
|
1462
|
+
for (const fact of slot.facts) {
|
|
1161
1463
|
const normalizedTitle = normalizeTitleKey(fact.title);
|
|
1162
1464
|
if (!seen.has(normalizedTitle)) {
|
|
1163
1465
|
seen.add(normalizedTitle);
|
|
1164
1466
|
dedupedFacts.push(fact);
|
|
1165
1467
|
}
|
|
1166
1468
|
}
|
|
1167
|
-
orderedChunkFacts.push({ facts: dedupedFacts, ontology_updates:
|
|
1469
|
+
orderedChunkFacts.push({ facts: dedupedFacts, ontology_updates: slot.ontology_updates });
|
|
1470
|
+
}
|
|
1471
|
+
if (failedChunks === chunks.length) {
|
|
1472
|
+
throw new WikiIngestEmptyError({
|
|
1473
|
+
parseFailures: failures,
|
|
1474
|
+
sourceRef,
|
|
1475
|
+
chunks: chunks.length
|
|
1476
|
+
});
|
|
1168
1477
|
}
|
|
1169
|
-
const now = Date.now();
|
|
1170
1478
|
const insertedFacts = [];
|
|
1171
1479
|
const deletedSourceFactIds = [];
|
|
1172
1480
|
try {
|
|
1173
|
-
|
|
1174
|
-
|
|
1175
|
-
|
|
1176
|
-
|
|
1177
|
-
|
|
1178
|
-
|
|
1179
|
-
|
|
1180
|
-
|
|
1181
|
-
|
|
1182
|
-
|
|
1183
|
-
|
|
1184
|
-
|
|
1185
|
-
|
|
1186
|
-
|
|
1187
|
-
for (const { facts, ontology_updates } of orderedChunkFacts) {
|
|
1188
|
-
if (mode === "emergent" && ontology_updates && this.ontologyService) {
|
|
1189
|
-
manifest = await this.ontologyService.mergeEmergentUpdates(entityId, ontology_updates, tx);
|
|
1190
|
-
ontologyState = await this.ontologyService.getEffectiveState(entityId, tx);
|
|
1191
|
-
mode = ontologyState.mode;
|
|
1192
|
-
}
|
|
1193
|
-
for (const fact of facts) {
|
|
1194
|
-
const ontologyFact = fact;
|
|
1195
|
-
const normalized = this.ontologyService?.validateAndNormalizeFact(ontologyFact, manifest, { strict: false }) ?? { okf_type: null, edges: [] };
|
|
1196
|
-
const id = generateId("fact_");
|
|
1197
|
-
hostNodes.push({
|
|
1198
|
-
id,
|
|
1199
|
-
type: ontologyFact.okf_type ?? "",
|
|
1200
|
-
title: fact.title,
|
|
1201
|
-
body: fact.body,
|
|
1202
|
-
// Forward the LLM-extracted tags/confidence through the
|
|
1203
|
-
// extracted-shape upsertGraphCore path so the entries row
|
|
1204
|
-
// stores them (search filterability, heal-candidate
|
|
1205
|
-
// selection, runReembed embedding-text signal). The public
|
|
1206
|
-
// WikiMemory.upsertGraph API doesn't accept tags/confidence
|
|
1207
|
-
// — host-supplied deterministic nodes default to [] and
|
|
1208
|
-
// 'certain' as documented.
|
|
1209
|
-
tags: fact.tags,
|
|
1210
|
-
confidence: fact.confidence
|
|
1211
|
-
});
|
|
1212
|
-
insertedFacts.push({ id, entity_id: entityId, title: fact.title, body: fact.body, tags: JSON.stringify(fact.tags) });
|
|
1213
|
-
titleIndex.set(normalizeTitleKey(fact.title), { id, okf_type: normalized.okf_type });
|
|
1214
|
-
if (normalized.edges.length > 0) {
|
|
1215
|
-
rawEdgeRequests.push({ sourceId: id, sourceType: normalized.okf_type, edges: normalized.edges });
|
|
1216
|
-
}
|
|
1217
|
-
}
|
|
1218
|
-
}
|
|
1219
|
-
const hostEdges = [];
|
|
1220
|
-
for (const req of rawEdgeRequests) {
|
|
1221
|
-
const resolved = this.ontologyService?.resolveEdges(
|
|
1222
|
-
entityId,
|
|
1223
|
-
req.sourceId,
|
|
1224
|
-
req.sourceType,
|
|
1225
|
-
req.edges,
|
|
1226
|
-
manifest,
|
|
1227
|
-
titleIndex,
|
|
1228
|
-
now
|
|
1229
|
-
) ?? [];
|
|
1230
|
-
for (const e of resolved) {
|
|
1231
|
-
hostEdges.push({ type: e.edge_type, sourceId: e.source_id, targetId: e.target_id });
|
|
1232
|
-
}
|
|
1233
|
-
}
|
|
1234
|
-
await this.upsertGraphCore(
|
|
1235
|
-
entityId,
|
|
1236
|
-
{ sourceRef, sourceHash, nodes: hostNodes, edges: hostEdges },
|
|
1237
|
-
tx,
|
|
1238
|
-
{ strict: false }
|
|
1239
|
-
);
|
|
1240
|
-
});
|
|
1481
|
+
if (failedChunks === 0) {
|
|
1482
|
+
const fullResult = await this.db.withTransactionAsync(async (tx) => {
|
|
1483
|
+
return await this.runFullUpsertGraph(entityId, sourceRef, sourceHash, orderedChunkFacts, tx);
|
|
1484
|
+
});
|
|
1485
|
+
deletedSourceFactIds.push(...fullResult.deletedSourceFactIds);
|
|
1486
|
+
insertedFacts.push(...fullResult.insertedFacts);
|
|
1487
|
+
} else {
|
|
1488
|
+
const partialResult = await this.db.withTransactionAsync(async (tx) => {
|
|
1489
|
+
const flat = [];
|
|
1490
|
+
for (const slot of orderedChunkFacts) flat.push(...slot.facts);
|
|
1491
|
+
return await this.appendPartialFacts(entityId, sourceRef, flat, tx);
|
|
1492
|
+
});
|
|
1493
|
+
insertedFacts.push(...partialResult.insertedDescriptors);
|
|
1494
|
+
}
|
|
1241
1495
|
} catch (err) {
|
|
1242
1496
|
const sqliteCode = err instanceof WikiTransactionError ? err.sqliteErrorCode : extractSqliteCode(err);
|
|
1243
1497
|
if (sqliteCode !== "SQLITE_CONSTRAINT_UNIQUE") throw err;
|
|
@@ -1246,22 +1500,31 @@ var IngestionService = class {
|
|
|
1246
1500
|
if (onDuplicateHash === "throw" || onDuplicateHash === "ingest") {
|
|
1247
1501
|
throw new WikiDuplicateHashError({ canonical, sourceHash, entityId });
|
|
1248
1502
|
}
|
|
1249
|
-
return
|
|
1503
|
+
return zeroChunkResult(canonical);
|
|
1250
1504
|
}
|
|
1251
1505
|
await this.searchService.sync(entityId);
|
|
1252
|
-
|
|
1253
|
-
|
|
1254
|
-
|
|
1255
|
-
|
|
1256
|
-
|
|
1257
|
-
|
|
1506
|
+
if (failedChunks === 0) {
|
|
1507
|
+
const uniqueDeletedSourceFactIds = Array.from(new Set(deletedSourceFactIds));
|
|
1508
|
+
for (const factId of uniqueDeletedSourceFactIds) {
|
|
1509
|
+
try {
|
|
1510
|
+
await this.embeddingService.notifyEmbeddingPersisted(entityId, factId, null);
|
|
1511
|
+
} catch (hookErr) {
|
|
1512
|
+
console.warn(`[WikiMemory] onEmbeddingPersisted hook failed during ingest for ${factId}:`, hookErr);
|
|
1513
|
+
}
|
|
1258
1514
|
}
|
|
1259
1515
|
}
|
|
1260
1516
|
for (const fact of insertedFacts) {
|
|
1261
1517
|
await this.embeddingService.embedFact(fact);
|
|
1262
1518
|
}
|
|
1263
1519
|
this.searchService.evictCache(entityId);
|
|
1264
|
-
|
|
1520
|
+
const result = {
|
|
1521
|
+
truncated,
|
|
1522
|
+
chunks: chunks.length,
|
|
1523
|
+
ingestedChunks,
|
|
1524
|
+
failedChunks
|
|
1525
|
+
};
|
|
1526
|
+
if (failedChunks > 0) result.parseFailures = failures;
|
|
1527
|
+
return result;
|
|
1265
1528
|
} finally {
|
|
1266
1529
|
releaseIngestLocks();
|
|
1267
1530
|
}
|
|
@@ -1386,6 +1649,156 @@ var IngestionService = class {
|
|
|
1386
1649
|
superseded: deletedFactIds.length + deletedEdgeCount
|
|
1387
1650
|
};
|
|
1388
1651
|
}
|
|
1652
|
+
/**
|
|
1653
|
+
* Full supersession + ownership path: identical to the pre-issue-#92 behavior
|
|
1654
|
+
* for the happy path. Called from the `failedChunks === 0` branch of
|
|
1655
|
+
* `ingestDocument`. Runs INSIDE the caller's tx.
|
|
1656
|
+
*
|
|
1657
|
+
* Returns `{ deletedSourceFactIds, insertedFacts }` so the caller can fire
|
|
1658
|
+
* post-commit hooks (embedding lifecycle) AFTER the transaction commits.
|
|
1659
|
+
* On the partial path the caller never invokes this method; the empty
|
|
1660
|
+
* arrays stay empty.
|
|
1661
|
+
*/
|
|
1662
|
+
async runFullUpsertGraph(entityId, sourceRef, sourceHash, orderedChunkFacts, tx) {
|
|
1663
|
+
const deletedSourceFactIds = [];
|
|
1664
|
+
const insertedFacts = [];
|
|
1665
|
+
deletedSourceFactIds.push(...await this.entryRepo.findIdsBySource(entityId, sourceRef, null, tx, false));
|
|
1666
|
+
const titleIndex = /* @__PURE__ */ new Map();
|
|
1667
|
+
const existingFacts = await this.entryRepo.findRecentByEntityId(entityId, 500, tx, sourceRef);
|
|
1668
|
+
for (const existing of existingFacts) {
|
|
1669
|
+
titleIndex.set(normalizeTitleKey(existing.title), {
|
|
1670
|
+
id: existing.id,
|
|
1671
|
+
okf_type: existing.okf_type ?? null
|
|
1672
|
+
});
|
|
1673
|
+
}
|
|
1674
|
+
let ontologyState = await this.ontologyService?.getEffectiveState(entityId, tx) ?? { mode: "off", manifest: { node_types: [], edge_types: [] } };
|
|
1675
|
+
let { mode, manifest } = ontologyState;
|
|
1676
|
+
const hostNodes = [];
|
|
1677
|
+
const rawEdgeRequests = [];
|
|
1678
|
+
const now = Date.now();
|
|
1679
|
+
for (const { facts, ontology_updates } of orderedChunkFacts) {
|
|
1680
|
+
if (mode === "emergent" && ontology_updates && this.ontologyService) {
|
|
1681
|
+
manifest = await this.ontologyService.mergeEmergentUpdates(entityId, ontology_updates, tx);
|
|
1682
|
+
ontologyState = await this.ontologyService.getEffectiveState(entityId, tx);
|
|
1683
|
+
mode = ontologyState.mode;
|
|
1684
|
+
}
|
|
1685
|
+
for (const fact of facts) {
|
|
1686
|
+
const ontologyFact = fact;
|
|
1687
|
+
const normalized = this.ontologyService?.validateAndNormalizeFact(ontologyFact, manifest, { strict: false }) ?? { okf_type: null, edges: [] };
|
|
1688
|
+
const id = generateId("fact_");
|
|
1689
|
+
hostNodes.push({
|
|
1690
|
+
id,
|
|
1691
|
+
type: ontologyFact.okf_type ?? "",
|
|
1692
|
+
title: fact.title,
|
|
1693
|
+
body: fact.body,
|
|
1694
|
+
// Forward the LLM-extracted tags/confidence through the
|
|
1695
|
+
// extracted-shape upsertGraphCore path so the entries row
|
|
1696
|
+
// stores them (search filterability, heal-candidate
|
|
1697
|
+
// selection, runReembed embedding-text signal). The public
|
|
1698
|
+
// WikiMemory.upsertGraph API doesn't accept tags/confidence
|
|
1699
|
+
// — host-supplied deterministic nodes default to [] and
|
|
1700
|
+
// 'certain' as documented.
|
|
1701
|
+
tags: fact.tags,
|
|
1702
|
+
confidence: fact.confidence
|
|
1703
|
+
});
|
|
1704
|
+
insertedFacts.push({ id, entity_id: entityId, title: fact.title, body: fact.body, tags: JSON.stringify(fact.tags) });
|
|
1705
|
+
titleIndex.set(normalizeTitleKey(fact.title), { id, okf_type: normalized.okf_type });
|
|
1706
|
+
if (normalized.edges.length > 0) {
|
|
1707
|
+
rawEdgeRequests.push({ sourceId: id, sourceType: normalized.okf_type, edges: normalized.edges });
|
|
1708
|
+
}
|
|
1709
|
+
}
|
|
1710
|
+
}
|
|
1711
|
+
const hostEdges = [];
|
|
1712
|
+
for (const req of rawEdgeRequests) {
|
|
1713
|
+
const resolved = this.ontologyService?.resolveEdges(
|
|
1714
|
+
entityId,
|
|
1715
|
+
req.sourceId,
|
|
1716
|
+
req.sourceType,
|
|
1717
|
+
req.edges,
|
|
1718
|
+
manifest,
|
|
1719
|
+
titleIndex,
|
|
1720
|
+
now
|
|
1721
|
+
) ?? [];
|
|
1722
|
+
for (const e of resolved) {
|
|
1723
|
+
hostEdges.push({ type: e.edge_type, sourceId: e.source_id, targetId: e.target_id });
|
|
1724
|
+
}
|
|
1725
|
+
}
|
|
1726
|
+
await this.upsertGraphCore(
|
|
1727
|
+
entityId,
|
|
1728
|
+
{ sourceRef, sourceHash, nodes: hostNodes, edges: hostEdges },
|
|
1729
|
+
tx,
|
|
1730
|
+
{ strict: false }
|
|
1731
|
+
);
|
|
1732
|
+
return { deletedSourceFactIds, insertedFacts };
|
|
1733
|
+
}
|
|
1734
|
+
/**
|
|
1735
|
+
* Partial-commit path for `ingestDocument`. Inserts ONLY facts whose
|
|
1736
|
+
* normalized title is not already in the live set for this (entityId,
|
|
1737
|
+
* sourceRef). Does NOT call `entryRepo.softDeleteBySource` (no
|
|
1738
|
+
* supersession) and does NOT call `sourceRefIndexRepo.upsert` (no
|
|
1739
|
+
* ownership). Edges are NOT resolved on this path (no ontology-context
|
|
1740
|
+
* build, no `resolveEdges`, no `mergeEmergentUpdates`). Stale edges from
|
|
1741
|
+
* prior attempts are recovered on the next full run's supersession.
|
|
1742
|
+
*
|
|
1743
|
+
* Source-hash is stored as NULL on partial rows. `findLatestSourceHash`
|
|
1744
|
+
* reads from the most recently updated live row for the sourceRef, so
|
|
1745
|
+
* storing the incoming hash here would cause `hasChanged` to return
|
|
1746
|
+
* `false` on a same-hash retry — the failed chunks would never get a
|
|
1747
|
+
* second chance. With NULL, a retry sees `storedHash === null`, `hasChanged`
|
|
1748
|
+
* returns `true`, and the partial commit's sibling rows remain live
|
|
1749
|
+
* (deduped, not superseded) for the next attempt to extend.
|
|
1750
|
+
*
|
|
1751
|
+
* Runs INSIDE the caller's `tx`. Does not open a nested transaction.
|
|
1752
|
+
* Returns `{ inserted, skippedDuplicate }` for observability.
|
|
1753
|
+
*/
|
|
1754
|
+
async appendPartialFacts(entityId, sourceRef, dedupedFacts, tx) {
|
|
1755
|
+
const liveIds = await this.entryRepo.findIdsBySource(entityId, sourceRef, null, tx, false);
|
|
1756
|
+
const liveFacts = liveIds.length === 0 ? [] : await this.entryRepo.findByIds(liveIds, void 0, tx);
|
|
1757
|
+
const liveTitles = new Set(liveFacts.map((f) => normalizeTitleKey(f.title)));
|
|
1758
|
+
let inserted = 0;
|
|
1759
|
+
let skippedDuplicate = 0;
|
|
1760
|
+
const now = Date.now();
|
|
1761
|
+
const insertedDescriptors = [];
|
|
1762
|
+
for (const fact of dedupedFacts) {
|
|
1763
|
+
const normalizedTitle = normalizeTitleKey(fact.title);
|
|
1764
|
+
if (liveTitles.has(normalizedTitle)) {
|
|
1765
|
+
skippedDuplicate++;
|
|
1766
|
+
continue;
|
|
1767
|
+
}
|
|
1768
|
+
liveTitles.add(normalizedTitle);
|
|
1769
|
+
const id = generateId("fact_");
|
|
1770
|
+
const wikiFact = {
|
|
1771
|
+
id,
|
|
1772
|
+
entity_id: entityId,
|
|
1773
|
+
title: fact.title,
|
|
1774
|
+
body: fact.body,
|
|
1775
|
+
tags: fact.tags,
|
|
1776
|
+
confidence: fact.confidence,
|
|
1777
|
+
source_type: "immutable_document",
|
|
1778
|
+
// source_hash: null on partial rows — see method docstring. The
|
|
1779
|
+
// full path stamps the actual hash; partial rows stay hash-less so
|
|
1780
|
+
// `hasChanged` keeps returning true for retries.
|
|
1781
|
+
source_hash: null,
|
|
1782
|
+
source_ref: sourceRef,
|
|
1783
|
+
created_at: now,
|
|
1784
|
+
updated_at: now,
|
|
1785
|
+
last_accessed_at: null,
|
|
1786
|
+
access_count: 0,
|
|
1787
|
+
deleted_at: null,
|
|
1788
|
+
okf_type: null
|
|
1789
|
+
};
|
|
1790
|
+
await this.entryRepo.upsert(wikiFact, tx);
|
|
1791
|
+
insertedDescriptors.push({
|
|
1792
|
+
id,
|
|
1793
|
+
entity_id: entityId,
|
|
1794
|
+
title: fact.title,
|
|
1795
|
+
body: fact.body,
|
|
1796
|
+
tags: JSON.stringify(fact.tags)
|
|
1797
|
+
});
|
|
1798
|
+
inserted++;
|
|
1799
|
+
}
|
|
1800
|
+
return { inserted, skippedDuplicate, insertedDescriptors };
|
|
1801
|
+
}
|
|
1389
1802
|
};
|
|
1390
1803
|
|
|
1391
1804
|
// src/utils/embedding.ts
|
|
@@ -1534,6 +1947,25 @@ var HEAL_ANCHOR_SEARCH_OVERFETCH = 4;
|
|
|
1534
1947
|
var HEAL_MAX_PROMPT_CHARS = 4e4;
|
|
1535
1948
|
var HEAL_BATCH_SIZE = 25;
|
|
1536
1949
|
var HEAL_RECHECK_MS = 7 * 24 * 60 * 60 * 1e3;
|
|
1950
|
+
var SKIP_ERROR_LOG_CHARS = 4096;
|
|
1951
|
+
var formatSkipError = (err) => {
|
|
1952
|
+
let base;
|
|
1953
|
+
if (err instanceof Error || typeof err !== "object" && typeof err !== "function") {
|
|
1954
|
+
base = safeErrorToString(err);
|
|
1955
|
+
} else {
|
|
1956
|
+
try {
|
|
1957
|
+
const json = JSON.stringify(err);
|
|
1958
|
+
base = typeof json === "string" ? json : safeErrorToString(err);
|
|
1959
|
+
} catch {
|
|
1960
|
+
base = safeErrorToString(err);
|
|
1961
|
+
}
|
|
1962
|
+
}
|
|
1963
|
+
if (base.length > SKIP_ERROR_LOG_CHARS) {
|
|
1964
|
+
const truncated = safeSlice(base, 0, SKIP_ERROR_LOG_CHARS);
|
|
1965
|
+
return `${truncated}\u2026[+${base.length - SKIP_ERROR_LOG_CHARS} chars truncated]`;
|
|
1966
|
+
}
|
|
1967
|
+
return base;
|
|
1968
|
+
};
|
|
1537
1969
|
var MaintenanceService = class {
|
|
1538
1970
|
constructor(db, prefix, options, entryRepo, sourceRefIndexRepo, taskRepo, eventRepo, metadataRepo, searchService, jobManager, embeddingService, promptService, ontologyService) {
|
|
1539
1971
|
this.db = db;
|
|
@@ -2014,8 +2446,7 @@ var MaintenanceService = class {
|
|
|
2014
2446
|
maxPromptChars: HEAL_MAX_PROMPT_CHARS,
|
|
2015
2447
|
onSkip: (fact, err) => {
|
|
2016
2448
|
console.warn(
|
|
2017
|
-
`[WikiMemory] heal skipped ${entityId}/${fact.id}: response could not be bounded
|
|
2018
|
-
err
|
|
2449
|
+
`[WikiMemory] heal skipped ${entityId}/${fact.id}: response could not be bounded: ${formatSkipError(err)}`
|
|
2019
2450
|
);
|
|
2020
2451
|
}
|
|
2021
2452
|
});
|
|
@@ -2153,8 +2584,7 @@ var MaintenanceService = class {
|
|
|
2153
2584
|
maxPromptChars: ONTOLOGY_BACKFILL_MAX_PROMPT_CHARS,
|
|
2154
2585
|
onSkip: (fact, err) => {
|
|
2155
2586
|
console.warn(
|
|
2156
|
-
`[WikiMemory] ontology backfill skipped ${entityId}/${fact.id}: response could not be bounded
|
|
2157
|
-
err
|
|
2587
|
+
`[WikiMemory] ontology backfill skipped ${entityId}/${fact.id}: response could not be bounded: ${formatSkipError(err)}`
|
|
2158
2588
|
);
|
|
2159
2589
|
}
|
|
2160
2590
|
});
|