@equationalapplications/core-llm-wiki 4.22.0 → 4.23.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +10 -6
- package/dist/{chunk-LCAU6NYC.mjs → chunk-MYZJLVX4.mjs} +311 -45
- package/dist/chunk-MYZJLVX4.mjs.map +1 -0
- package/dist/index.d.mts +2 -2
- package/dist/index.d.ts +2 -2
- package/dist/index.js +348 -43
- package/dist/index.js.map +1 -1
- package/dist/index.mjs +41 -2
- package/dist/index.mjs.map +1 -1
- package/dist/{testing-Bk8J6QLX.d.mts → testing-CBjAuTSl.d.mts} +72 -0
- package/dist/{testing-Bk8J6QLX.d.ts → testing-CBjAuTSl.d.ts} +72 -0
- package/dist/testing.d.mts +1 -1
- package/dist/testing.d.ts +1 -1
- package/dist/testing.js +309 -43
- package/dist/testing.js.map +1 -1
- package/dist/testing.mjs +1 -1
- package/package.json +6 -6
- package/dist/chunk-LCAU6NYC.mjs.map +0 -1
package/dist/index.d.mts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
import { M as MemoryBundle, F as FormatContextOptions, G as GraphNeighborhood, a as MemoryDump, b as FormattedMemoryDump, O as OntologyManifest, R as ReadOptions, S as SQLiteAdapter, W as WikiOptions, c as WikiMemory } from './testing-
|
|
2
|
-
export { E as EntityStatus, d as ExtractedFact, e as ExtractedFactEdge, f as ExtractedFactWithOntology, g as ExtractedTask, h as GraphTraversalOptions, H as HOOK_TIMEOUT_MARKER, L as LLMProvider, i as ONTOLOGY_BACKFILL_BATCH_SIZE, j as ONTOLOGY_BACKFILL_MAX_PROMPT_CHARS, k as ONTOLOGY_BACKFILL_RECHECK_MS, l as OntologyBackfillResult, m as OntologyConfig, n as OntologyEdgeType, o as OntologyMode, p as OntologyNodeType, q as OntologyPromptContext, r as OntologyUpdates, P as PromptOverrides, s as PromptService, t as PrunePartialFailureError, V as VectorRanker, u as VectorRankerFallback, v as VectorRankerRankArgs, w as VectorRankerSemanticResult, x as WikiBusyError, y as WikiBusyOperation, z as WikiCheckpoint, A as WikiConfig, B as WikiEdge, C as WikiEvent, D as WikiFact, I as WikiMemoryTestAccess, J as WikiOutboxEvent, K as WikiTask, N as WikiTransactionError } from './testing-
|
|
1
|
+
import { M as MemoryBundle, F as FormatContextOptions, G as GraphNeighborhood, a as MemoryDump, b as FormattedMemoryDump, O as OntologyManifest, R as ReadOptions, S as SQLiteAdapter, W as WikiOptions, c as WikiMemory } from './testing-CBjAuTSl.mjs';
|
|
2
|
+
export { E as EntityStatus, d as ExtractedFact, e as ExtractedFactEdge, f as ExtractedFactWithOntology, g as ExtractedTask, h as GraphTraversalOptions, H as HOOK_TIMEOUT_MARKER, L as LLMProvider, i as ONTOLOGY_BACKFILL_BATCH_SIZE, j as ONTOLOGY_BACKFILL_MAX_PROMPT_CHARS, k as ONTOLOGY_BACKFILL_RECHECK_MS, l as OntologyBackfillResult, m as OntologyConfig, n as OntologyEdgeType, o as OntologyMode, p as OntologyNodeType, q as OntologyPromptContext, r as OntologyUpdates, P as PromptOverrides, s as PromptService, t as PrunePartialFailureError, V as VectorRanker, u as VectorRankerFallback, v as VectorRankerRankArgs, w as VectorRankerSemanticResult, x as WikiBusyError, y as WikiBusyOperation, z as WikiCheckpoint, A as WikiConfig, B as WikiEdge, C as WikiEvent, D as WikiFact, I as WikiMemoryTestAccess, J as WikiOutboxEvent, K as WikiTask, N as WikiTransactionError } from './testing-CBjAuTSl.mjs';
|
|
3
3
|
import { OkfFile } from '@equationalapplications/core-okf';
|
|
4
4
|
import 'minisearch';
|
|
5
5
|
|
package/dist/index.d.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
import { M as MemoryBundle, F as FormatContextOptions, G as GraphNeighborhood, a as MemoryDump, b as FormattedMemoryDump, O as OntologyManifest, R as ReadOptions, S as SQLiteAdapter, W as WikiOptions, c as WikiMemory } from './testing-
|
|
2
|
-
export { E as EntityStatus, d as ExtractedFact, e as ExtractedFactEdge, f as ExtractedFactWithOntology, g as ExtractedTask, h as GraphTraversalOptions, H as HOOK_TIMEOUT_MARKER, L as LLMProvider, i as ONTOLOGY_BACKFILL_BATCH_SIZE, j as ONTOLOGY_BACKFILL_MAX_PROMPT_CHARS, k as ONTOLOGY_BACKFILL_RECHECK_MS, l as OntologyBackfillResult, m as OntologyConfig, n as OntologyEdgeType, o as OntologyMode, p as OntologyNodeType, q as OntologyPromptContext, r as OntologyUpdates, P as PromptOverrides, s as PromptService, t as PrunePartialFailureError, V as VectorRanker, u as VectorRankerFallback, v as VectorRankerRankArgs, w as VectorRankerSemanticResult, x as WikiBusyError, y as WikiBusyOperation, z as WikiCheckpoint, A as WikiConfig, B as WikiEdge, C as WikiEvent, D as WikiFact, I as WikiMemoryTestAccess, J as WikiOutboxEvent, K as WikiTask, N as WikiTransactionError } from './testing-
|
|
1
|
+
import { M as MemoryBundle, F as FormatContextOptions, G as GraphNeighborhood, a as MemoryDump, b as FormattedMemoryDump, O as OntologyManifest, R as ReadOptions, S as SQLiteAdapter, W as WikiOptions, c as WikiMemory } from './testing-CBjAuTSl.js';
|
|
2
|
+
export { E as EntityStatus, d as ExtractedFact, e as ExtractedFactEdge, f as ExtractedFactWithOntology, g as ExtractedTask, h as GraphTraversalOptions, H as HOOK_TIMEOUT_MARKER, L as LLMProvider, i as ONTOLOGY_BACKFILL_BATCH_SIZE, j as ONTOLOGY_BACKFILL_MAX_PROMPT_CHARS, k as ONTOLOGY_BACKFILL_RECHECK_MS, l as OntologyBackfillResult, m as OntologyConfig, n as OntologyEdgeType, o as OntologyMode, p as OntologyNodeType, q as OntologyPromptContext, r as OntologyUpdates, P as PromptOverrides, s as PromptService, t as PrunePartialFailureError, V as VectorRanker, u as VectorRankerFallback, v as VectorRankerRankArgs, w as VectorRankerSemanticResult, x as WikiBusyError, y as WikiBusyOperation, z as WikiCheckpoint, A as WikiConfig, B as WikiEdge, C as WikiEvent, D as WikiFact, I as WikiMemoryTestAccess, J as WikiOutboxEvent, K as WikiTask, N as WikiTransactionError } from './testing-CBjAuTSl.js';
|
|
3
3
|
import { OkfFile } from '@equationalapplications/core-okf';
|
|
4
4
|
import 'minisearch';
|
|
5
5
|
|
package/dist/index.js
CHANGED
|
@@ -707,6 +707,45 @@ var EntryRepository = class extends BaseRepository {
|
|
|
707
707
|
);
|
|
708
708
|
return rows.map(mapRowToFact);
|
|
709
709
|
}
|
|
710
|
+
/**
|
|
711
|
+
* Fetch live, mutable entries for an entity — everything heal is allowed to
|
|
712
|
+
* downgrade or delete. Heal previously loaded every row via
|
|
713
|
+
* findAllByEntityId and filtered in JS, which on a document-heavy corpus
|
|
714
|
+
* meant loading 2560 rows to keep 31.
|
|
715
|
+
*/
|
|
716
|
+
async findHealCandidatesByEntityId(entityId, tx) {
|
|
717
|
+
const executor = this.getExecutor(tx);
|
|
718
|
+
const rows = await executor.getAllAsync(
|
|
719
|
+
`SELECT * FROM ${this.prefix}entries
|
|
720
|
+
WHERE entity_id = ? AND deleted_at IS NULL AND source_type != 'immutable_document'
|
|
721
|
+
ORDER BY updated_at DESC`,
|
|
722
|
+
[entityId]
|
|
723
|
+
);
|
|
724
|
+
return rows.map(mapRowToFact);
|
|
725
|
+
}
|
|
726
|
+
/**
|
|
727
|
+
* Resolve search hits to document anchors. The MiniSearch index holds every
|
|
728
|
+
* fact, not only immutable_document rows, so the source-type restriction has
|
|
729
|
+
* to be applied here, after retrieval.
|
|
730
|
+
*/
|
|
731
|
+
async findAnchorRowsByIds(entityId, ids, tx) {
|
|
732
|
+
if (ids.length === 0) return [];
|
|
733
|
+
const executor = this.getExecutor(tx);
|
|
734
|
+
const rows = [];
|
|
735
|
+
for (let i = 0; i < ids.length; i += this.chunkSize) {
|
|
736
|
+
const chunk = ids.slice(i, i + this.chunkSize);
|
|
737
|
+
const placeholders = chunk.map(() => "?").join(", ");
|
|
738
|
+
const chunkRows = await executor.getAllAsync(
|
|
739
|
+
`SELECT id, title, source_ref FROM ${this.prefix}entries
|
|
740
|
+
WHERE entity_id = ? AND deleted_at IS NULL
|
|
741
|
+
AND source_type = 'immutable_document'
|
|
742
|
+
AND id IN (${placeholders})`,
|
|
743
|
+
[entityId, ...chunk]
|
|
744
|
+
);
|
|
745
|
+
rows.push(...chunkRows);
|
|
746
|
+
}
|
|
747
|
+
return rows;
|
|
748
|
+
}
|
|
710
749
|
/**
|
|
711
750
|
* Fetch recent non-deleted entries for an entity (limited), ordered by updated_at DESC.
|
|
712
751
|
* Used by MaintenanceService.doRunLibrarian().
|
|
@@ -2063,9 +2102,23 @@ var _SearchService = class _SearchService {
|
|
|
2063
2102
|
this.entryRepo = entryRepo;
|
|
2064
2103
|
this.miniSearchEntryIdsByEntity = /* @__PURE__ */ new Map();
|
|
2065
2104
|
this.vectorCache = /* @__PURE__ */ new Map();
|
|
2105
|
+
/**
|
|
2106
|
+
* Serializes rebuilds. `rebuildIndex` awaits a repository read between
|
|
2107
|
+
* snapshotting the previous id set and discarding it, so two concurrent
|
|
2108
|
+
* sync() calls for one entity can interleave: a slow, stale read lands last
|
|
2109
|
+
* and discards documents the fresh read just added. Chaining also keeps
|
|
2110
|
+
* discard()/addAll() out of each other's way, which is what accrued the
|
|
2111
|
+
* auto-vacuum debt behind the TypeError in #64.
|
|
2112
|
+
*/
|
|
2113
|
+
this.syncChain = Promise.resolve();
|
|
2066
2114
|
this.miniSearch = new MiniSearch__default.default({
|
|
2067
2115
|
fields: ["title", "body", "tags"],
|
|
2068
2116
|
storeFields: ["entity_id"],
|
|
2117
|
+
// Vacuuming is driven explicitly at the end of each serialized rebuild
|
|
2118
|
+
// (see sync). Auto-vacuum fires on its own schedule, asynchronously with
|
|
2119
|
+
// respect to the caller, and traversing the tree mid-rebuild is what
|
|
2120
|
+
// threw the uncaught TypeError in MiniSearch.performVacuuming (#64).
|
|
2121
|
+
autoVacuum: false,
|
|
2069
2122
|
searchOptions: {
|
|
2070
2123
|
boost: { title: 2 },
|
|
2071
2124
|
fuzzy: 0.2,
|
|
@@ -2076,10 +2129,26 @@ var _SearchService = class _SearchService {
|
|
|
2076
2129
|
/**
|
|
2077
2130
|
* Rebuilds the search index and clears the vector cache for a given entity.
|
|
2078
2131
|
* A direct replacement for manually syncing state after a DB transaction.
|
|
2132
|
+
*
|
|
2133
|
+
* Rebuilds are serialized per instance and never reject: the MiniSearch index
|
|
2134
|
+
* is a rebuildable cache over SQLite, so degraded keyword search is the
|
|
2135
|
+
* correct failure mode and killing the host process is not.
|
|
2079
2136
|
*/
|
|
2080
2137
|
async sync(entityId) {
|
|
2081
|
-
|
|
2082
|
-
|
|
2138
|
+
const work = this.syncChain.then(async () => {
|
|
2139
|
+
try {
|
|
2140
|
+
try {
|
|
2141
|
+
await this.rebuildIndex(entityId);
|
|
2142
|
+
await this.miniSearch.vacuum();
|
|
2143
|
+
} finally {
|
|
2144
|
+
this.evictCache(entityId);
|
|
2145
|
+
}
|
|
2146
|
+
} catch (err) {
|
|
2147
|
+
console.warn(`[WikiMemory] search index rebuild failed for ${entityId ?? "*"}:`, err);
|
|
2148
|
+
}
|
|
2149
|
+
});
|
|
2150
|
+
this.syncChain = work;
|
|
2151
|
+
return work;
|
|
2083
2152
|
}
|
|
2084
2153
|
/**
|
|
2085
2154
|
* Clears the parsed vector cache. Useful for mid-loop flush guarantees
|
|
@@ -3067,12 +3136,122 @@ var IngestionService = class {
|
|
|
3067
3136
|
}
|
|
3068
3137
|
};
|
|
3069
3138
|
|
|
3139
|
+
// src/services/BoundedLlmCall.ts
|
|
3140
|
+
var DEFAULT_BATCH_SIZE = 10;
|
|
3141
|
+
var ESTIMATED_OUTPUT_TOKENS_PER_ITEM = 150;
|
|
3142
|
+
var OUTPUT_BUDGET_FRACTION = 0.8;
|
|
3143
|
+
var TRUNCATION_PATTERNS = [
|
|
3144
|
+
/truncat/i,
|
|
3145
|
+
/token limit/i,
|
|
3146
|
+
/max(imum)?[ _-]?tokens?/i,
|
|
3147
|
+
/output limit/i,
|
|
3148
|
+
/length limit/i,
|
|
3149
|
+
/finish[_ ]?reason/i
|
|
3150
|
+
];
|
|
3151
|
+
var EXCEEDS_LIMIT_PATTERN = /exceed[a-z]*[^.]{0,40}\b(model|context)?[ _-]?limit/i;
|
|
3152
|
+
function isTruncationError(err) {
|
|
3153
|
+
const message = err instanceof Error ? err.message : String(err ?? "");
|
|
3154
|
+
if (EXCEEDS_LIMIT_PATTERN.test(message)) return false;
|
|
3155
|
+
return TRUNCATION_PATTERNS.some((pattern) => pattern.test(message));
|
|
3156
|
+
}
|
|
3157
|
+
function initialBatchSize(maxOutputTokens) {
|
|
3158
|
+
if (!maxOutputTokens || !Number.isFinite(maxOutputTokens) || maxOutputTokens <= 0) {
|
|
3159
|
+
return DEFAULT_BATCH_SIZE;
|
|
3160
|
+
}
|
|
3161
|
+
const estimate = Math.floor(
|
|
3162
|
+
maxOutputTokens * OUTPUT_BUDGET_FRACTION / ESTIMATED_OUTPUT_TOKENS_PER_ITEM
|
|
3163
|
+
);
|
|
3164
|
+
return Math.max(DEFAULT_BATCH_SIZE, estimate);
|
|
3165
|
+
}
|
|
3166
|
+
var promptLength = (prompts) => prompts.systemPrompt.length + prompts.userPrompt.length;
|
|
3167
|
+
async function runBatched(args) {
|
|
3168
|
+
const { items, buildPrompt, call, parse, maxPromptChars, maxOutputTokens, onSkip } = args;
|
|
3169
|
+
const results = [];
|
|
3170
|
+
const skipped = [];
|
|
3171
|
+
let batches = 0;
|
|
3172
|
+
let batchSize = initialBatchSize(maxOutputTokens);
|
|
3173
|
+
const trim = async (candidate) => {
|
|
3174
|
+
const whole = await buildPrompt(candidate);
|
|
3175
|
+
if (candidate.length <= 1 || promptLength(whole) <= maxPromptChars) {
|
|
3176
|
+
return { batch: candidate, prompts: whole };
|
|
3177
|
+
}
|
|
3178
|
+
let low = 2;
|
|
3179
|
+
let high = candidate.length - 1;
|
|
3180
|
+
let best;
|
|
3181
|
+
let bestPrompts;
|
|
3182
|
+
while (low <= high) {
|
|
3183
|
+
const mid = Math.floor((low + high) / 2);
|
|
3184
|
+
const batch = candidate.slice(0, mid);
|
|
3185
|
+
const prompts = await buildPrompt(batch);
|
|
3186
|
+
if (promptLength(prompts) <= maxPromptChars) {
|
|
3187
|
+
best = batch;
|
|
3188
|
+
bestPrompts = prompts;
|
|
3189
|
+
low = mid + 1;
|
|
3190
|
+
} else {
|
|
3191
|
+
high = mid - 1;
|
|
3192
|
+
}
|
|
3193
|
+
}
|
|
3194
|
+
if (best && bestPrompts) return { batch: best, prompts: bestPrompts };
|
|
3195
|
+
const single = candidate.slice(0, 1);
|
|
3196
|
+
return { batch: single, prompts: await buildPrompt(single) };
|
|
3197
|
+
};
|
|
3198
|
+
const onFailure = async (batch, err) => {
|
|
3199
|
+
if (batch.length <= 1) {
|
|
3200
|
+
if (batch.length === 1) {
|
|
3201
|
+
skipped.push(batch[0]);
|
|
3202
|
+
onSkip?.(batch[0], err);
|
|
3203
|
+
}
|
|
3204
|
+
return;
|
|
3205
|
+
}
|
|
3206
|
+
const mid = Math.ceil(batch.length / 2);
|
|
3207
|
+
if (mid < batchSize) batchSize = mid;
|
|
3208
|
+
let i = 0;
|
|
3209
|
+
while (i < batch.length) {
|
|
3210
|
+
const size = Math.min(batchSize, batch.length - i);
|
|
3211
|
+
const trimmed = await trim(batch.slice(i, i + size));
|
|
3212
|
+
await attempt(trimmed.batch, trimmed.prompts);
|
|
3213
|
+
i += trimmed.batch.length;
|
|
3214
|
+
}
|
|
3215
|
+
};
|
|
3216
|
+
const attempt = async (batch, prebuilt) => {
|
|
3217
|
+
if (batch.length === 0) return;
|
|
3218
|
+
const prompts = prebuilt ?? await buildPrompt(batch);
|
|
3219
|
+
batches++;
|
|
3220
|
+
let responseText;
|
|
3221
|
+
try {
|
|
3222
|
+
responseText = await call(prompts);
|
|
3223
|
+
} catch (err) {
|
|
3224
|
+
if (!isTruncationError(err)) throw err;
|
|
3225
|
+
await onFailure(batch, err);
|
|
3226
|
+
return;
|
|
3227
|
+
}
|
|
3228
|
+
let result;
|
|
3229
|
+
try {
|
|
3230
|
+
result = parse(responseText, batch);
|
|
3231
|
+
} catch (err) {
|
|
3232
|
+
await onFailure(batch, err);
|
|
3233
|
+
return;
|
|
3234
|
+
}
|
|
3235
|
+
results.push(result);
|
|
3236
|
+
};
|
|
3237
|
+
let index = 0;
|
|
3238
|
+
while (index < items.length) {
|
|
3239
|
+
const { batch, prompts } = await trim(items.slice(index, index + batchSize));
|
|
3240
|
+
index += batch.length;
|
|
3241
|
+
await attempt(batch, prompts);
|
|
3242
|
+
}
|
|
3243
|
+
return { results, skipped, batches };
|
|
3244
|
+
}
|
|
3245
|
+
|
|
3070
3246
|
// src/services/MaintenanceService.ts
|
|
3071
3247
|
var FUZZY_THRESHOLD = 0.5;
|
|
3072
3248
|
var MIN_TOKENS_TO_QUALIFY = 3;
|
|
3073
3249
|
var ONTOLOGY_BACKFILL_BATCH_SIZE = 25;
|
|
3074
3250
|
var ONTOLOGY_BACKFILL_MAX_PROMPT_CHARS = 4e4;
|
|
3075
3251
|
var ONTOLOGY_BACKFILL_RECHECK_MS = 7 * 24 * 60 * 60 * 1e3;
|
|
3252
|
+
var HEAL_MAX_ANCHORS = 50;
|
|
3253
|
+
var HEAL_ANCHOR_SEARCH_OVERFETCH = 4;
|
|
3254
|
+
var HEAL_MAX_PROMPT_CHARS = 4e4;
|
|
3076
3255
|
var MaintenanceService = class {
|
|
3077
3256
|
constructor(db, prefix, options, entryRepo, taskRepo, eventRepo, metadataRepo, searchService, jobManager, embeddingService, promptService, ontologyService) {
|
|
3078
3257
|
this.db = db;
|
|
@@ -3455,30 +3634,56 @@ var MaintenanceService = class {
|
|
|
3455
3634
|
console.warn(`[WikiMemory] onEmbeddingPersisted hook failed during heal orphan pass for ${factId}:`, hookErr);
|
|
3456
3635
|
}
|
|
3457
3636
|
}
|
|
3458
|
-
const
|
|
3637
|
+
const healCandidates = await this.entryRepo.findHealCandidatesByEntityId(entityId);
|
|
3459
3638
|
const allTasks = await this.taskRepo.findAllPending([entityId]);
|
|
3460
3639
|
const recentEvents = await this.eventRepo.getRecent(entityId, 20);
|
|
3461
|
-
const
|
|
3462
|
-
const documentAnchors = allFactsRows.filter((f) => f.source_type === "immutable_document").map(({ id, title, source_ref }) => ({ id, title, source_ref }));
|
|
3463
|
-
const healCandidatesForPrompt = healCandidates.map((f) => {
|
|
3640
|
+
const toPromptShape = (f) => {
|
|
3464
3641
|
const { embedding: _embedding, embedding_blob: _blob, ...rest } = f;
|
|
3465
3642
|
return { ...rest, tags: typeof rest.tags === "string" ? JSON.parse(rest.tags) : rest.tags };
|
|
3643
|
+
};
|
|
3644
|
+
const anchorCache = /* @__PURE__ */ new Map();
|
|
3645
|
+
const outcome = await runBatched({
|
|
3646
|
+
items: healCandidates,
|
|
3647
|
+
buildPrompt: async (batch) => {
|
|
3648
|
+
const documentAnchors = await this._selectHealAnchors(entityId, batch, anchorCache);
|
|
3649
|
+
return this.promptService.buildHealPrompt(
|
|
3650
|
+
batch.map(toPromptShape),
|
|
3651
|
+
documentAnchors,
|
|
3652
|
+
allTasks,
|
|
3653
|
+
recentEvents,
|
|
3654
|
+
promptOverride
|
|
3655
|
+
);
|
|
3656
|
+
},
|
|
3657
|
+
call: (prompts) => this.options.llmProvider.generateText(prompts),
|
|
3658
|
+
parse: (responseText, batch) => {
|
|
3659
|
+
const result = parseJsonResponse(responseText);
|
|
3660
|
+
return {
|
|
3661
|
+
batch,
|
|
3662
|
+
downgraded: Array.isArray(result.downgraded) ? result.downgraded : [],
|
|
3663
|
+
deleted: Array.isArray(result.deleted) ? result.deleted : [],
|
|
3664
|
+
newFacts: Array.isArray(result.newFacts) ? result.newFacts : []
|
|
3665
|
+
};
|
|
3666
|
+
},
|
|
3667
|
+
maxOutputTokens: this.options.llmProvider.maxOutputTokens,
|
|
3668
|
+
maxPromptChars: HEAL_MAX_PROMPT_CHARS,
|
|
3669
|
+
onSkip: (fact, err) => {
|
|
3670
|
+
console.warn(
|
|
3671
|
+
`[WikiMemory] heal skipped ${entityId}/${fact.id}: response could not be bounded`,
|
|
3672
|
+
err
|
|
3673
|
+
);
|
|
3674
|
+
}
|
|
3466
3675
|
});
|
|
3467
|
-
const
|
|
3468
|
-
|
|
3469
|
-
|
|
3470
|
-
|
|
3471
|
-
|
|
3472
|
-
|
|
3473
|
-
|
|
3474
|
-
|
|
3475
|
-
|
|
3476
|
-
const
|
|
3477
|
-
const
|
|
3478
|
-
const deleted = Array.isArray(result.deleted) ? result.deleted : [];
|
|
3479
|
-
const newFacts = Array.isArray(result.newFacts) ? result.newFacts : [];
|
|
3480
|
-
const safeDowngraded = Array.from(new Set(downgraded.filter((id) => mutableIds.has(id))));
|
|
3481
|
-
const safeDeleted = Array.from(new Set(deleted.filter((id) => mutableIds.has(id))));
|
|
3676
|
+
const safeDowngradedSet = /* @__PURE__ */ new Set();
|
|
3677
|
+
const safeDeletedSet = /* @__PURE__ */ new Set();
|
|
3678
|
+
const newFacts = [];
|
|
3679
|
+
for (const batchResult of outcome.results) {
|
|
3680
|
+
const mutableIds = new Set(batchResult.batch.map((f) => f.id));
|
|
3681
|
+
for (const id of batchResult.downgraded) if (mutableIds.has(id)) safeDowngradedSet.add(id);
|
|
3682
|
+
for (const id of batchResult.deleted) if (mutableIds.has(id)) safeDeletedSet.add(id);
|
|
3683
|
+
newFacts.push(...batchResult.newFacts);
|
|
3684
|
+
}
|
|
3685
|
+
const safeDowngraded = Array.from(safeDowngradedSet);
|
|
3686
|
+
const safeDeleted = Array.from(safeDeletedSet);
|
|
3482
3687
|
const validNewFacts = newFacts.map(validateFact).filter((f) => f !== null);
|
|
3483
3688
|
const insertedFacts = [];
|
|
3484
3689
|
const uniqueDeletedFactIds = Array.from(new Set(safeDeleted));
|
|
@@ -3545,7 +3750,7 @@ var MaintenanceService = class {
|
|
|
3545
3750
|
}
|
|
3546
3751
|
const now = Date.now();
|
|
3547
3752
|
const recheckCutoff = now - ONTOLOGY_BACKFILL_RECHECK_MS;
|
|
3548
|
-
const zeroed = { scanned: 0, typed: 0, failedValidation: 0, edgesAdded: 0 };
|
|
3753
|
+
const zeroed = { scanned: 0, typed: 0, failedValidation: 0, edgesAdded: 0, skipped: 0 };
|
|
3549
3754
|
const ontologyService = this.ontologyService;
|
|
3550
3755
|
if (!ontologyService) {
|
|
3551
3756
|
return { ...zeroed, remaining: 0, deferred: 0 };
|
|
@@ -3567,18 +3772,79 @@ var MaintenanceService = class {
|
|
|
3567
3772
|
options?.promptOverride,
|
|
3568
3773
|
ontologyContext
|
|
3569
3774
|
);
|
|
3570
|
-
const
|
|
3571
|
-
|
|
3572
|
-
|
|
3573
|
-
|
|
3574
|
-
|
|
3575
|
-
|
|
3576
|
-
|
|
3577
|
-
|
|
3578
|
-
|
|
3579
|
-
|
|
3580
|
-
|
|
3581
|
-
|
|
3775
|
+
const outcome = await runBatched({
|
|
3776
|
+
items: candidates,
|
|
3777
|
+
buildPrompt,
|
|
3778
|
+
call: (prompts) => this.options.llmProvider.generateText(prompts),
|
|
3779
|
+
parse: (responseText, batch) => {
|
|
3780
|
+
const parsed = parseJsonResponse(responseText);
|
|
3781
|
+
return {
|
|
3782
|
+
batch,
|
|
3783
|
+
classifications: Array.isArray(parsed.classifications) ? parsed.classifications : [],
|
|
3784
|
+
ontologyUpdates: parsed.ontology_updates
|
|
3785
|
+
};
|
|
3786
|
+
},
|
|
3787
|
+
maxOutputTokens: this.options.llmProvider.maxOutputTokens,
|
|
3788
|
+
maxPromptChars: ONTOLOGY_BACKFILL_MAX_PROMPT_CHARS,
|
|
3789
|
+
onSkip: (fact, err) => {
|
|
3790
|
+
console.warn(
|
|
3791
|
+
`[WikiMemory] ontology backfill skipped ${entityId}/${fact.id}: response could not be bounded`,
|
|
3792
|
+
err
|
|
3793
|
+
);
|
|
3794
|
+
}
|
|
3795
|
+
});
|
|
3796
|
+
let typed = 0;
|
|
3797
|
+
let failedValidation = 0;
|
|
3798
|
+
let edgesAdded = 0;
|
|
3799
|
+
let scanned = 0;
|
|
3800
|
+
let abortedOntologyOff = false;
|
|
3801
|
+
for (const batchResult of outcome.results) {
|
|
3802
|
+
const applied = await this._applyOntologyBackfillBatch(entityId, batchResult, now);
|
|
3803
|
+
if (applied.abortedOntologyOff) {
|
|
3804
|
+
abortedOntologyOff = true;
|
|
3805
|
+
break;
|
|
3806
|
+
}
|
|
3807
|
+
typed += applied.typed;
|
|
3808
|
+
failedValidation += applied.failedValidation;
|
|
3809
|
+
edgesAdded += applied.edgesAdded;
|
|
3810
|
+
scanned += batchResult.batch.length;
|
|
3811
|
+
}
|
|
3812
|
+
if (abortedOntologyOff) {
|
|
3813
|
+
const counts2 = await this.entryRepo.countUntypedByEntityId(entityId, recheckCutoff);
|
|
3814
|
+
return {
|
|
3815
|
+
scanned,
|
|
3816
|
+
typed,
|
|
3817
|
+
failedValidation,
|
|
3818
|
+
edgesAdded,
|
|
3819
|
+
skipped: outcome.skipped.length,
|
|
3820
|
+
remaining: 0,
|
|
3821
|
+
deferred: counts2.deferred
|
|
3822
|
+
};
|
|
3823
|
+
}
|
|
3824
|
+
if (outcome.skipped.length > 0) {
|
|
3825
|
+
await this.entryRepo.markOntologyChecked(outcome.skipped.map((f) => f.id), entityId, now, this.db);
|
|
3826
|
+
}
|
|
3827
|
+
this.searchService.evictCache(entityId);
|
|
3828
|
+
const counts = await this.entryRepo.countUntypedByEntityId(entityId, recheckCutoff);
|
|
3829
|
+
return {
|
|
3830
|
+
scanned,
|
|
3831
|
+
typed,
|
|
3832
|
+
failedValidation,
|
|
3833
|
+
edgesAdded,
|
|
3834
|
+
skipped: outcome.skipped.length,
|
|
3835
|
+
remaining: counts.eligible,
|
|
3836
|
+
deferred: counts.deferred
|
|
3837
|
+
};
|
|
3838
|
+
}
|
|
3839
|
+
/**
|
|
3840
|
+
* Applies one parsed backfill batch in its own transaction. Per-batch rather
|
|
3841
|
+
* than one transaction for the pass, so mergeEmergentUpdates semantics and
|
|
3842
|
+
* the mid-flight `mode === 'off'` abort check keep the shape they had when a
|
|
3843
|
+
* pass was a single call.
|
|
3844
|
+
*/
|
|
3845
|
+
async _applyOntologyBackfillBatch(entityId, batchResult, now) {
|
|
3846
|
+
const ontologyService = this.ontologyService;
|
|
3847
|
+
const { batch, classifications, ontologyUpdates } = batchResult;
|
|
3582
3848
|
let typed = 0;
|
|
3583
3849
|
let failedValidation = 0;
|
|
3584
3850
|
let edgesAdded = 0;
|
|
@@ -3589,8 +3855,8 @@ var MaintenanceService = class {
|
|
|
3589
3855
|
abortedOntologyOff = true;
|
|
3590
3856
|
return;
|
|
3591
3857
|
}
|
|
3592
|
-
if (txMode === "emergent" &&
|
|
3593
|
-
manifest = await ontologyService.mergeEmergentUpdates(entityId,
|
|
3858
|
+
if (txMode === "emergent" && ontologyUpdates) {
|
|
3859
|
+
manifest = await ontologyService.mergeEmergentUpdates(entityId, ontologyUpdates, tx);
|
|
3594
3860
|
}
|
|
3595
3861
|
const titleRows = await this.entryRepo.findTitleIndexByEntityId(entityId, tx);
|
|
3596
3862
|
const titleIndex = /* @__PURE__ */ new Map();
|
|
@@ -3644,19 +3910,58 @@ var MaintenanceService = class {
|
|
|
3644
3910
|
}
|
|
3645
3911
|
await this.entryRepo.markOntologyChecked(batch.map((f) => f.id), entityId, now, tx);
|
|
3646
3912
|
});
|
|
3647
|
-
|
|
3648
|
-
const counts2 = await this.entryRepo.countUntypedByEntityId(entityId, recheckCutoff);
|
|
3649
|
-
return { ...zeroed, remaining: 0, deferred: counts2.deferred };
|
|
3650
|
-
}
|
|
3651
|
-
this.searchService.evictCache(entityId);
|
|
3652
|
-
const counts = await this.entryRepo.countUntypedByEntityId(entityId, recheckCutoff);
|
|
3653
|
-
return { scanned: batch.length, typed, failedValidation, edgesAdded, remaining: counts.eligible, deferred: counts.deferred };
|
|
3913
|
+
return { typed, failedValidation, edgesAdded, abortedOntologyOff };
|
|
3654
3914
|
}
|
|
3655
3915
|
_validatePruneDuration(value, name) {
|
|
3656
3916
|
if (value !== null && value !== void 0 && (typeof value !== "number" || !isFinite(value) || value < 0)) {
|
|
3657
3917
|
throw new Error(`Invalid ${name}: must be a non-negative finite number or null`);
|
|
3658
3918
|
}
|
|
3659
3919
|
}
|
|
3920
|
+
/**
|
|
3921
|
+
* Anchors relevant to one batch of heal candidates.
|
|
3922
|
+
*
|
|
3923
|
+
* Heal used to pass every immutable_document fact for the entity — 2560 rows
|
|
3924
|
+
* against 31 candidates on the corpus behind #63 — which is what blew the
|
|
3925
|
+
* output ceiling. Anchors are now retrieved by keyword relevance to the batch
|
|
3926
|
+
* and capped.
|
|
3927
|
+
*
|
|
3928
|
+
* The MiniSearch index holds all facts, not only anchors, so hits are
|
|
3929
|
+
* overfetched and the source_type restriction is applied after retrieval, in
|
|
3930
|
+
* SQL. Search rank order is preserved through the filter.
|
|
3931
|
+
*
|
|
3932
|
+
* Accepted tradeoff: an anchor that contradicts a candidate while sharing no
|
|
3933
|
+
* vocabulary with it is now missed. Exhaustive-but-broken traded for
|
|
3934
|
+
* relevance-scoped-and-working.
|
|
3935
|
+
*
|
|
3936
|
+
* `cache` is keyed by the derived query rather than by the batch, so two
|
|
3937
|
+
* batches that reduce to the same query share one lookup. Caller-owned and
|
|
3938
|
+
* per-pass — see the call site in doRunHeal.
|
|
3939
|
+
*/
|
|
3940
|
+
async _selectHealAnchors(entityId, batch, cache) {
|
|
3941
|
+
const query = batch.map((f) => f.title).join(" ").trim();
|
|
3942
|
+
if (!query) return [];
|
|
3943
|
+
const cached = cache?.get(query);
|
|
3944
|
+
if (cached) return cached;
|
|
3945
|
+
const hits = this.searchService.searchKeyword(
|
|
3946
|
+
query,
|
|
3947
|
+
[entityId],
|
|
3948
|
+
HEAL_MAX_ANCHORS * HEAL_ANCHOR_SEARCH_OVERFETCH
|
|
3949
|
+
);
|
|
3950
|
+
const hitIds = hits.map((h) => h.id);
|
|
3951
|
+
const anchors = [];
|
|
3952
|
+
if (hitIds.length > 0) {
|
|
3953
|
+
const rows = await this.entryRepo.findAnchorRowsByIds(entityId, hitIds);
|
|
3954
|
+
const byId = new Map(rows.map((r) => [r.id, r]));
|
|
3955
|
+
for (const id of hitIds) {
|
|
3956
|
+
const row = byId.get(id);
|
|
3957
|
+
if (!row) continue;
|
|
3958
|
+
anchors.push(row);
|
|
3959
|
+
if (anchors.length >= HEAL_MAX_ANCHORS) break;
|
|
3960
|
+
}
|
|
3961
|
+
}
|
|
3962
|
+
cache?.set(query, anchors);
|
|
3963
|
+
return anchors;
|
|
3964
|
+
}
|
|
3660
3965
|
_sanitizeRankerError(err) {
|
|
3661
3966
|
return sanitizeRankerError(err, this.options.sanitizeRankerErrors);
|
|
3662
3967
|
}
|