@equationalapplications/core-llm-wiki 5.1.0 → 5.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/{chunk-DM7BKDCV.mjs → chunk-J3N3WRK7.mjs} +85 -8
- package/dist/chunk-J3N3WRK7.mjs.map +1 -0
- package/dist/index.d.mts +2 -2
- package/dist/index.d.ts +2 -2
- package/dist/index.js +305 -23
- package/dist/index.js.map +1 -1
- package/dist/index.mjs +224 -20
- package/dist/index.mjs.map +1 -1
- package/dist/{testing-DCRNie7k.d.mts → testing-DyXISRwS.d.mts} +137 -1
- package/dist/{testing-DCRNie7k.d.ts → testing-DyXISRwS.d.ts} +137 -1
- package/dist/testing.d.mts +1 -1
- package/dist/testing.d.ts +1 -1
- package/dist/testing.js +82 -5
- package/dist/testing.js.map +1 -1
- package/dist/testing.mjs +1 -1
- package/package.json +2 -2
- package/dist/chunk-DM7BKDCV.mjs.map +0 -1
package/dist/index.mjs
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
import { __privateAdd, MetadataRepository, EmbeddingService, SearchService, JobManager, PromptService, IngestionService, MaintenanceService, ImportExportService, RetrievalService, WriteService, __privateGet, __privateSet, normalizeSourceRef, normalizeSourceHash, entitySummaryMetaKey, generateId, BaseRepository, emptyManifest, resolveNodeType, validateInlineEdges, resolveEdgeDefinitions, normalizeTitleKey, WikiTransactionError, extractSqliteCode } from './chunk-
|
|
2
|
-
export { DEFAULT_CHUNK_OVERLAP, DEFAULT_MAX_CHUNK_LENGTH, HEAL_BATCH_SIZE, HEAL_RECHECK_MS, HOOK_TIMEOUT_MARKER, ONTOLOGY_BACKFILL_BATCH_SIZE, ONTOLOGY_BACKFILL_MAX_PROMPT_CHARS, ONTOLOGY_BACKFILL_RECHECK_MS, ONTOLOGY_BACKFILL_SYSTEM_PROMPT, PromptService, PrunePartialFailureError, WikiBusyError, WikiTransactionError, chunkText, configureRandomSource, parseEmbedding, safeSlice, validateManifest } from './chunk-
|
|
1
|
+
import { __privateAdd, MetadataRepository, EmbeddingService, SearchService, JobManager, PromptService, IngestionService, MaintenanceService, ImportExportService, RetrievalService, WriteService, __privateGet, __privateSet, normalizeSourceRef, normalizeSourceHash, entitySummaryMetaKey, generateId, BaseRepository, emptyManifest, resolveNodeType, validateInlineEdges, resolveEdgeDefinitions, normalizeTitleKey, WikiTransactionError, extractSqliteCode } from './chunk-J3N3WRK7.mjs';
|
|
2
|
+
export { DEFAULT_CHUNK_OVERLAP, DEFAULT_MAX_CHUNK_LENGTH, HEAL_BATCH_SIZE, HEAL_RECHECK_MS, HOOK_TIMEOUT_MARKER, ONTOLOGY_BACKFILL_BATCH_SIZE, ONTOLOGY_BACKFILL_MAX_PROMPT_CHARS, ONTOLOGY_BACKFILL_RECHECK_MS, ONTOLOGY_BACKFILL_SYSTEM_PROMPT, PromptService, PrunePartialFailureError, WikiBusyError, WikiDuplicateHashError, WikiTransactionError, chunkText, configureRandomSource, parseEmbedding, safeSlice, validateManifest } from './chunk-J3N3WRK7.mjs';
|
|
3
3
|
import { appendRelatedSection, buildConceptDocument, buildLogMd, buildEntityIndexMd, buildRootIndexMd, isAllowedOkfPath, parseRootIndexMd, parseEntityIndexMd, parseConcept, splitRelatedSection, extractMarkdownLinks, parseLogMd, parseEventIdComment, appendEventIdComment } from '@equationalapplications/core-okf';
|
|
4
4
|
|
|
5
5
|
// src/db/schema.ts
|
|
@@ -975,12 +975,154 @@ var EntryRepository = class extends BaseRepository {
|
|
|
975
975
|
const row = await executor.getFirstAsync(
|
|
976
976
|
`SELECT source_hash FROM ${this.prefix}entries
|
|
977
977
|
WHERE entity_id = ? AND source_ref = ? AND deleted_at IS NULL
|
|
978
|
-
ORDER BY updated_at DESC
|
|
978
|
+
ORDER BY updated_at DESC, id ASC
|
|
979
979
|
LIMIT 1`,
|
|
980
980
|
[entityId, sourceRef]
|
|
981
981
|
);
|
|
982
982
|
return row?.source_hash ?? null;
|
|
983
983
|
}
|
|
984
|
+
/**
|
|
985
|
+
* Batch version of {@link findLatestSourceHash}. Returns a Map covering every
|
|
986
|
+
* requested ref, where the value is the source_hash from the row with the
|
|
987
|
+
* most-recently-updated live fact for that ref, or null when no live row exists.
|
|
988
|
+
*
|
|
989
|
+
* The SQL uses ROW_NUMBER() OVER (PARTITION BY source_ref ORDER BY updated_at
|
|
990
|
+
* DESC, id ASC) — NOT MAX(source_hash). Aggregation with MAX(source_hash) is
|
|
991
|
+
* wrong because MAX computes independently across grouped rows; the hash must
|
|
992
|
+
* come from the exact row that wins MAX(updated_at). The `id ASC` tie-break
|
|
993
|
+
* keeps selection deterministic when two live rows share `updated_at`.
|
|
994
|
+
*
|
|
995
|
+
* Source refs are de-duplicated and processed in chunks so the per-query bind
|
|
996
|
+
* parameter count stays under SQLite's `SQLITE_MAX_VARIABLE_NUMBER` (default
|
|
997
|
+
* 999, leaving one slot for `entity_id`).
|
|
998
|
+
*
|
|
999
|
+
* Empty input is a synchronous early return with zero SQL calls.
|
|
1000
|
+
*/
|
|
1001
|
+
async findLatestSourceHashes(entityId, sourceRefs, tx) {
|
|
1002
|
+
const out = /* @__PURE__ */ new Map();
|
|
1003
|
+
if (sourceRefs.length === 0) return out;
|
|
1004
|
+
const dedupedRefs = Array.from(new Set(sourceRefs));
|
|
1005
|
+
for (const ref of dedupedRefs) {
|
|
1006
|
+
out.set(ref, null);
|
|
1007
|
+
}
|
|
1008
|
+
const executor = this.getExecutor(tx);
|
|
1009
|
+
const chunkLimit = Math.max(1, this.chunkSize - 1);
|
|
1010
|
+
for (let i = 0; i < dedupedRefs.length; i += chunkLimit) {
|
|
1011
|
+
const chunk = dedupedRefs.slice(i, i + chunkLimit);
|
|
1012
|
+
const placeholders = chunk.map(() => "?").join(",");
|
|
1013
|
+
const rows = await executor.getAllAsync(
|
|
1014
|
+
`WITH ranked AS (
|
|
1015
|
+
SELECT source_ref, source_hash,
|
|
1016
|
+
ROW_NUMBER() OVER (
|
|
1017
|
+
PARTITION BY source_ref
|
|
1018
|
+
ORDER BY updated_at DESC, id ASC
|
|
1019
|
+
) as rn
|
|
1020
|
+
FROM ${this.prefix}entries
|
|
1021
|
+
WHERE entity_id = ? AND source_ref IN (${placeholders}) AND deleted_at IS NULL
|
|
1022
|
+
)
|
|
1023
|
+
SELECT source_ref, source_hash
|
|
1024
|
+
FROM ranked
|
|
1025
|
+
WHERE rn = 1`,
|
|
1026
|
+
[entityId, ...chunk]
|
|
1027
|
+
);
|
|
1028
|
+
for (const r of rows) {
|
|
1029
|
+
out.set(r.source_ref, r.source_hash);
|
|
1030
|
+
}
|
|
1031
|
+
}
|
|
1032
|
+
return out;
|
|
1033
|
+
}
|
|
1034
|
+
/**
|
|
1035
|
+
* Return the live source_refs for an entity that hold the given source_hash.
|
|
1036
|
+
* Used by the ingestDocument duplicate-hash guard and by hosts auditing
|
|
1037
|
+
* duplicate-content collisions. Sorted `COLLATE BINARY` so the canonical
|
|
1038
|
+
* ref (the code-unit-minimum of the set) is stable across deploys.
|
|
1039
|
+
*/
|
|
1040
|
+
async findSourceRefsByHash(entityId, sourceHash, tx) {
|
|
1041
|
+
const executor = this.getExecutor(tx);
|
|
1042
|
+
const rows = await executor.getAllAsync(
|
|
1043
|
+
`SELECT source_ref FROM ${this.prefix}entries
|
|
1044
|
+
WHERE entity_id = ? AND source_hash = ? AND deleted_at IS NULL
|
|
1045
|
+
AND source_ref IS NOT NULL
|
|
1046
|
+
GROUP BY source_ref
|
|
1047
|
+
ORDER BY source_ref COLLATE BINARY`,
|
|
1048
|
+
[entityId, sourceHash]
|
|
1049
|
+
);
|
|
1050
|
+
return rows.map((r) => r.source_ref);
|
|
1051
|
+
}
|
|
1052
|
+
/**
|
|
1053
|
+
* Per-sourceRef rollup for an entity: one row per live source_ref, with the
|
|
1054
|
+
* most-recently-updated live hash (NOT the lexically-max hash) and a live
|
|
1055
|
+
* fact count.
|
|
1056
|
+
*
|
|
1057
|
+
* The hash comes from the same row that wins ROW_NUMBER() over updated_at
|
|
1058
|
+
* DESC, so a ref with multiple live hashes (e.g. from import) reports the
|
|
1059
|
+
* latest one — matching single-doc `findLatestSourceHash` semantics.
|
|
1060
|
+
*
|
|
1061
|
+
* Sort is COLLATE BINARY: locale-dependent ordering re-mints identity on
|
|
1062
|
+
* every deploy, which is the bug this whole section exists to prevent.
|
|
1063
|
+
*/
|
|
1064
|
+
async listSourceRefs(entityId, tx) {
|
|
1065
|
+
const executor = this.getExecutor(tx);
|
|
1066
|
+
const rows = await executor.getAllAsync(
|
|
1067
|
+
`WITH ranked AS (
|
|
1068
|
+
SELECT source_ref, source_hash, updated_at,
|
|
1069
|
+
ROW_NUMBER() OVER (
|
|
1070
|
+
PARTITION BY source_ref
|
|
1071
|
+
ORDER BY updated_at DESC, id ASC
|
|
1072
|
+
) AS rn,
|
|
1073
|
+
COUNT(*) OVER (PARTITION BY source_ref) AS fact_count
|
|
1074
|
+
FROM ${this.prefix}entries
|
|
1075
|
+
WHERE entity_id = ? AND deleted_at IS NULL AND source_ref IS NOT NULL
|
|
1076
|
+
)
|
|
1077
|
+
SELECT source_ref,
|
|
1078
|
+
source_hash AS source_hash,
|
|
1079
|
+
fact_count AS fact_count,
|
|
1080
|
+
updated_at AS last_ingested_at
|
|
1081
|
+
FROM ranked
|
|
1082
|
+
WHERE rn = 1
|
|
1083
|
+
ORDER BY source_ref COLLATE BINARY`,
|
|
1084
|
+
[entityId]
|
|
1085
|
+
);
|
|
1086
|
+
return rows.map((r) => ({
|
|
1087
|
+
sourceRef: r.source_ref,
|
|
1088
|
+
sourceHash: r.source_hash,
|
|
1089
|
+
factCount: Number(r.fact_count),
|
|
1090
|
+
lastIngestedAt: Number(r.last_ingested_at)
|
|
1091
|
+
}));
|
|
1092
|
+
}
|
|
1093
|
+
/**
|
|
1094
|
+
* Count live entries for an entity across all source_refs/source_hashes.
|
|
1095
|
+
* Used by forget({ dryRun, clearAll: true }).
|
|
1096
|
+
*/
|
|
1097
|
+
async countLiveByEntityId(entityId, tx) {
|
|
1098
|
+
const executor = this.getExecutor(tx);
|
|
1099
|
+
const row = await executor.getFirstAsync(
|
|
1100
|
+
`SELECT COUNT(*) AS cnt FROM ${this.prefix}entries
|
|
1101
|
+
WHERE entity_id = ? AND deleted_at IS NULL`,
|
|
1102
|
+
[entityId]
|
|
1103
|
+
);
|
|
1104
|
+
return row?.cnt ?? 0;
|
|
1105
|
+
}
|
|
1106
|
+
/**
|
|
1107
|
+
* Count live entries matching a source filter (either or both may be null).
|
|
1108
|
+
* Used by forget({ dryRun: true }) with sourceRef/sourceHash.
|
|
1109
|
+
*/
|
|
1110
|
+
async countLiveBySource(entityId, sourceRef, sourceHash, tx) {
|
|
1111
|
+
const executor = this.getExecutor(tx);
|
|
1112
|
+
let q = `SELECT COUNT(*) AS cnt FROM ${this.prefix}entries
|
|
1113
|
+
WHERE entity_id = ? AND deleted_at IS NULL`;
|
|
1114
|
+
const args = [entityId];
|
|
1115
|
+
if (sourceRef !== null) {
|
|
1116
|
+
q += ` AND source_ref = ?`;
|
|
1117
|
+
args.push(sourceRef);
|
|
1118
|
+
}
|
|
1119
|
+
if (sourceHash !== null) {
|
|
1120
|
+
q += ` AND source_hash = ?`;
|
|
1121
|
+
args.push(sourceHash);
|
|
1122
|
+
}
|
|
1123
|
+
const row = await executor.getFirstAsync(q, args);
|
|
1124
|
+
return row?.cnt ?? 0;
|
|
1125
|
+
}
|
|
984
1126
|
async findMetadataByIds(ids, tx) {
|
|
985
1127
|
if (ids.length === 0) return [];
|
|
986
1128
|
const executor = this.getExecutor(tx);
|
|
@@ -1508,6 +1650,18 @@ var TaskRepository = class extends BaseRepository {
|
|
|
1508
1650
|
}
|
|
1509
1651
|
return result.changes;
|
|
1510
1652
|
}
|
|
1653
|
+
/**
|
|
1654
|
+
* Count live tasks for an entity. Used by forget({ dryRun, clearAll: true }).
|
|
1655
|
+
*/
|
|
1656
|
+
async countLiveByEntityId(entityId, tx) {
|
|
1657
|
+
const executor = this.getExecutor(tx);
|
|
1658
|
+
const row = await executor.getFirstAsync(
|
|
1659
|
+
`SELECT COUNT(*) AS cnt FROM ${this.prefix}tasks
|
|
1660
|
+
WHERE entity_id = ? AND deleted_at IS NULL`,
|
|
1661
|
+
[entityId]
|
|
1662
|
+
);
|
|
1663
|
+
return row?.cnt ?? 0;
|
|
1664
|
+
}
|
|
1511
1665
|
};
|
|
1512
1666
|
|
|
1513
1667
|
// src/repositories/EventRepository.ts
|
|
@@ -2058,19 +2212,69 @@ var WikiMemory = class {
|
|
|
2058
2212
|
});
|
|
2059
2213
|
await this.searchService.sync();
|
|
2060
2214
|
}
|
|
2061
|
-
async hasChanged(entityId,
|
|
2062
|
-
|
|
2063
|
-
|
|
2064
|
-
|
|
2065
|
-
|
|
2066
|
-
|
|
2067
|
-
|
|
2068
|
-
|
|
2069
|
-
|
|
2070
|
-
|
|
2071
|
-
|
|
2072
|
-
|
|
2073
|
-
|
|
2215
|
+
async hasChanged(entityId, sourceRefOrEntries, sourceHashArg) {
|
|
2216
|
+
if (typeof sourceRefOrEntries === "string") {
|
|
2217
|
+
const sourceRef = normalizeSourceRef(sourceRefOrEntries);
|
|
2218
|
+
if (!sourceRef) {
|
|
2219
|
+
throw new Error(`Invalid sourceRef: ${JSON.stringify(sourceRefOrEntries)}`);
|
|
2220
|
+
}
|
|
2221
|
+
const sourceHash = normalizeSourceHash(sourceHashArg);
|
|
2222
|
+
if (!sourceHash) {
|
|
2223
|
+
throw new Error("Invalid sourceHash: must be a 64-character hex string (normalized to lowercase)");
|
|
2224
|
+
}
|
|
2225
|
+
const storedHash = await this.entryRepo.findLatestSourceHash(entityId, sourceRef);
|
|
2226
|
+
if (storedHash === null) return true;
|
|
2227
|
+
const normalizedStoredHash = normalizeSourceHash(storedHash);
|
|
2228
|
+
return normalizedStoredHash !== sourceHash;
|
|
2229
|
+
}
|
|
2230
|
+
const entries = sourceRefOrEntries;
|
|
2231
|
+
if (entries.length === 0) return [];
|
|
2232
|
+
const normalized = entries.map((e) => {
|
|
2233
|
+
const r = normalizeSourceRef(e.sourceRef);
|
|
2234
|
+
if (!r) throw new Error(`Invalid sourceRef: ${JSON.stringify(e.sourceRef)}`);
|
|
2235
|
+
const h = normalizeSourceHash(e.sourceHash);
|
|
2236
|
+
if (!h) throw new Error("Invalid sourceHash: must be a 64-character hex string (normalized to lowercase)");
|
|
2237
|
+
return { rawSourceRef: e.sourceRef, sourceRef: r, sourceHash: h };
|
|
2238
|
+
});
|
|
2239
|
+
const latestHashes = await this.entryRepo.findLatestSourceHashes(entityId, normalized.map((e) => e.sourceRef));
|
|
2240
|
+
const distinctHashes = Array.from(new Set(normalized.map((e) => e.sourceHash)));
|
|
2241
|
+
const dupRefs = await Promise.all(
|
|
2242
|
+
distinctHashes.map((h) => this.entryRepo.findSourceRefsByHash(entityId, h))
|
|
2243
|
+
);
|
|
2244
|
+
const dupMap = /* @__PURE__ */ new Map();
|
|
2245
|
+
for (let i = 0; i < distinctHashes.length; i++) {
|
|
2246
|
+
dupMap.set(distinctHashes[i], dupRefs[i]);
|
|
2247
|
+
}
|
|
2248
|
+
return normalized.map((e) => {
|
|
2249
|
+
const stored = latestHashes.get(e.sourceRef);
|
|
2250
|
+
const changed = stored === void 0 || stored === null || normalizeSourceHash(stored) !== e.sourceHash;
|
|
2251
|
+
const allRefs = dupMap.get(e.sourceHash) ?? [];
|
|
2252
|
+
const others = allRefs.filter((r) => r !== e.sourceRef);
|
|
2253
|
+
if (others.length === 0) {
|
|
2254
|
+
return { sourceRef: e.rawSourceRef, changed };
|
|
2255
|
+
}
|
|
2256
|
+
const canonical = others[0];
|
|
2257
|
+
return { sourceRef: e.rawSourceRef, changed, duplicateOf: canonical };
|
|
2258
|
+
});
|
|
2259
|
+
}
|
|
2260
|
+
/**
|
|
2261
|
+
* Returns the live source_refs for an entity, one row per ref, with the most
|
|
2262
|
+
* recently-updated live `source_hash` and a live fact count. Refs are sorted
|
|
2263
|
+
* `COLLATE BINARY` (no locale dependency). Used by hosts to reconcile stored
|
|
2264
|
+
* state against a live source, or audit duplicate-hash collisions via
|
|
2265
|
+
* `findSourceRefsByHash`.
|
|
2266
|
+
*/
|
|
2267
|
+
async listSourceRefs(entityId) {
|
|
2268
|
+
return this.entryRepo.listSourceRefs(entityId);
|
|
2269
|
+
}
|
|
2270
|
+
/**
|
|
2271
|
+
* Returns the live source_refs for an entity that hold the given source_hash,
|
|
2272
|
+
* sorted `COLLATE BINARY` ascending. The first element is the canonical ref
|
|
2273
|
+
* under the code-unit-minimum rule (no locale dependency). Used by the
|
|
2274
|
+
* ingestDocument guard and by hosts auditing duplicate-content collisions.
|
|
2275
|
+
*/
|
|
2276
|
+
async findSourceRefsByHash(entityId, sourceHash) {
|
|
2277
|
+
return this.entryRepo.findSourceRefsByHash(entityId, sourceHash);
|
|
2074
2278
|
}
|
|
2075
2279
|
async runPrune(entityId, options) {
|
|
2076
2280
|
return this.maintenanceService.runPrune(entityId, options);
|
|
@@ -2157,16 +2361,16 @@ var WikiMemory = class {
|
|
|
2157
2361
|
async importDump(dump, opts) {
|
|
2158
2362
|
return this.importExportService.importDump(dump, opts);
|
|
2159
2363
|
}
|
|
2160
|
-
async forget(entityId, params) {
|
|
2161
|
-
return this.maintenanceService.forget(entityId, params);
|
|
2364
|
+
async forget(entityId, params, opts) {
|
|
2365
|
+
return this.maintenanceService.forget(entityId, params, opts);
|
|
2162
2366
|
}
|
|
2163
2367
|
/**
|
|
2164
2368
|
* @param params.promptOverride - Overrides the system prompt for this ingest call only.
|
|
2165
2369
|
* For persistent customization, set `options.config.prompts.ingestSystemPrompt` at
|
|
2166
2370
|
* WikiMemory construction time.
|
|
2167
2371
|
*/
|
|
2168
|
-
async ingestDocument(entityId, params) {
|
|
2169
|
-
return this.ingestionService.ingestDocument(entityId, params);
|
|
2372
|
+
async ingestDocument(entityId, params, opts) {
|
|
2373
|
+
return this.ingestionService.ingestDocument(entityId, params, opts);
|
|
2170
2374
|
}
|
|
2171
2375
|
/**
|
|
2172
2376
|
* Returns up to `limit` unprocessed outbox events, oldest first.
|