@equationalapplications/core-llm-wiki 5.1.1 → 5.2.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +14 -0
- package/dist/{chunk-YKXCMOHH.mjs → chunk-6HNIOKT4.mjs} +258 -70
- package/dist/chunk-6HNIOKT4.mjs.map +1 -0
- package/dist/index.d.mts +2 -2
- package/dist/index.d.ts +2 -2
- package/dist/index.js +608 -85
- package/dist/index.js.map +1 -1
- package/dist/index.mjs +354 -20
- package/dist/index.mjs.map +1 -1
- package/dist/{testing-DCRNie7k.d.mts → testing-CAk9oFvw.d.mts} +242 -3
- package/dist/{testing-DCRNie7k.d.ts → testing-CAk9oFvw.d.ts} +242 -3
- package/dist/testing.d.mts +1 -1
- package/dist/testing.d.ts +1 -1
- package/dist/testing.js +309 -67
- package/dist/testing.js.map +1 -1
- package/dist/testing.mjs +1 -1
- package/package.json +2 -2
- package/dist/chunk-YKXCMOHH.mjs.map +0 -1
package/dist/index.mjs
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
import { __privateAdd, MetadataRepository, EmbeddingService, SearchService, JobManager, PromptService, IngestionService, MaintenanceService, ImportExportService, RetrievalService, WriteService, __privateGet, __privateSet, normalizeSourceRef, normalizeSourceHash, entitySummaryMetaKey, generateId, BaseRepository, emptyManifest, resolveNodeType, validateInlineEdges, resolveEdgeDefinitions, normalizeTitleKey, WikiTransactionError, extractSqliteCode } from './chunk-
|
|
2
|
-
export { DEFAULT_CHUNK_OVERLAP, DEFAULT_MAX_CHUNK_LENGTH, HEAL_BATCH_SIZE, HEAL_RECHECK_MS, HOOK_TIMEOUT_MARKER, ONTOLOGY_BACKFILL_BATCH_SIZE, ONTOLOGY_BACKFILL_MAX_PROMPT_CHARS, ONTOLOGY_BACKFILL_RECHECK_MS, ONTOLOGY_BACKFILL_SYSTEM_PROMPT, PromptService, PrunePartialFailureError, WikiBusyError, WikiTransactionError, chunkText, configureRandomSource, parseEmbedding, safeSlice, validateManifest } from './chunk-
|
|
1
|
+
import { __privateAdd, MetadataRepository, EmbeddingService, SearchService, JobManager, PromptService, IngestionService, MaintenanceService, ImportExportService, RetrievalService, WriteService, __privateGet, __privateSet, normalizeSourceRef, normalizeSourceHash, entitySummaryMetaKey, generateId, BaseRepository, emptyManifest, resolveNodeType, validateInlineEdges, resolveEdgeDefinitions, normalizeTitleKey, WikiTransactionError, extractSqliteCode } from './chunk-6HNIOKT4.mjs';
|
|
2
|
+
export { DEFAULT_CHUNK_OVERLAP, DEFAULT_MAX_CHUNK_LENGTH, HEAL_BATCH_SIZE, HEAL_RECHECK_MS, HOOK_TIMEOUT_MARKER, ONTOLOGY_BACKFILL_BATCH_SIZE, ONTOLOGY_BACKFILL_MAX_PROMPT_CHARS, ONTOLOGY_BACKFILL_RECHECK_MS, ONTOLOGY_BACKFILL_SYSTEM_PROMPT, PromptService, PrunePartialFailureError, WikiBusyError, WikiDuplicateHashError, WikiTransactionError, chunkText, configureRandomSource, parseEmbedding, safeSlice, validateManifest } from './chunk-6HNIOKT4.mjs';
|
|
3
3
|
import { appendRelatedSection, buildConceptDocument, buildLogMd, buildEntityIndexMd, buildRootIndexMd, isAllowedOkfPath, parseRootIndexMd, parseEntityIndexMd, parseConcept, splitRelatedSection, extractMarkdownLinks, parseLogMd, parseEventIdComment, appendEventIdComment } from '@equationalapplications/core-okf';
|
|
4
4
|
|
|
5
5
|
// src/db/schema.ts
|
|
@@ -32,6 +32,24 @@ async function setupDatabase(db, prefix) {
|
|
|
32
32
|
CREATE INDEX IF NOT EXISTS ${prefix}entries_source_hash_idx ON ${prefix}entries(entity_id, source_hash) WHERE source_hash IS NOT NULL;
|
|
33
33
|
CREATE INDEX IF NOT EXISTS ${prefix}entries_updated_idx ON ${prefix}entries(updated_at DESC);
|
|
34
34
|
|
|
35
|
+
-- source_ref_index: per-(entity, source_hash) record of the canonical sourceRef
|
|
36
|
+
-- currently holding that hash. The partial UNIQUE index on (entity_id, source_hash)
|
|
37
|
+
-- WHERE deleted_at IS NULL enforces the sourceRef-level TOCTOU-race invariant;
|
|
38
|
+
-- entries-level uniqueness cannot express it because a single ingestDocument call
|
|
39
|
+
-- writes N facts that all share (entity_id, source_ref, source_hash). See
|
|
40
|
+
-- docs/superpowers/specs/2026-08-07-dependabot-concurrency-release-hygiene-design.md \xA7B1.
|
|
41
|
+
CREATE TABLE IF NOT EXISTS ${prefix}source_ref_index (
|
|
42
|
+
id TEXT PRIMARY KEY,
|
|
43
|
+
entity_id TEXT NOT NULL,
|
|
44
|
+
source_hash TEXT NOT NULL,
|
|
45
|
+
source_ref TEXT NOT NULL,
|
|
46
|
+
created_at INTEGER NOT NULL,
|
|
47
|
+
deleted_at INTEGER
|
|
48
|
+
);
|
|
49
|
+
CREATE UNIQUE INDEX IF NOT EXISTS ${prefix}idx_source_ref_hash
|
|
50
|
+
ON ${prefix}source_ref_index (entity_id, source_hash)
|
|
51
|
+
WHERE deleted_at IS NULL;
|
|
52
|
+
|
|
35
53
|
CREATE TABLE IF NOT EXISTS ${prefix}tasks (
|
|
36
54
|
id TEXT PRIMARY KEY,
|
|
37
55
|
entity_id TEXT NOT NULL,
|
|
@@ -278,6 +296,61 @@ var MIGRATIONS = [
|
|
|
278
296
|
);
|
|
279
297
|
}
|
|
280
298
|
}
|
|
299
|
+
},
|
|
300
|
+
{
|
|
301
|
+
version: 9,
|
|
302
|
+
description: "add_source_ref_index",
|
|
303
|
+
run: async (db, prefix) => {
|
|
304
|
+
const duplicates = await db.getAllAsync(
|
|
305
|
+
`SELECT entity_id, source_hash, COUNT(DISTINCT source_ref) AS n_refs
|
|
306
|
+
FROM ${prefix}entries
|
|
307
|
+
WHERE deleted_at IS NULL AND source_hash IS NOT NULL
|
|
308
|
+
GROUP BY entity_id, source_hash
|
|
309
|
+
HAVING COUNT(DISTINCT source_ref) > 1`
|
|
310
|
+
);
|
|
311
|
+
if (duplicates.length > 0) {
|
|
312
|
+
const sample = duplicates.slice(0, 5).map((d) => `(entity_id=${d.entity_id}, source_hash=${d.source_hash.slice(0, 12)}\u2026, n_refs=${d.n_refs})`).join(", ");
|
|
313
|
+
throw new Error(
|
|
314
|
+
`Migration v9 (add_source_ref_index) failed: existing live rows have multiple sourceRefs sharing a hash. Found ${duplicates.length} duplicate (entity_id, source_hash) groups. First ${Math.min(5, duplicates.length)}: ${sample}. Resolve each by calling forget({ sourceRef: <loser> }) for the offending sourceRef, then re-run setup.`
|
|
315
|
+
);
|
|
316
|
+
}
|
|
317
|
+
await db.execAsync(
|
|
318
|
+
`CREATE TABLE IF NOT EXISTS ${prefix}source_ref_index (
|
|
319
|
+
id TEXT PRIMARY KEY,
|
|
320
|
+
entity_id TEXT NOT NULL,
|
|
321
|
+
source_hash TEXT NOT NULL,
|
|
322
|
+
source_ref TEXT NOT NULL,
|
|
323
|
+
created_at INTEGER NOT NULL,
|
|
324
|
+
deleted_at INTEGER
|
|
325
|
+
);
|
|
326
|
+
CREATE UNIQUE INDEX IF NOT EXISTS ${prefix}idx_source_ref_hash
|
|
327
|
+
ON ${prefix}source_ref_index (entity_id, source_hash)
|
|
328
|
+
WHERE deleted_at IS NULL;`
|
|
329
|
+
);
|
|
330
|
+
await db.execAsync(
|
|
331
|
+
`INSERT OR IGNORE INTO ${prefix}source_ref_index (id, entity_id, source_hash, source_ref, created_at, deleted_at)
|
|
332
|
+
SELECT
|
|
333
|
+
'sri:' || entity_id || ':' || source_hash,
|
|
334
|
+
entity_id,
|
|
335
|
+
source_hash,
|
|
336
|
+
source_ref,
|
|
337
|
+
updated_at,
|
|
338
|
+
NULL
|
|
339
|
+
FROM (
|
|
340
|
+
SELECT
|
|
341
|
+
entity_id, source_hash, source_ref, updated_at,
|
|
342
|
+
ROW_NUMBER() OVER (
|
|
343
|
+
PARTITION BY entity_id, source_hash
|
|
344
|
+
ORDER BY updated_at ASC, id ASC
|
|
345
|
+
) AS rn
|
|
346
|
+
FROM ${prefix}entries
|
|
347
|
+
WHERE deleted_at IS NULL
|
|
348
|
+
AND source_hash IS NOT NULL
|
|
349
|
+
AND source_ref IS NOT NULL
|
|
350
|
+
) ranked
|
|
351
|
+
WHERE rn = 1;`
|
|
352
|
+
);
|
|
353
|
+
}
|
|
281
354
|
}
|
|
282
355
|
];
|
|
283
356
|
for (let i = 1; i < MIGRATIONS.length; i++) {
|
|
@@ -975,12 +1048,154 @@ var EntryRepository = class extends BaseRepository {
|
|
|
975
1048
|
const row = await executor.getFirstAsync(
|
|
976
1049
|
`SELECT source_hash FROM ${this.prefix}entries
|
|
977
1050
|
WHERE entity_id = ? AND source_ref = ? AND deleted_at IS NULL
|
|
978
|
-
ORDER BY updated_at DESC
|
|
1051
|
+
ORDER BY updated_at DESC, id ASC
|
|
979
1052
|
LIMIT 1`,
|
|
980
1053
|
[entityId, sourceRef]
|
|
981
1054
|
);
|
|
982
1055
|
return row?.source_hash ?? null;
|
|
983
1056
|
}
|
|
1057
|
+
/**
|
|
1058
|
+
* Batch version of {@link findLatestSourceHash}. Returns a Map covering every
|
|
1059
|
+
* requested ref, where the value is the source_hash from the row with the
|
|
1060
|
+
* most-recently-updated live fact for that ref, or null when no live row exists.
|
|
1061
|
+
*
|
|
1062
|
+
* The SQL uses ROW_NUMBER() OVER (PARTITION BY source_ref ORDER BY updated_at
|
|
1063
|
+
* DESC, id ASC) — NOT MAX(source_hash). Aggregation with MAX(source_hash) is
|
|
1064
|
+
* wrong because MAX computes independently across grouped rows; the hash must
|
|
1065
|
+
* come from the exact row that wins MAX(updated_at). The `id ASC` tie-break
|
|
1066
|
+
* keeps selection deterministic when two live rows share `updated_at`.
|
|
1067
|
+
*
|
|
1068
|
+
* Source refs are de-duplicated and processed in chunks so the per-query bind
|
|
1069
|
+
* parameter count stays under SQLite's `SQLITE_MAX_VARIABLE_NUMBER` (default
|
|
1070
|
+
* 999, leaving one slot for `entity_id`).
|
|
1071
|
+
*
|
|
1072
|
+
* Empty input is a synchronous early return with zero SQL calls.
|
|
1073
|
+
*/
|
|
1074
|
+
async findLatestSourceHashes(entityId, sourceRefs, tx) {
|
|
1075
|
+
const out = /* @__PURE__ */ new Map();
|
|
1076
|
+
if (sourceRefs.length === 0) return out;
|
|
1077
|
+
const dedupedRefs = Array.from(new Set(sourceRefs));
|
|
1078
|
+
for (const ref of dedupedRefs) {
|
|
1079
|
+
out.set(ref, null);
|
|
1080
|
+
}
|
|
1081
|
+
const executor = this.getExecutor(tx);
|
|
1082
|
+
const chunkLimit = Math.max(1, this.chunkSize - 1);
|
|
1083
|
+
for (let i = 0; i < dedupedRefs.length; i += chunkLimit) {
|
|
1084
|
+
const chunk = dedupedRefs.slice(i, i + chunkLimit);
|
|
1085
|
+
const placeholders = chunk.map(() => "?").join(",");
|
|
1086
|
+
const rows = await executor.getAllAsync(
|
|
1087
|
+
`WITH ranked AS (
|
|
1088
|
+
SELECT source_ref, source_hash,
|
|
1089
|
+
ROW_NUMBER() OVER (
|
|
1090
|
+
PARTITION BY source_ref
|
|
1091
|
+
ORDER BY updated_at DESC, id ASC
|
|
1092
|
+
) as rn
|
|
1093
|
+
FROM ${this.prefix}entries
|
|
1094
|
+
WHERE entity_id = ? AND source_ref IN (${placeholders}) AND deleted_at IS NULL
|
|
1095
|
+
)
|
|
1096
|
+
SELECT source_ref, source_hash
|
|
1097
|
+
FROM ranked
|
|
1098
|
+
WHERE rn = 1`,
|
|
1099
|
+
[entityId, ...chunk]
|
|
1100
|
+
);
|
|
1101
|
+
for (const r of rows) {
|
|
1102
|
+
out.set(r.source_ref, r.source_hash);
|
|
1103
|
+
}
|
|
1104
|
+
}
|
|
1105
|
+
return out;
|
|
1106
|
+
}
|
|
1107
|
+
/**
|
|
1108
|
+
* Return the live source_refs for an entity that hold the given source_hash.
|
|
1109
|
+
* Used by the ingestDocument duplicate-hash guard and by hosts auditing
|
|
1110
|
+
* duplicate-content collisions. Sorted `COLLATE BINARY` so the canonical
|
|
1111
|
+
* ref (the code-unit-minimum of the set) is stable across deploys.
|
|
1112
|
+
*/
|
|
1113
|
+
async findSourceRefsByHash(entityId, sourceHash, tx) {
|
|
1114
|
+
const executor = this.getExecutor(tx);
|
|
1115
|
+
const rows = await executor.getAllAsync(
|
|
1116
|
+
`SELECT source_ref FROM ${this.prefix}entries
|
|
1117
|
+
WHERE entity_id = ? AND source_hash = ? AND deleted_at IS NULL
|
|
1118
|
+
AND source_ref IS NOT NULL
|
|
1119
|
+
GROUP BY source_ref
|
|
1120
|
+
ORDER BY source_ref COLLATE BINARY`,
|
|
1121
|
+
[entityId, sourceHash]
|
|
1122
|
+
);
|
|
1123
|
+
return rows.map((r) => r.source_ref);
|
|
1124
|
+
}
|
|
1125
|
+
/**
|
|
1126
|
+
* Per-sourceRef rollup for an entity: one row per live source_ref, with the
|
|
1127
|
+
* most-recently-updated live hash (NOT the lexically-max hash) and a live
|
|
1128
|
+
* fact count.
|
|
1129
|
+
*
|
|
1130
|
+
* The hash comes from the same row that wins ROW_NUMBER() over updated_at
|
|
1131
|
+
* DESC, so a ref with multiple live hashes (e.g. from import) reports the
|
|
1132
|
+
* latest one — matching single-doc `findLatestSourceHash` semantics.
|
|
1133
|
+
*
|
|
1134
|
+
* Sort is COLLATE BINARY: locale-dependent ordering re-mints identity on
|
|
1135
|
+
* every deploy, which is the bug this whole section exists to prevent.
|
|
1136
|
+
*/
|
|
1137
|
+
async listSourceRefs(entityId, tx) {
|
|
1138
|
+
const executor = this.getExecutor(tx);
|
|
1139
|
+
const rows = await executor.getAllAsync(
|
|
1140
|
+
`WITH ranked AS (
|
|
1141
|
+
SELECT source_ref, source_hash, updated_at,
|
|
1142
|
+
ROW_NUMBER() OVER (
|
|
1143
|
+
PARTITION BY source_ref
|
|
1144
|
+
ORDER BY updated_at DESC, id ASC
|
|
1145
|
+
) AS rn,
|
|
1146
|
+
COUNT(*) OVER (PARTITION BY source_ref) AS fact_count
|
|
1147
|
+
FROM ${this.prefix}entries
|
|
1148
|
+
WHERE entity_id = ? AND deleted_at IS NULL AND source_ref IS NOT NULL
|
|
1149
|
+
)
|
|
1150
|
+
SELECT source_ref,
|
|
1151
|
+
source_hash AS source_hash,
|
|
1152
|
+
fact_count AS fact_count,
|
|
1153
|
+
updated_at AS last_ingested_at
|
|
1154
|
+
FROM ranked
|
|
1155
|
+
WHERE rn = 1
|
|
1156
|
+
ORDER BY source_ref COLLATE BINARY`,
|
|
1157
|
+
[entityId]
|
|
1158
|
+
);
|
|
1159
|
+
return rows.map((r) => ({
|
|
1160
|
+
sourceRef: r.source_ref,
|
|
1161
|
+
sourceHash: r.source_hash,
|
|
1162
|
+
factCount: Number(r.fact_count),
|
|
1163
|
+
lastIngestedAt: Number(r.last_ingested_at)
|
|
1164
|
+
}));
|
|
1165
|
+
}
|
|
1166
|
+
/**
|
|
1167
|
+
* Count live entries for an entity across all source_refs/source_hashes.
|
|
1168
|
+
* Used by forget({ dryRun, clearAll: true }).
|
|
1169
|
+
*/
|
|
1170
|
+
async countLiveByEntityId(entityId, tx) {
|
|
1171
|
+
const executor = this.getExecutor(tx);
|
|
1172
|
+
const row = await executor.getFirstAsync(
|
|
1173
|
+
`SELECT COUNT(*) AS cnt FROM ${this.prefix}entries
|
|
1174
|
+
WHERE entity_id = ? AND deleted_at IS NULL`,
|
|
1175
|
+
[entityId]
|
|
1176
|
+
);
|
|
1177
|
+
return row?.cnt ?? 0;
|
|
1178
|
+
}
|
|
1179
|
+
/**
|
|
1180
|
+
* Count live entries matching a source filter (either or both may be null).
|
|
1181
|
+
* Used by forget({ dryRun: true }) with sourceRef/sourceHash.
|
|
1182
|
+
*/
|
|
1183
|
+
async countLiveBySource(entityId, sourceRef, sourceHash, tx) {
|
|
1184
|
+
const executor = this.getExecutor(tx);
|
|
1185
|
+
let q = `SELECT COUNT(*) AS cnt FROM ${this.prefix}entries
|
|
1186
|
+
WHERE entity_id = ? AND deleted_at IS NULL`;
|
|
1187
|
+
const args = [entityId];
|
|
1188
|
+
if (sourceRef !== null) {
|
|
1189
|
+
q += ` AND source_ref = ?`;
|
|
1190
|
+
args.push(sourceRef);
|
|
1191
|
+
}
|
|
1192
|
+
if (sourceHash !== null) {
|
|
1193
|
+
q += ` AND source_hash = ?`;
|
|
1194
|
+
args.push(sourceHash);
|
|
1195
|
+
}
|
|
1196
|
+
const row = await executor.getFirstAsync(q, args);
|
|
1197
|
+
return row?.cnt ?? 0;
|
|
1198
|
+
}
|
|
984
1199
|
async findMetadataByIds(ids, tx) {
|
|
985
1200
|
if (ids.length === 0) return [];
|
|
986
1201
|
const executor = this.getExecutor(tx);
|
|
@@ -1252,6 +1467,57 @@ var OutboxRepository = class extends BaseRepository {
|
|
|
1252
1467
|
}
|
|
1253
1468
|
};
|
|
1254
1469
|
|
|
1470
|
+
// src/repositories/SourceRefIndexRepository.ts
|
|
1471
|
+
var SourceRefIndexRepository = class extends BaseRepository {
|
|
1472
|
+
/**
|
|
1473
|
+
* Idempotent insert: the partial UNIQUE index catches concurrent inserts for
|
|
1474
|
+
* the same (entity_id, source_hash). The caller (IngestionService) catches
|
|
1475
|
+
* the resulting SQLITE_CONSTRAINT_UNIQUE and translates it to the per-mode
|
|
1476
|
+
* duplicate-hash outcome. Runtime IDs use the `sri_` prefix to avoid
|
|
1477
|
+
* collision with the deterministic `sri:<entity>:<hash>` IDs used by the v9
|
|
1478
|
+
* backfill (the colon separator keeps the two ID spaces disjoint).
|
|
1479
|
+
*/
|
|
1480
|
+
async upsert(entityId, sourceHash, sourceRef, tx) {
|
|
1481
|
+
const executor = this.getExecutor(tx);
|
|
1482
|
+
await executor.runAsync(
|
|
1483
|
+
`INSERT INTO ${this.prefix}source_ref_index (id, entity_id, source_hash, source_ref, created_at, deleted_at)
|
|
1484
|
+
VALUES (?, ?, ?, ?, ?, NULL)`,
|
|
1485
|
+
[generateId("sri_"), entityId, sourceHash, sourceRef, Date.now()]
|
|
1486
|
+
);
|
|
1487
|
+
}
|
|
1488
|
+
/**
|
|
1489
|
+
* Idempotent soft-delete of the live row for (entity_id, source_ref).
|
|
1490
|
+
* Called at the start of every ingestDocument to remove the prior run's
|
|
1491
|
+
* index row, so the new upsert doesn't collide with itself. No-op when the
|
|
1492
|
+
* row is already soft-deleted or never existed.
|
|
1493
|
+
*/
|
|
1494
|
+
async softDeleteByEntityAndSourceRef(entityId, sourceRef, tx) {
|
|
1495
|
+
const executor = this.getExecutor(tx);
|
|
1496
|
+
const now = Date.now();
|
|
1497
|
+
const result = await executor.runAsync(
|
|
1498
|
+
`UPDATE ${this.prefix}source_ref_index
|
|
1499
|
+
SET deleted_at = ?, created_at = ?
|
|
1500
|
+
WHERE entity_id = ? AND source_ref = ? AND deleted_at IS NULL`,
|
|
1501
|
+
[now, now, entityId, sourceRef]
|
|
1502
|
+
);
|
|
1503
|
+
return result.changes;
|
|
1504
|
+
}
|
|
1505
|
+
/**
|
|
1506
|
+
* Returns the live sourceRef holding the given hash, or null when no live
|
|
1507
|
+
* row exists. Used by the IngestionService pre-check (line 82) and the
|
|
1508
|
+
* catch-and-translate canonical lookup (line 212).
|
|
1509
|
+
*/
|
|
1510
|
+
async findActiveByEntityAndHash(entityId, sourceHash, tx) {
|
|
1511
|
+
const executor = this.getExecutor(tx);
|
|
1512
|
+
const row = await executor.getFirstAsync(
|
|
1513
|
+
`SELECT source_ref FROM ${this.prefix}source_ref_index
|
|
1514
|
+
WHERE entity_id = ? AND source_hash = ? AND deleted_at IS NULL`,
|
|
1515
|
+
[entityId, sourceHash]
|
|
1516
|
+
);
|
|
1517
|
+
return row?.source_ref ?? null;
|
|
1518
|
+
}
|
|
1519
|
+
};
|
|
1520
|
+
|
|
1255
1521
|
// src/repositories/TaskRepository.ts
|
|
1256
1522
|
function mapRowToTask(row) {
|
|
1257
1523
|
return {
|
|
@@ -1508,6 +1774,18 @@ var TaskRepository = class extends BaseRepository {
|
|
|
1508
1774
|
}
|
|
1509
1775
|
return result.changes;
|
|
1510
1776
|
}
|
|
1777
|
+
/**
|
|
1778
|
+
* Count live tasks for an entity. Used by forget({ dryRun, clearAll: true }).
|
|
1779
|
+
*/
|
|
1780
|
+
async countLiveByEntityId(entityId, tx) {
|
|
1781
|
+
const executor = this.getExecutor(tx);
|
|
1782
|
+
const row = await executor.getFirstAsync(
|
|
1783
|
+
`SELECT COUNT(*) AS cnt FROM ${this.prefix}tasks
|
|
1784
|
+
WHERE entity_id = ? AND deleted_at IS NULL`,
|
|
1785
|
+
[entityId]
|
|
1786
|
+
);
|
|
1787
|
+
return row?.cnt ?? 0;
|
|
1788
|
+
}
|
|
1511
1789
|
};
|
|
1512
1790
|
|
|
1513
1791
|
// src/repositories/EventRepository.ts
|
|
@@ -1916,6 +2194,7 @@ var WikiMemory = class {
|
|
|
1916
2194
|
}
|
|
1917
2195
|
this.outboxRepo = new OutboxRepository(this.db, this.prefix, !!options.config?.enableOutbox);
|
|
1918
2196
|
this.entryRepo = new EntryRepository(this.db, this.prefix, this.outboxRepo);
|
|
2197
|
+
this.sourceRefIndexRepo = new SourceRefIndexRepository(this.db, this.prefix);
|
|
1919
2198
|
this.taskRepo = new TaskRepository(this.db, this.prefix, this.outboxRepo);
|
|
1920
2199
|
this.eventRepo = new EventRepository(this.db, this.prefix);
|
|
1921
2200
|
this.edgeRepo = new EdgeRepository(this.db, this.prefix);
|
|
@@ -1934,6 +2213,7 @@ var WikiMemory = class {
|
|
|
1934
2213
|
this.prefix,
|
|
1935
2214
|
this.options,
|
|
1936
2215
|
this.entryRepo,
|
|
2216
|
+
this.sourceRefIndexRepo,
|
|
1937
2217
|
this.searchService,
|
|
1938
2218
|
this.jobManager,
|
|
1939
2219
|
this.embeddingService,
|
|
@@ -1945,6 +2225,7 @@ var WikiMemory = class {
|
|
|
1945
2225
|
this.prefix,
|
|
1946
2226
|
this.options,
|
|
1947
2227
|
this.entryRepo,
|
|
2228
|
+
this.sourceRefIndexRepo,
|
|
1948
2229
|
this.taskRepo,
|
|
1949
2230
|
this.eventRepo,
|
|
1950
2231
|
this.metadataRepo,
|
|
@@ -2009,6 +2290,7 @@ var WikiMemory = class {
|
|
|
2009
2290
|
promptService: this.promptService,
|
|
2010
2291
|
graphTraversalService: this.graphTraversalService,
|
|
2011
2292
|
entryRepo: this.entryRepo,
|
|
2293
|
+
sourceRefIndexRepo: this.sourceRefIndexRepo,
|
|
2012
2294
|
metadataRepo: this.metadataRepo,
|
|
2013
2295
|
jobManager: this.jobManager
|
|
2014
2296
|
};
|
|
@@ -2058,19 +2340,71 @@ var WikiMemory = class {
|
|
|
2058
2340
|
});
|
|
2059
2341
|
await this.searchService.sync();
|
|
2060
2342
|
}
|
|
2061
|
-
async hasChanged(entityId,
|
|
2062
|
-
|
|
2063
|
-
|
|
2064
|
-
|
|
2065
|
-
|
|
2066
|
-
|
|
2067
|
-
|
|
2068
|
-
|
|
2069
|
-
|
|
2070
|
-
|
|
2071
|
-
|
|
2072
|
-
|
|
2073
|
-
|
|
2343
|
+
async hasChanged(entityId, sourceRefOrEntries, sourceHashArg) {
|
|
2344
|
+
if (typeof sourceRefOrEntries === "string") {
|
|
2345
|
+
const sourceRef = normalizeSourceRef(sourceRefOrEntries);
|
|
2346
|
+
if (!sourceRef) {
|
|
2347
|
+
throw new Error(`Invalid sourceRef: ${JSON.stringify(sourceRefOrEntries)}`);
|
|
2348
|
+
}
|
|
2349
|
+
const sourceHash = normalizeSourceHash(sourceHashArg);
|
|
2350
|
+
if (!sourceHash) {
|
|
2351
|
+
throw new Error("Invalid sourceHash: must be a 64-character hex string (normalized to lowercase)");
|
|
2352
|
+
}
|
|
2353
|
+
const storedHash = await this.entryRepo.findLatestSourceHash(entityId, sourceRef);
|
|
2354
|
+
if (storedHash === null) return true;
|
|
2355
|
+
const normalizedStoredHash = normalizeSourceHash(storedHash);
|
|
2356
|
+
return normalizedStoredHash !== sourceHash;
|
|
2357
|
+
}
|
|
2358
|
+
const entries = sourceRefOrEntries;
|
|
2359
|
+
if (entries.length === 0) return [];
|
|
2360
|
+
const normalized = entries.map((e) => {
|
|
2361
|
+
const r = normalizeSourceRef(e.sourceRef);
|
|
2362
|
+
if (!r) throw new Error(`Invalid sourceRef: ${JSON.stringify(e.sourceRef)}`);
|
|
2363
|
+
const h = normalizeSourceHash(e.sourceHash);
|
|
2364
|
+
if (!h) throw new Error("Invalid sourceHash: must be a 64-character hex string (normalized to lowercase)");
|
|
2365
|
+
return { rawSourceRef: e.sourceRef, sourceRef: r, sourceHash: h };
|
|
2366
|
+
});
|
|
2367
|
+
const latestHashes = await this.entryRepo.findLatestSourceHashes(entityId, normalized.map((e) => e.sourceRef));
|
|
2368
|
+
const distinctHashes = Array.from(new Set(normalized.map((e) => e.sourceHash)));
|
|
2369
|
+
const dupRefs = await Promise.all(
|
|
2370
|
+
distinctHashes.map((h) => this.sourceRefIndexRepo.findActiveByEntityAndHash(entityId, h))
|
|
2371
|
+
);
|
|
2372
|
+
const dupMap = /* @__PURE__ */ new Map();
|
|
2373
|
+
for (let i = 0; i < distinctHashes.length; i++) {
|
|
2374
|
+
dupMap.set(distinctHashes[i], dupRefs[i]);
|
|
2375
|
+
}
|
|
2376
|
+
return normalized.map((e) => {
|
|
2377
|
+
const stored = latestHashes.get(e.sourceRef);
|
|
2378
|
+
const changed = stored === void 0 || stored === null || normalizeSourceHash(stored) !== e.sourceHash;
|
|
2379
|
+
const canonical = dupMap.get(e.sourceHash) ?? null;
|
|
2380
|
+
if (canonical === null || canonical === e.sourceRef) {
|
|
2381
|
+
return { sourceRef: e.rawSourceRef, changed };
|
|
2382
|
+
}
|
|
2383
|
+
return { sourceRef: e.rawSourceRef, changed, duplicateOf: canonical };
|
|
2384
|
+
});
|
|
2385
|
+
}
|
|
2386
|
+
/**
|
|
2387
|
+
* Returns the live source_refs for an entity, one row per ref, with the most
|
|
2388
|
+
* recently-updated live `source_hash` and a live fact count. Refs are sorted
|
|
2389
|
+
* `COLLATE BINARY` (no locale dependency). Used by hosts to reconcile stored
|
|
2390
|
+
* state against a live source, or audit duplicate-hash collisions via
|
|
2391
|
+
* `findSourceRefsByHash`.
|
|
2392
|
+
*/
|
|
2393
|
+
async listSourceRefs(entityId) {
|
|
2394
|
+
return this.entryRepo.listSourceRefs(entityId);
|
|
2395
|
+
}
|
|
2396
|
+
/**
|
|
2397
|
+
* Returns the live source_refs for an entity that hold the given source_hash.
|
|
2398
|
+
* With v9, source_ref_index is the source of truth for the sourceRef-level
|
|
2399
|
+
* TOCTOU-race invariant: at most one sourceRef can hold a given
|
|
2400
|
+
* (entity_id, source_hash). The result is either a single-element array
|
|
2401
|
+
* (one canonical ref) or empty (no live ref holds the hash). Returned as
|
|
2402
|
+
* an array to preserve the existing public-API shape used by hosts
|
|
2403
|
+
* auditing duplicate-content collisions.
|
|
2404
|
+
*/
|
|
2405
|
+
async findSourceRefsByHash(entityId, sourceHash) {
|
|
2406
|
+
const canonical = await this.sourceRefIndexRepo.findActiveByEntityAndHash(entityId, sourceHash);
|
|
2407
|
+
return canonical === null ? [] : [canonical];
|
|
2074
2408
|
}
|
|
2075
2409
|
async runPrune(entityId, options) {
|
|
2076
2410
|
return this.maintenanceService.runPrune(entityId, options);
|
|
@@ -2157,16 +2491,16 @@ var WikiMemory = class {
|
|
|
2157
2491
|
async importDump(dump, opts) {
|
|
2158
2492
|
return this.importExportService.importDump(dump, opts);
|
|
2159
2493
|
}
|
|
2160
|
-
async forget(entityId, params) {
|
|
2161
|
-
return this.maintenanceService.forget(entityId, params);
|
|
2494
|
+
async forget(entityId, params, opts) {
|
|
2495
|
+
return this.maintenanceService.forget(entityId, params, opts);
|
|
2162
2496
|
}
|
|
2163
2497
|
/**
|
|
2164
2498
|
* @param params.promptOverride - Overrides the system prompt for this ingest call only.
|
|
2165
2499
|
* For persistent customization, set `options.config.prompts.ingestSystemPrompt` at
|
|
2166
2500
|
* WikiMemory construction time.
|
|
2167
2501
|
*/
|
|
2168
|
-
async ingestDocument(entityId, params) {
|
|
2169
|
-
return this.ingestionService.ingestDocument(entityId, params);
|
|
2502
|
+
async ingestDocument(entityId, params, opts) {
|
|
2503
|
+
return this.ingestionService.ingestDocument(entityId, params, opts);
|
|
2170
2504
|
}
|
|
2171
2505
|
/**
|
|
2172
2506
|
* Returns up to `limit` unprocessed outbox events, oldest first.
|