@equationalapplications/core-llm-wiki 7.7.5 → 7.7.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1171,6 +1171,14 @@ type ReembedCandidateRow = WikiFact & {
1171
1171
  };
1172
1172
  declare class EntryRepository extends BaseRepository {
1173
1173
  private outbox;
1174
+ /**
1175
+ * Column list and liveness predicate shared by findMiniSearchRows and
1176
+ * findMiniSearchRowsByIds, so the full rebuild and the incremental read
1177
+ * can never drift apart (spec 2026-09-28 §4.4: both must produce
1178
+ * identical documents).
1179
+ */
1180
+ private static readonly MINI_SEARCH_COLUMNS;
1181
+ private static readonly MINI_SEARCH_LIVE_WHERE;
1174
1182
  private chunkSize;
1175
1183
  constructor(db: SQLiteAdapter, prefix: string, outbox: OutboxRepository);
1176
1184
  /**
@@ -1375,6 +1383,19 @@ declare class EntryRepository extends BaseRepository {
1375
1383
  body: string;
1376
1384
  tags: string;
1377
1385
  }>>;
1386
+ /**
1387
+ * Keyword-index rows for specific ids — the incremental counterpart of
1388
+ * {@link findMiniSearchRows}, with identical columns. Soft-deleted rows and
1389
+ * rows of other entities are omitted, so a caller can pass every id a write
1390
+ * touched and get back exactly what should be indexed. Spec 2026-09-28 §4.4.
1391
+ */
1392
+ findMiniSearchRowsByIds(entityId: string, ids: readonly string[], tx?: SQLiteAdapter): Promise<Array<{
1393
+ id: string;
1394
+ entity_id: string;
1395
+ title: string;
1396
+ body: string;
1397
+ tags: string;
1398
+ }>>;
1378
1399
  updateEmbeddingBlob(id: string, blob: Uint8Array, tx?: SQLiteAdapter): Promise<void>;
1379
1400
  /**
1380
1401
  * Record a failed embedding attempt. Deliberately does NOT touch updated_at
@@ -1702,7 +1723,29 @@ declare class SearchService {
1702
1723
  * auto-vacuum debt behind the TypeError in #64.
1703
1724
  */
1704
1725
  private syncChain;
1726
+ /**
1727
+ * Entities whose index may have drifted from SQLite: an incremental update
1728
+ * failed part-way, or core wrote rows it could not index (upsertGraph runs in
1729
+ * the host's transaction). Their next syncEntries rebuilds the entity in full.
1730
+ * See spec 2026-09-28 §5.
1731
+ */
1732
+ private staleEntities;
1733
+ /**
1734
+ * Per-entity count of markStale() calls. A rebuild clears an entity's stale
1735
+ * flag only when the count it snapshotted before its read still holds
1736
+ * afterwards: markStale() runs inside the host's still-open transaction, so
1737
+ * a call landing mid-read belongs to rows the read cannot have seen, and
1738
+ * its flag must outlive the turn (#233 review).
1739
+ */
1740
+ private staleEpochs;
1705
1741
  constructor(entryRepo: EntryRepository);
1742
+ /**
1743
+ * A fresh index with the production options. clearAll() swaps one in because
1744
+ * MiniSearch.removeAll() empties the index but leaves dirtCount (and its
1745
+ * vacuum bookkeeping) at its old value, which would trip syncEntries'
1746
+ * conditional vacuum early after a clear.
1747
+ */
1748
+ private createMiniSearch;
1706
1749
  /**
1707
1750
  * Rebuilds the search index and clears the vector cache for a given entity.
1708
1751
  * A direct replacement for manually syncing state after a DB transaction.
@@ -1712,6 +1755,24 @@ declare class SearchService {
1712
1755
  * correct failure mode and killing the host process is not.
1713
1756
  */
1714
1757
  sync(entityId?: string): Promise<void>;
1758
+ /**
1759
+ * Re-indexes only `ids` for `entityId`: drops each from the index, then
1760
+ * re-adds the ones still live in SQLite, so soft-deleted or missing ids end
1761
+ * up absent. Costs O(ids), where sync(entityId) costs O(entity) — callers that
1762
+ * know what a write touched use this so chunked imports stay linear (#232).
1763
+ *
1764
+ * Serialized with sync() on the same chain and, like it, never rejects. An
1765
+ * entity that is stale or has never been indexed gets a full rebuild instead
1766
+ * — including on an empty `ids` list, which is a no-op only for an entity
1767
+ * the index already tracks, and even then one that waits for rebuilds
1768
+ * already queued on the chain.
1769
+ */
1770
+ syncEntries(entityId: string, ids: Iterable<string>): Promise<void>;
1771
+ /**
1772
+ * Forces the entity's next syncEntries to rebuild it in full. For writes core
1773
+ * cannot index itself, such as upsertGraph inside a host transaction.
1774
+ */
1775
+ markStale(entityId: string): void;
1715
1776
  /**
1716
1777
  * Clears the parsed vector cache. Useful for mid-loop flush guarantees
1717
1778
  * or memory pressure evictions.
@@ -1738,6 +1799,14 @@ declare class SearchService {
1738
1799
  private normalizeMiniSearchRow;
1739
1800
  private _tieBreakSort;
1740
1801
  private _compareScoredRows;
1802
+ /**
1803
+ * MiniSearch breaks equal-score ties by internal insertion order, which
1804
+ * syncEntries changes: a discarded-and-re-added id moves to the end. Every
1805
+ * caller truncates these results to a limit, so which ids survive a tie at
1806
+ * the boundary must not depend on write history. Re-sort exact score ties
1807
+ * by id — the same final tie-break _compareScoredRows applies.
1808
+ */
1809
+ private _compareSearchResults;
1741
1810
  }
1742
1811
 
1743
1812
  type OperationType = 'prune' | 'librarian' | 'heal' | 'ingest' | 'reembed' | 'global_reembed' | 'import' | 'global_import' | 'forget' | 'ontologyBackfill';
@@ -1171,6 +1171,14 @@ type ReembedCandidateRow = WikiFact & {
1171
1171
  };
1172
1172
  declare class EntryRepository extends BaseRepository {
1173
1173
  private outbox;
1174
+ /**
1175
+ * Column list and liveness predicate shared by findMiniSearchRows and
1176
+ * findMiniSearchRowsByIds, so the full rebuild and the incremental read
1177
+ * can never drift apart (spec 2026-09-28 §4.4: both must produce
1178
+ * identical documents).
1179
+ */
1180
+ private static readonly MINI_SEARCH_COLUMNS;
1181
+ private static readonly MINI_SEARCH_LIVE_WHERE;
1174
1182
  private chunkSize;
1175
1183
  constructor(db: SQLiteAdapter, prefix: string, outbox: OutboxRepository);
1176
1184
  /**
@@ -1375,6 +1383,19 @@ declare class EntryRepository extends BaseRepository {
1375
1383
  body: string;
1376
1384
  tags: string;
1377
1385
  }>>;
1386
+ /**
1387
+ * Keyword-index rows for specific ids — the incremental counterpart of
1388
+ * {@link findMiniSearchRows}, with identical columns. Soft-deleted rows and
1389
+ * rows of other entities are omitted, so a caller can pass every id a write
1390
+ * touched and get back exactly what should be indexed. Spec 2026-09-28 §4.4.
1391
+ */
1392
+ findMiniSearchRowsByIds(entityId: string, ids: readonly string[], tx?: SQLiteAdapter): Promise<Array<{
1393
+ id: string;
1394
+ entity_id: string;
1395
+ title: string;
1396
+ body: string;
1397
+ tags: string;
1398
+ }>>;
1378
1399
  updateEmbeddingBlob(id: string, blob: Uint8Array, tx?: SQLiteAdapter): Promise<void>;
1379
1400
  /**
1380
1401
  * Record a failed embedding attempt. Deliberately does NOT touch updated_at
@@ -1702,7 +1723,29 @@ declare class SearchService {
1702
1723
  * auto-vacuum debt behind the TypeError in #64.
1703
1724
  */
1704
1725
  private syncChain;
1726
+ /**
1727
+ * Entities whose index may have drifted from SQLite: an incremental update
1728
+ * failed part-way, or core wrote rows it could not index (upsertGraph runs in
1729
+ * the host's transaction). Their next syncEntries rebuilds the entity in full.
1730
+ * See spec 2026-09-28 §5.
1731
+ */
1732
+ private staleEntities;
1733
+ /**
1734
+ * Per-entity count of markStale() calls. A rebuild clears an entity's stale
1735
+ * flag only when the count it snapshotted before its read still holds
1736
+ * afterwards: markStale() runs inside the host's still-open transaction, so
1737
+ * a call landing mid-read belongs to rows the read cannot have seen, and
1738
+ * its flag must outlive the turn (#233 review).
1739
+ */
1740
+ private staleEpochs;
1705
1741
  constructor(entryRepo: EntryRepository);
1742
+ /**
1743
+ * A fresh index with the production options. clearAll() swaps one in because
1744
+ * MiniSearch.removeAll() empties the index but leaves dirtCount (and its
1745
+ * vacuum bookkeeping) at its old value, which would trip syncEntries'
1746
+ * conditional vacuum early after a clear.
1747
+ */
1748
+ private createMiniSearch;
1706
1749
  /**
1707
1750
  * Rebuilds the search index and clears the vector cache for a given entity.
1708
1751
  * A direct replacement for manually syncing state after a DB transaction.
@@ -1712,6 +1755,24 @@ declare class SearchService {
1712
1755
  * correct failure mode and killing the host process is not.
1713
1756
  */
1714
1757
  sync(entityId?: string): Promise<void>;
1758
+ /**
1759
+ * Re-indexes only `ids` for `entityId`: drops each from the index, then
1760
+ * re-adds the ones still live in SQLite, so soft-deleted or missing ids end
1761
+ * up absent. Costs O(ids), where sync(entityId) costs O(entity) — callers that
1762
+ * know what a write touched use this so chunked imports stay linear (#232).
1763
+ *
1764
+ * Serialized with sync() on the same chain and, like it, never rejects. An
1765
+ * entity that is stale or has never been indexed gets a full rebuild instead
1766
+ * — including on an empty `ids` list, which is a no-op only for an entity
1767
+ * the index already tracks, and even then one that waits for rebuilds
1768
+ * already queued on the chain.
1769
+ */
1770
+ syncEntries(entityId: string, ids: Iterable<string>): Promise<void>;
1771
+ /**
1772
+ * Forces the entity's next syncEntries to rebuild it in full. For writes core
1773
+ * cannot index itself, such as upsertGraph inside a host transaction.
1774
+ */
1775
+ markStale(entityId: string): void;
1715
1776
  /**
1716
1777
  * Clears the parsed vector cache. Useful for mid-loop flush guarantees
1717
1778
  * or memory pressure evictions.
@@ -1738,6 +1799,14 @@ declare class SearchService {
1738
1799
  private normalizeMiniSearchRow;
1739
1800
  private _tieBreakSort;
1740
1801
  private _compareScoredRows;
1802
+ /**
1803
+ * MiniSearch breaks equal-score ties by internal insertion order, which
1804
+ * syncEntries changes: a discarded-and-re-added id moves to the end. Every
1805
+ * caller truncates these results to a limit, so which ids survive a tie at
1806
+ * the boundary must not depend on write history. Re-sort exact score ties
1807
+ * by id — the same final tie-break _compareScoredRows applies.
1808
+ */
1809
+ private _compareSearchResults;
1741
1810
  }
1742
1811
 
1743
1812
  type OperationType = 'prune' | 'librarian' | 'heal' | 'ingest' | 'reembed' | 'global_reembed' | 'import' | 'global_import' | 'forget' | 'ontologyBackfill';
@@ -1,3 +1,3 @@
1
- export { aq as EmbeddingService, ar as ImportExportService, as as IngestionService, at as JobManager, at as JobManagerType, au as MaintenanceService, av as RetrievalService, aw as SearchService, aw as SearchServiceType, ai as WikiMemoryTestAccess, ax as WriteService } from './testing-CH_Kk34-.mjs';
1
+ export { aq as EmbeddingService, ar as ImportExportService, as as IngestionService, at as JobManager, at as JobManagerType, au as MaintenanceService, av as RetrievalService, aw as SearchService, aw as SearchServiceType, ai as WikiMemoryTestAccess, ax as WriteService } from './testing-BIFNLUva.mjs';
2
2
  import '@equationalapplications/core-okf';
3
3
  import 'minisearch';
package/dist/testing.d.ts CHANGED
@@ -1,3 +1,3 @@
1
- export { aq as EmbeddingService, ar as ImportExportService, as as IngestionService, at as JobManager, at as JobManagerType, au as MaintenanceService, av as RetrievalService, aw as SearchService, aw as SearchServiceType, ai as WikiMemoryTestAccess, ax as WriteService } from './testing-CH_Kk34-.js';
1
+ export { aq as EmbeddingService, ar as ImportExportService, as as IngestionService, at as JobManager, at as JobManagerType, au as MaintenanceService, av as RetrievalService, aw as SearchService, aw as SearchServiceType, ai as WikiMemoryTestAccess, ax as WriteService } from './testing-BIFNLUva.js';
2
2
  import '@equationalapplications/core-okf';
3
3
  import 'minisearch';
package/dist/testing.js CHANGED
@@ -1343,7 +1343,11 @@ var ImportExportService = class {
1343
1343
  );
1344
1344
  }
1345
1345
  });
1346
- await this.searchService.sync(entityId);
1346
+ if (merge) {
1347
+ await this.searchService.syncEntries(entityId, [...upsertedFactIds]);
1348
+ } else {
1349
+ await this.searchService.sync(entityId);
1350
+ }
1347
1351
  for (const fact of bundle.facts) {
1348
1352
  if (!fact.deleted_at && upsertedFactIds.has(fact.id) && !factsWithPreservedBlob.has(fact.id)) {
1349
1353
  const clipped = clippedTextByFactId.get(fact.id);
@@ -2095,7 +2099,10 @@ var IngestionService = class {
2095
2099
  return zeroChunkResult(canonical);
2096
2100
  }
2097
2101
  diagBuffer.flush(this.options);
2098
- await this.searchService.sync(entityId);
2102
+ await this.searchService.syncEntries(
2103
+ entityId,
2104
+ [...deletedSourceFactIds, ...insertedFacts.map((fact) => fact.id)]
2105
+ );
2099
2106
  if (failedChunks === 0) {
2100
2107
  const uniqueDeletedSourceFactIds = Array.from(new Set(deletedSourceFactIds));
2101
2108
  for (const factId of uniqueDeletedSourceFactIds) {
@@ -4506,7 +4513,31 @@ var _SearchService = class _SearchService {
4506
4513
  * auto-vacuum debt behind the TypeError in #64.
4507
4514
  */
4508
4515
  this.syncChain = Promise.resolve();
4509
- this.miniSearch = new MiniSearch__default.default({
4516
+ /**
4517
+ * Entities whose index may have drifted from SQLite: an incremental update
4518
+ * failed part-way, or core wrote rows it could not index (upsertGraph runs in
4519
+ * the host's transaction). Their next syncEntries rebuilds the entity in full.
4520
+ * See spec 2026-09-28 §5.
4521
+ */
4522
+ this.staleEntities = /* @__PURE__ */ new Set();
4523
+ /**
4524
+ * Per-entity count of markStale() calls. A rebuild clears an entity's stale
4525
+ * flag only when the count it snapshotted before its read still holds
4526
+ * afterwards: markStale() runs inside the host's still-open transaction, so
4527
+ * a call landing mid-read belongs to rows the read cannot have seen, and
4528
+ * its flag must outlive the turn (#233 review).
4529
+ */
4530
+ this.staleEpochs = /* @__PURE__ */ new Map();
4531
+ this.miniSearch = this.createMiniSearch();
4532
+ }
4533
+ /**
4534
+ * A fresh index with the production options. clearAll() swaps one in because
4535
+ * MiniSearch.removeAll() empties the index but leaves dirtCount (and its
4536
+ * vacuum bookkeeping) at its old value, which would trip syncEntries'
4537
+ * conditional vacuum early after a clear.
4538
+ */
4539
+ createMiniSearch() {
4540
+ return new MiniSearch__default.default({
4510
4541
  fields: ["title", "body", "tags"],
4511
4542
  storeFields: ["entity_id"],
4512
4543
  // Vacuuming is driven explicitly at the end of each serialized rebuild
@@ -4533,7 +4564,19 @@ var _SearchService = class _SearchService {
4533
4564
  const work = this.syncChain.then(async () => {
4534
4565
  try {
4535
4566
  try {
4567
+ const epochsBefore = new Map(this.staleEpochs);
4536
4568
  await this.rebuildIndex(entityId);
4569
+ if (entityId) {
4570
+ if ((this.staleEpochs.get(entityId) ?? 0) === (epochsBefore.get(entityId) ?? 0)) {
4571
+ this.staleEntities.delete(entityId);
4572
+ }
4573
+ } else {
4574
+ for (const id of [...this.staleEntities]) {
4575
+ if ((this.staleEpochs.get(id) ?? 0) === (epochsBefore.get(id) ?? 0)) {
4576
+ this.staleEntities.delete(id);
4577
+ }
4578
+ }
4579
+ }
4537
4580
  await this.miniSearch.vacuum();
4538
4581
  } finally {
4539
4582
  this.evictCache(entityId);
@@ -4545,6 +4588,68 @@ var _SearchService = class _SearchService {
4545
4588
  this.syncChain = work;
4546
4589
  return work;
4547
4590
  }
4591
+ /**
4592
+ * Re-indexes only `ids` for `entityId`: drops each from the index, then
4593
+ * re-adds the ones still live in SQLite, so soft-deleted or missing ids end
4594
+ * up absent. Costs O(ids), where sync(entityId) costs O(entity) — callers that
4595
+ * know what a write touched use this so chunked imports stay linear (#232).
4596
+ *
4597
+ * Serialized with sync() on the same chain and, like it, never rejects. An
4598
+ * entity that is stale or has never been indexed gets a full rebuild instead
4599
+ * — including on an empty `ids` list, which is a no-op only for an entity
4600
+ * the index already tracks, and even then one that waits for rebuilds
4601
+ * already queued on the chain.
4602
+ */
4603
+ async syncEntries(entityId, ids) {
4604
+ const uniqueIds = [...new Set(ids)];
4605
+ if (uniqueIds.length === 0 && !this.staleEntities.has(entityId) && this.miniSearchEntryIdsByEntity.has(entityId)) {
4606
+ return this.syncChain;
4607
+ }
4608
+ const work = this.syncChain.then(async () => {
4609
+ try {
4610
+ try {
4611
+ const epochsBefore = new Map(this.staleEpochs);
4612
+ const needsRebuild = () => this.staleEntities.has(entityId) || !this.miniSearchEntryIdsByEntity.has(entityId);
4613
+ if (!needsRebuild()) {
4614
+ const rows = await this.entryRepo.findMiniSearchRowsByIds(entityId, uniqueIds);
4615
+ const tracked = this.miniSearchEntryIdsByEntity.get(entityId);
4616
+ if (tracked && !this.staleEntities.has(entityId)) {
4617
+ for (const id of uniqueIds) {
4618
+ if (tracked.delete(id) && this.miniSearch.has(id)) this.miniSearch.discard(id);
4619
+ }
4620
+ const documents = rows.map((row) => this.normalizeMiniSearchRow(row));
4621
+ if (documents.length > 0) this.miniSearch.addAll(documents);
4622
+ for (const document of documents) tracked.add(document.id);
4623
+ if (this.miniSearch.dirtCount > 0) {
4624
+ await this.miniSearch.vacuum();
4625
+ }
4626
+ return;
4627
+ }
4628
+ }
4629
+ await this.rebuildIndex(entityId);
4630
+ if ((this.staleEpochs.get(entityId) ?? 0) === (epochsBefore.get(entityId) ?? 0)) {
4631
+ this.staleEntities.delete(entityId);
4632
+ }
4633
+ await this.miniSearch.vacuum();
4634
+ } finally {
4635
+ this.evictCache(entityId);
4636
+ }
4637
+ } catch (err) {
4638
+ this.staleEntities.add(entityId);
4639
+ console.warn(`[WikiMemory] search index incremental sync failed for ${entityId}:`, err);
4640
+ }
4641
+ });
4642
+ this.syncChain = work;
4643
+ return work;
4644
+ }
4645
+ /**
4646
+ * Forces the entity's next syncEntries to rebuild it in full. For writes core
4647
+ * cannot index itself, such as upsertGraph inside a host transaction.
4648
+ */
4649
+ markStale(entityId) {
4650
+ this.staleEntities.add(entityId);
4651
+ this.staleEpochs.set(entityId, (this.staleEpochs.get(entityId) ?? 0) + 1);
4652
+ }
4548
4653
  /**
4549
4654
  * Clears the parsed vector cache. Useful for mid-loop flush guarantees
4550
4655
  * or memory pressure evictions.
@@ -4561,8 +4666,10 @@ var _SearchService = class _SearchService {
4561
4666
  */
4562
4667
  clearAll() {
4563
4668
  this.vectorCache.clear();
4564
- this.miniSearch.removeAll();
4669
+ this.miniSearch = this.createMiniSearch();
4565
4670
  this.miniSearchEntryIdsByEntity.clear();
4671
+ this.staleEntities.clear();
4672
+ this.staleEpochs.clear();
4566
4673
  }
4567
4674
  /**
4568
4675
  * Executes a keyword search against the active MiniSearch index.
@@ -4573,7 +4680,7 @@ var _SearchService = class _SearchService {
4573
4680
  filter: (r) => entityIdSet.has(r.entity_id),
4574
4681
  combineWith: "OR"
4575
4682
  });
4576
- return results.slice(0, limit);
4683
+ return results.sort((a, b) => this._compareSearchResults(a, b)).slice(0, limit);
4577
4684
  }
4578
4685
  /**
4579
4686
  * Pre-fetches MiniSearch scores for candidate hydration, used during hybrid weighting.
@@ -4583,7 +4690,7 @@ var _SearchService = class _SearchService {
4583
4690
  let results = this.miniSearch.search(query, {
4584
4691
  filter: (r) => entityIdSet.has(r.entity_id),
4585
4692
  combineWith: "OR"
4586
- });
4693
+ }).sort((a, b) => this._compareSearchResults(a, b));
4587
4694
  if (preFilterLimit !== void 0) {
4588
4695
  results = results.slice(0, preFilterLimit);
4589
4696
  }
@@ -4711,6 +4818,18 @@ var _SearchService = class _SearchService {
4711
4818
  if (updatedAtDiff !== 0) return updatedAtDiff;
4712
4819
  return a.id.localeCompare(b.id);
4713
4820
  }
4821
+ /**
4822
+ * MiniSearch breaks equal-score ties by internal insertion order, which
4823
+ * syncEntries changes: a discarded-and-re-added id moves to the end. Every
4824
+ * caller truncates these results to a limit, so which ids survive a tie at
4825
+ * the boundary must not depend on write history. Re-sort exact score ties
4826
+ * by id — the same final tie-break _compareScoredRows applies.
4827
+ */
4828
+ _compareSearchResults(a, b) {
4829
+ const scoreDiff = b.score - a.score;
4830
+ if (!Number.isNaN(scoreDiff) && scoreDiff !== 0) return scoreDiff;
4831
+ return a.id.localeCompare(b.id);
4832
+ }
4714
4833
  };
4715
4834
  /**
4716
4835
  * Maximum number of entities whose parsed embedding vectors are held in