@equationalapplications/core-llm-wiki 4.23.1 → 5.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -92,8 +92,8 @@ const wikiMemory = new WikiMemory(db, {
92
92
  maxResults: 10, // default: 10
93
93
  autoLibrarianThreshold: 20, // default: 20 — events before librarian auto-runs
94
94
  autoHealThreshold: 100, // default: 100 — events before heal auto-runs
95
- maxChunkLength: 12000, // default: 12000 (char count per ingestDocument chunk)
96
- chunkOverlap: 400, // default: 400 (overlap between chunks in characters)
95
+ maxChunkLength: 12000, // default: 12000 (char count per ingestDocument chunk; exported as DEFAULT_MAX_CHUNK_LENGTH)
96
+ chunkOverlap: 400, // default: 400 (overlap between chunks in characters; exported as DEFAULT_CHUNK_OVERLAP)
97
97
  chunkConcurrency: 1, // default: 1 (parallel LLM calls per ingestDocument)
98
98
  pruneRetainSoftDeletedFor: 7, // default: 7 (days before hard-deleting soft-deleted facts)
99
99
  pruneEventsAfter: 30, // default: 30 (days before hard-deleting old events)
@@ -798,6 +798,33 @@ configureRandomSource(getRandomValues);
798
798
 
799
799
  `@equationalapplications/expo-llm-wiki` does this automatically on import (main entry and `/factory` subpath). If you use `@equationalapplications/core-llm-wiki` directly on React Native without the expo package, you must call `configureRandomSource()` yourself or polyfill `globalThis.crypto.getRandomValues`.
800
800
 
801
+ ## Chunking Utilities
802
+
803
+ `ingestDocument()` splits a document into chunks before extraction. That same chunking is exported as a pure function, so a consumer can reproduce ingest-time chunk boundaries exactly — useful for recovering the passage a fact was extracted from, by re-chunking the source and ranking chunks against the fact's stored embedding.
804
+
805
+ ```typescript
806
+ import {
807
+ chunkText,
808
+ safeSlice,
809
+ DEFAULT_MAX_CHUNK_LENGTH, // 12000
810
+ DEFAULT_CHUNK_OVERLAP, // 400
811
+ } from '@equationalapplications/core-llm-wiki';
812
+
813
+ const { chunks, truncated } = chunkText(
814
+ sourceDocument,
815
+ DEFAULT_MAX_CHUNK_LENGTH,
816
+ DEFAULT_CHUNK_OVERLAP
817
+ );
818
+ ```
819
+
820
+ `chunkText(input, maxChunkLength, overlap)` returns `{ chunks, truncated }`. It prefers to split on a paragraph break, then a sentence terminator, then whitespace, falling back to a hard cut only when none is found within the window — `truncated` is `true` if any split needed that fallback. Consecutive chunks repeat up to `overlap` characters so context isn't lost at a boundary (less than `overlap` when the previous chunk was shorter than that). Empty/whitespace-only input returns `{ chunks: [], truncated: false }` without validating `maxChunkLength`/`overlap`; otherwise throws if `maxChunkLength` is not an integer >= 2, or if `overlap` is not a non-negative integer < `maxChunkLength`.
821
+
822
+ `safeSlice(value, start, end?)` slices like `String.prototype.slice` but clamps out-of-range indices, swaps a start-after-end range, and never splits a UTF-16 surrogate pair.
823
+
824
+ `DEFAULT_MAX_CHUNK_LENGTH` and `DEFAULT_CHUNK_OVERLAP` are the values `ingestDocument()` uses when neither the call nor `WikiConfig` overrides them. Pass the same values your host app configured (`config.maxChunkLength` / `config.chunkOverlap`) to match a non-default ingest.
825
+
826
+ > **Note:** `ingestDocument()` clamps the resolved overlap — whether it came from `chunkOverlap`, `WikiConfig`, or `DEFAULT_CHUNK_OVERLAP` — to `maxChunkLength - 1`. The clamp is evaluated on every call and is a no-op at the shipped defaults (`400 < 12000 - 1`), but it also bites when a custom `maxChunkLength` alone leaves the resolved overlap too large (e.g. `maxChunkLength: 100` with the default overlap `400` ingests at an effective overlap of `99`). That clamp is internal: passing the unclamped pair straight to `chunkText` throws, since it requires `overlap < maxChunkLength`. Apply the same `Math.min(overlap, maxChunkLength - 1)` yourself when re-chunking under a custom config.
827
+
801
828
  ## Adapter Interface
802
829
 
803
830
  Implement `SQLiteAdapter` to use your platform's SQLite driver:
@@ -789,144 +789,6 @@ var JobManager = class {
789
789
  }
790
790
  };
791
791
 
792
- // src/prompts.ts
793
- var LIBRARIAN_SYSTEM_PROMPT = `You are a knowledge extraction agent. Your job is to analyze recent episodic events and extract stable facts and actionable tasks about the user or entity.
794
- Return ONLY a valid JSON object matching this schema:
795
- {
796
- "facts": [{ "title": "string (max 80 chars)", "body": "string (max 800 chars)", "tags": ["string"], "confidence": "certain|inferred|tentative" }],
797
- "tasks": [{ "description": "string", "priority": "number (0-10)" }]
798
- }
799
- Keep facts concise. Do not return markdown, just raw JSON.`;
800
- var HEAL_SYSTEM_PROMPT = `You are a memory grooming agent. Your job is to review a full dump of facts and recent events to resolve contradictions, downgrade stale claims, and flag obsolete facts for deletion.
801
- Return ONLY a valid JSON object matching this schema:
802
- {
803
- "downgraded": ["string (fact IDs)"],
804
- "deleted": ["string (fact IDs)"],
805
- "newFacts": [{ "title": "string (max 80 chars)", "body": "string (max 800 chars)", "tags": ["string"], "confidence": "certain|inferred|tentative" }]
806
- }
807
- Do not return markdown, just raw JSON.`;
808
- var INGEST_SYSTEM_PROMPT = `You are a document ingestion agent. Your job is to extract factual knowledge from the provided document chunk.
809
- Return ONLY a valid JSON object matching this schema:
810
- {
811
- "facts": [{ "title": "string (max 80 chars)", "body": "string (max 800 chars)", "tags": ["string"], "confidence": "certain|inferred|tentative" }]
812
- }
813
- Extract verbatim factual content. Do not return markdown, just raw JSON.`;
814
- var ONTOLOGY_BACKFILL_SYSTEM_PROMPT = `You are a knowledge classification agent. You will receive existing memory facts that currently have no ontology type. For each input fact { "id", "title", "body", "tags" }, assign the best matching okf_type from the ontology manifest and optionally propose edges to related facts by title.
815
- Return ONLY a valid JSON object matching this schema:
816
- {
817
- "classifications": [
818
- { "id": "string (input fact id, copied verbatim)", "okf_type": "string (manifest node type slug)", "edges": [{ "edge_type": "string", "target_title": "string" }] }
819
- ]
820
- }
821
- If no manifest type fits a fact, omit that fact from "classifications" entirely \u2014 do not guess.
822
- Do not return markdown, just raw JSON.`;
823
-
824
- // src/services/PromptService.ts
825
- var PromptService = class {
826
- constructor(globalOverrides) {
827
- this.globalOverrides = globalOverrides;
828
- }
829
- hydrate(template, variables) {
830
- return template.replace(/\{\{\s*(\w+)\s*\}\}/g, (_match, key) => {
831
- const value = variables[key];
832
- if (value === void 0) return _match;
833
- return typeof value === "string" ? value : JSON.stringify(value, null, 2);
834
- });
835
- }
836
- hasOntologyPlaceholders(template) {
837
- return /\{\{\s*ontology(?:Manifest|ModeInstructions)\s*\}\}/.test(template);
838
- }
839
- buildSystemPrompt(template, variables, ontologyContext) {
840
- const shouldHydrate = Object.keys(variables).some(
841
- (key) => new RegExp(`\\{\\{\\s*${key}\\s*\\}\\}`).test(template)
842
- ) || ontologyContext != null && this.hasOntologyPlaceholders(template);
843
- const hydrated = shouldHydrate ? this.hydrate(template, { ...variables, ...ontologyContext ?? {} }) : template;
844
- return this.hasOntologyPlaceholders(template) ? ontologyContext != null ? hydrated : hydrated.replace(/\{\{\s*ontology(?:Manifest|ModeInstructions)\s*\}\}/g, "") : this.appendOntology(hydrated, ontologyContext);
845
- }
846
- appendOntology(systemPrompt, ctx) {
847
- if (!ctx) return systemPrompt;
848
- return `${systemPrompt}
849
-
850
- ${ctx.ontologyModeInstructions}`;
851
- }
852
- buildIngestPrompt(documentChunk, runtimeOverride, ontologyContext) {
853
- const template = runtimeOverride ?? this.globalOverrides?.ingestSystemPrompt ?? INGEST_SYSTEM_PROMPT;
854
- const hasDocumentChunk = /\{\{\s*documentChunk\s*\}\}/.test(template);
855
- if (hasDocumentChunk || this.hasOntologyPlaceholders(template)) {
856
- return {
857
- systemPrompt: this.buildSystemPrompt(template, { documentChunk }, ontologyContext),
858
- userPrompt: hasDocumentChunk ? "Please extract the facts." : `Document Chunk:
859
- ${documentChunk}`
860
- };
861
- }
862
- return {
863
- systemPrompt: this.appendOntology(template, ontologyContext),
864
- userPrompt: `Document Chunk:
865
- ${documentChunk}`
866
- };
867
- }
868
- buildLibrarianPrompt(events, currentFacts, runtimeOverride, ontologyContext) {
869
- const template = runtimeOverride ?? this.globalOverrides?.librarianSystemPrompt ?? LIBRARIAN_SYSTEM_PROMPT;
870
- const hasEvents = /\{\{\s*events\s*\}\}/.test(template);
871
- const hasCurrentFacts = /\{\{\s*currentFacts\s*\}\}/.test(template);
872
- if (hasEvents || hasCurrentFacts || this.hasOntologyPlaceholders(template)) {
873
- return {
874
- systemPrompt: this.buildSystemPrompt(template, { events, currentFacts }, ontologyContext),
875
- userPrompt: hasEvents || hasCurrentFacts ? "Please synthesize the context." : `Events:
876
- ${JSON.stringify(events, null, 2)}
877
-
878
- Current Facts:
879
- ${JSON.stringify(currentFacts, null, 2)}`
880
- };
881
- }
882
- return {
883
- systemPrompt: this.appendOntology(template, ontologyContext),
884
- userPrompt: `Events:
885
- ${JSON.stringify(events, null, 2)}
886
-
887
- Current Facts:
888
- ${JSON.stringify(currentFacts, null, 2)}`
889
- };
890
- }
891
- buildHealPrompt(healCandidates, documentAnchors, allTasks, recentEvents, runtimeOverride) {
892
- const template = runtimeOverride ?? this.globalOverrides?.healSystemPrompt ?? HEAL_SYSTEM_PROMPT;
893
- if (/\{\{\s*healCandidates\s*\}\}/.test(template) || /\{\{\s*documentAnchors\s*\}\}/.test(template) || /\{\{\s*allTasks\s*\}\}/.test(template) || /\{\{\s*recentEvents\s*\}\}/.test(template)) {
894
- return {
895
- systemPrompt: this.hydrate(template, { healCandidates, documentAnchors, allTasks, recentEvents }),
896
- userPrompt: "Please heal the memory graph."
897
- };
898
- }
899
- return {
900
- systemPrompt: template,
901
- userPrompt: `Heal Candidates:
902
- ${JSON.stringify(healCandidates, null, 2)}
903
- Document Anchors (DO NOT MODIFY OR DELETE):
904
- ${JSON.stringify(documentAnchors, null, 2)}
905
- All Tasks:
906
- ${JSON.stringify(allTasks, null, 2)}
907
- Recent Events:
908
- ${JSON.stringify(recentEvents, null, 2)}
909
- The following document anchors are provided for contradiction detection only. Do not include them in \`downgraded\`, \`deleted\`, or \`newFacts\`.`
910
- };
911
- }
912
- buildOntologyBackfillPrompt(facts, runtimeOverride, ontologyContext) {
913
- const template = runtimeOverride ?? this.globalOverrides?.ontologyBackfillSystemPrompt ?? ONTOLOGY_BACKFILL_SYSTEM_PROMPT;
914
- const hasFacts = /\{\{\s*facts\s*\}\}/.test(template);
915
- if (hasFacts || this.hasOntologyPlaceholders(template)) {
916
- return {
917
- systemPrompt: this.buildSystemPrompt(template, { facts }, ontologyContext),
918
- userPrompt: hasFacts ? "Please classify the facts." : `Facts:
919
- ${JSON.stringify(facts, null, 2)}`
920
- };
921
- }
922
- return {
923
- systemPrompt: this.appendOntology(template, ontologyContext),
924
- userPrompt: `Facts:
925
- ${JSON.stringify(facts, null, 2)}`
926
- };
927
- }
928
- };
929
-
930
792
  // src/utils/pure.ts
931
793
  function parseJsonResponse(text) {
932
794
  const firstBrace = text.indexOf("{");
@@ -1142,6 +1004,148 @@ function jaccardScore(a, b) {
1142
1004
  return intersection.size / union.size;
1143
1005
  }
1144
1006
 
1007
+ // src/prompts.ts
1008
+ var LIBRARIAN_SYSTEM_PROMPT = `You are a knowledge extraction agent. Your job is to analyze recent episodic events and extract stable facts and actionable tasks about the user or entity.
1009
+ Return ONLY a valid JSON object matching this schema:
1010
+ {
1011
+ "facts": [{ "title": "string (max 80 chars)", "body": "string (max 800 chars)", "tags": ["string"], "confidence": "certain|inferred|tentative" }],
1012
+ "tasks": [{ "description": "string", "priority": "number (0-10)" }]
1013
+ }
1014
+ Keep facts concise. Do not return markdown, just raw JSON.`;
1015
+ var HEAL_SYSTEM_PROMPT = `You are a memory grooming agent. Your job is to review a full dump of facts and recent events to resolve contradictions, downgrade stale claims, and flag obsolete facts for deletion.
1016
+ Return ONLY a valid JSON object matching this schema:
1017
+ {
1018
+ "downgraded": ["string (fact IDs)"],
1019
+ "deleted": ["string (fact IDs)"],
1020
+ "newFacts": [{ "title": "string (max 80 chars)", "body": "string (max 800 chars)", "tags": ["string"], "confidence": "certain|inferred|tentative" }]
1021
+ }
1022
+ Do not return markdown, just raw JSON.`;
1023
+ var INGEST_SYSTEM_PROMPT = `You are a document ingestion agent. Your job is to extract factual knowledge from the provided document chunk.
1024
+ Return ONLY a valid JSON object matching this schema:
1025
+ {
1026
+ "facts": [{ "title": "string (max 80 chars)", "body": "string (max 800 chars)", "tags": ["string"], "confidence": "certain|inferred|tentative" }]
1027
+ }
1028
+ Extract verbatim factual content. Do not return markdown, just raw JSON.`;
1029
+ var ONTOLOGY_BACKFILL_SYSTEM_PROMPT = `You are a knowledge classification agent. You will receive existing memory facts that currently have no ontology type. For each input fact { "id", "title", "body", "tags" }, assign the best matching okf_type from the ontology manifest and optionally propose edges to related facts by title.
1030
+ Return ONLY a valid JSON object matching this schema:
1031
+ {
1032
+ "classifications": [
1033
+ { "id": "string (input fact id, copied verbatim)", "okf_type": "string (manifest node type slug)", "edges": [{ "edge_type": "string", "target_title": "string" }] }
1034
+ ]
1035
+ }
1036
+ If no manifest type fits a fact, omit that fact from "classifications" entirely \u2014 do not guess.
1037
+ Do not return markdown, just raw JSON.`;
1038
+
1039
+ // src/services/PromptService.ts
1040
+ var PromptService = class {
1041
+ constructor(globalOverrides) {
1042
+ this.globalOverrides = globalOverrides;
1043
+ }
1044
+ hydrate(template, variables) {
1045
+ return template.replace(/\{\{\s*(\w+)\s*\}\}/g, (_match, key) => {
1046
+ const value = variables[key];
1047
+ if (value === void 0) return _match;
1048
+ return typeof value === "string" ? value : JSON.stringify(value, null, 2);
1049
+ });
1050
+ }
1051
+ hasOntologyPlaceholders(template) {
1052
+ return /\{\{\s*ontology(?:Manifest|ModeInstructions)\s*\}\}/.test(template);
1053
+ }
1054
+ buildSystemPrompt(template, variables, ontologyContext) {
1055
+ const shouldHydrate = Object.keys(variables).some(
1056
+ (key) => new RegExp(`\\{\\{\\s*${key}\\s*\\}\\}`).test(template)
1057
+ ) || ontologyContext != null && this.hasOntologyPlaceholders(template);
1058
+ const hydrated = shouldHydrate ? this.hydrate(template, { ...variables, ...ontologyContext ?? {} }) : template;
1059
+ return this.hasOntologyPlaceholders(template) ? ontologyContext != null ? hydrated : hydrated.replace(/\{\{\s*ontology(?:Manifest|ModeInstructions)\s*\}\}/g, "") : this.appendOntology(hydrated, ontologyContext);
1060
+ }
1061
+ appendOntology(systemPrompt, ctx) {
1062
+ if (!ctx) return systemPrompt;
1063
+ return `${systemPrompt}
1064
+
1065
+ ${ctx.ontologyModeInstructions}`;
1066
+ }
1067
+ buildIngestPrompt(documentChunk, runtimeOverride, ontologyContext) {
1068
+ const template = runtimeOverride ?? this.globalOverrides?.ingestSystemPrompt ?? INGEST_SYSTEM_PROMPT;
1069
+ const hasDocumentChunk = /\{\{\s*documentChunk\s*\}\}/.test(template);
1070
+ if (hasDocumentChunk || this.hasOntologyPlaceholders(template)) {
1071
+ return {
1072
+ systemPrompt: this.buildSystemPrompt(template, { documentChunk }, ontologyContext),
1073
+ userPrompt: hasDocumentChunk ? "Please extract the facts." : `Document Chunk:
1074
+ ${documentChunk}`
1075
+ };
1076
+ }
1077
+ return {
1078
+ systemPrompt: this.appendOntology(template, ontologyContext),
1079
+ userPrompt: `Document Chunk:
1080
+ ${documentChunk}`
1081
+ };
1082
+ }
1083
+ buildLibrarianPrompt(events, currentFacts, runtimeOverride, ontologyContext) {
1084
+ const template = runtimeOverride ?? this.globalOverrides?.librarianSystemPrompt ?? LIBRARIAN_SYSTEM_PROMPT;
1085
+ const hasEvents = /\{\{\s*events\s*\}\}/.test(template);
1086
+ const hasCurrentFacts = /\{\{\s*currentFacts\s*\}\}/.test(template);
1087
+ if (hasEvents || hasCurrentFacts || this.hasOntologyPlaceholders(template)) {
1088
+ return {
1089
+ systemPrompt: this.buildSystemPrompt(template, { events, currentFacts }, ontologyContext),
1090
+ userPrompt: hasEvents || hasCurrentFacts ? "Please synthesize the context." : `Events:
1091
+ ${JSON.stringify(events, null, 2)}
1092
+
1093
+ Current Facts:
1094
+ ${JSON.stringify(currentFacts, null, 2)}`
1095
+ };
1096
+ }
1097
+ return {
1098
+ systemPrompt: this.appendOntology(template, ontologyContext),
1099
+ userPrompt: `Events:
1100
+ ${JSON.stringify(events, null, 2)}
1101
+
1102
+ Current Facts:
1103
+ ${JSON.stringify(currentFacts, null, 2)}`
1104
+ };
1105
+ }
1106
+ buildHealPrompt(healCandidates, documentAnchors, allTasks, recentEvents, runtimeOverride) {
1107
+ const template = runtimeOverride ?? this.globalOverrides?.healSystemPrompt ?? HEAL_SYSTEM_PROMPT;
1108
+ if (/\{\{\s*healCandidates\s*\}\}/.test(template) || /\{\{\s*documentAnchors\s*\}\}/.test(template) || /\{\{\s*allTasks\s*\}\}/.test(template) || /\{\{\s*recentEvents\s*\}\}/.test(template)) {
1109
+ return {
1110
+ systemPrompt: this.hydrate(template, { healCandidates, documentAnchors, allTasks, recentEvents }),
1111
+ userPrompt: "Please heal the memory graph."
1112
+ };
1113
+ }
1114
+ return {
1115
+ systemPrompt: template,
1116
+ userPrompt: `Heal Candidates:
1117
+ ${JSON.stringify(healCandidates, null, 2)}
1118
+ Document Anchors (DO NOT MODIFY OR DELETE):
1119
+ ${JSON.stringify(documentAnchors, null, 2)}
1120
+ All Tasks:
1121
+ ${JSON.stringify(allTasks, null, 2)}
1122
+ Recent Events:
1123
+ ${JSON.stringify(recentEvents, null, 2)}
1124
+ The following document anchors are provided for contradiction detection only. Do not include them in \`downgraded\`, \`deleted\`, or \`newFacts\`.`
1125
+ };
1126
+ }
1127
+ buildOntologyBackfillPrompt(facts, runtimeOverride, ontologyContext) {
1128
+ const template = runtimeOverride ?? this.globalOverrides?.ontologyBackfillSystemPrompt ?? ONTOLOGY_BACKFILL_SYSTEM_PROMPT;
1129
+ const hasFacts = /\{\{\s*facts\s*\}\}/.test(template);
1130
+ if (hasFacts || this.hasOntologyPlaceholders(template)) {
1131
+ return {
1132
+ systemPrompt: this.buildSystemPrompt(template, { facts }, ontologyContext),
1133
+ userPrompt: hasFacts ? "Please classify the facts." : `Facts:
1134
+ ${JSON.stringify(facts, null, 2)}`
1135
+ };
1136
+ }
1137
+ return {
1138
+ systemPrompt: this.appendOntology(template, ontologyContext),
1139
+ userPrompt: `Facts:
1140
+ ${JSON.stringify(facts, null, 2)}`
1141
+ };
1142
+ }
1143
+ };
1144
+
1145
+ // src/utils/chunkingDefaults.ts
1146
+ var DEFAULT_MAX_CHUNK_LENGTH = 12e3;
1147
+ var DEFAULT_CHUNK_OVERLAP = 400;
1148
+
1145
1149
  // src/services/IngestionService.ts
1146
1150
  var IngestionService = class {
1147
1151
  constructor(db, prefix, options, entryRepo, searchService, jobManager, embeddingService, promptService, ontologyService) {
@@ -1160,10 +1164,10 @@ var IngestionService = class {
1160
1164
  if (!sourceRef) throw new Error("Invalid sourceRef");
1161
1165
  const sourceHash = normalizeSourceHash(params.sourceHash);
1162
1166
  if (!sourceHash) throw new Error("Invalid sourceHash (must be 64-char hex string)");
1163
- const maxChunkLength = params.maxChunkLength ?? this.options.config?.maxChunkLength ?? 12e3;
1164
- const rawOverlap = params.chunkOverlap ?? this.options.config?.chunkOverlap ?? 400;
1167
+ const maxChunkLength = params.maxChunkLength ?? this.options.config?.maxChunkLength ?? DEFAULT_MAX_CHUNK_LENGTH;
1168
+ const rawOverlap = params.chunkOverlap ?? this.options.config?.chunkOverlap ?? DEFAULT_CHUNK_OVERLAP;
1165
1169
  const chunkOverlap = Math.min(
1166
- Number.isFinite(rawOverlap) && rawOverlap >= 0 ? Math.floor(rawOverlap) : 400,
1170
+ Number.isFinite(rawOverlap) && rawOverlap >= 0 ? Math.floor(rawOverlap) : DEFAULT_CHUNK_OVERLAP,
1167
1171
  maxChunkLength - 1
1168
1172
  );
1169
1173
  const rawConcurrency = params.chunkConcurrency ?? this.options.config?.chunkConcurrency ?? 1;
@@ -1569,6 +1573,8 @@ var ONTOLOGY_BACKFILL_RECHECK_MS = 7 * 24 * 60 * 60 * 1e3;
1569
1573
  var HEAL_MAX_ANCHORS = 50;
1570
1574
  var HEAL_ANCHOR_SEARCH_OVERFETCH = 4;
1571
1575
  var HEAL_MAX_PROMPT_CHARS = 4e4;
1576
+ var HEAL_BATCH_SIZE = 25;
1577
+ var HEAL_RECHECK_MS = 7 * 24 * 60 * 60 * 1e3;
1572
1578
  var MaintenanceService = class {
1573
1579
  constructor(db, prefix, options, entryRepo, taskRepo, eventRepo, metadataRepo, searchService, jobManager, embeddingService, promptService, ontologyService) {
1574
1580
  this.db = db;
@@ -1669,7 +1675,7 @@ var MaintenanceService = class {
1669
1675
  async runHeal(entityId, options) {
1670
1676
  this.jobManager.acquireLock("heal", entityId);
1671
1677
  try {
1672
- await this.doRunHeal(entityId, options?.promptOverride);
1678
+ return await this.doRunHeal(entityId, options);
1673
1679
  } finally {
1674
1680
  this.jobManager.releaseLock("heal", entityId);
1675
1681
  }
@@ -1921,9 +1927,21 @@ var MaintenanceService = class {
1921
1927
  }
1922
1928
  this.searchService.evictCache(entityId);
1923
1929
  }
1924
- /** Core heal pass (locks handled by {@link runHeal}). Package-internal orchestration hook. */
1925
- async doRunHeal(entityId, promptOverride) {
1930
+ /**
1931
+ * Core heal pass (locks handled by {@link runHeal}). Package-internal orchestration hook.
1932
+ *
1933
+ * Bounded: at most `batchSize` candidates per pass (#67). Loop on
1934
+ * `result.remaining > 0` for convergence — see {@link HealResult.remaining}
1935
+ * for what convergence means here.
1936
+ */
1937
+ async doRunHeal(entityId, options) {
1938
+ const promptOverride = options?.promptOverride;
1939
+ const batchSize = options?.batchSize ?? HEAL_BATCH_SIZE;
1940
+ if (!Number.isInteger(batchSize) || batchSize < 1) {
1941
+ throw new Error("Invalid batchSize: must be an integer >= 1");
1942
+ }
1926
1943
  const now = Date.now();
1944
+ const recheckCutoff = now - HEAL_RECHECK_MS;
1927
1945
  const orphanAfterDays = this.options.config?.orphanAfterDays !== void 0 ? this.options.config?.orphanAfterDays : 30;
1928
1946
  const staleInferredAfterDays = this.options.config?.staleInferredAfterDays !== void 0 ? this.options.config?.staleInferredAfterDays : 60;
1929
1947
  const MS_PER_DAY = 24 * 60 * 60 * 1e3;
@@ -1934,6 +1952,7 @@ var MaintenanceService = class {
1934
1952
  throw new Error("Invalid staleInferredAfterDays: must be a finite number >= 0 or null");
1935
1953
  }
1936
1954
  const orphanedIds = [];
1955
+ const staleDowngradedIds = [];
1937
1956
  await this.db.withTransactionAsync(async (tx) => {
1938
1957
  if (orphanAfterDays !== null) {
1939
1958
  const orphanThreshold = now - orphanAfterDays * MS_PER_DAY;
@@ -1941,7 +1960,7 @@ var MaintenanceService = class {
1941
1960
  }
1942
1961
  if (staleInferredAfterDays !== null) {
1943
1962
  const staleThreshold = now - staleInferredAfterDays * MS_PER_DAY;
1944
- await this.entryRepo.downgradeStaleInferred(entityId, staleThreshold, tx);
1963
+ staleDowngradedIds.push(...await this.entryRepo.downgradeStaleInferred(entityId, staleThreshold, tx));
1945
1964
  }
1946
1965
  });
1947
1966
  for (const factId of orphanedIds) {
@@ -1951,7 +1970,21 @@ var MaintenanceService = class {
1951
1970
  console.warn(`[WikiMemory] onEmbeddingPersisted hook failed during heal orphan pass for ${factId}:`, hookErr);
1952
1971
  }
1953
1972
  }
1954
- const healCandidates = await this.entryRepo.findHealCandidatesByEntityId(entityId);
1973
+ const healCandidates = await this.entryRepo.findHealCandidatesByEntityId(entityId, batchSize, recheckCutoff);
1974
+ if (healCandidates.length === 0) {
1975
+ await this.searchService.sync(entityId);
1976
+ this.searchService.evictCache(entityId);
1977
+ const counts2 = await this.entryRepo.countHealCandidatesByEntityId(entityId, recheckCutoff);
1978
+ return {
1979
+ scanned: 0,
1980
+ downgraded: staleDowngradedIds.length,
1981
+ deleted: orphanedIds.length,
1982
+ newFactsCreated: 0,
1983
+ skipped: 0,
1984
+ remaining: counts2.eligible,
1985
+ deferred: counts2.deferred
1986
+ };
1987
+ }
1955
1988
  const allTasks = await this.taskRepo.findAllPending([entityId]);
1956
1989
  const recentEvents = await this.eventRepo.getRecent(entityId, 20);
1957
1990
  const toPromptShape = (f) => {
@@ -2004,7 +2037,7 @@ var MaintenanceService = class {
2004
2037
  const validNewFacts = newFacts.map(validateFact).filter((f) => f !== null);
2005
2038
  const insertedFacts = [];
2006
2039
  const uniqueDeletedFactIds = Array.from(new Set(safeDeleted));
2007
- const healFactsForDedupe = [...healCandidates];
2040
+ const healFactsForDedupe = (await this.entryRepo.findInferredTitlesByEntityId(entityId)).filter((f) => !safeDeletedSet.has(f.id));
2008
2041
  await this.db.withTransactionAsync(async (tx) => {
2009
2042
  await this.entryRepo.downgradeByIds(safeDowngraded, entityId, tx);
2010
2043
  await this.entryRepo.softDeleteByIds(safeDeleted, entityId, tx);
@@ -2013,7 +2046,6 @@ var MaintenanceService = class {
2013
2046
  let skip = false;
2014
2047
  if (newTokens.size >= MIN_TOKENS_TO_QUALIFY) {
2015
2048
  for (const existing of healFactsForDedupe) {
2016
- if (existing.source_type !== "librarian_inferred") continue;
2017
2049
  const existingTokens = titleTokens(existing.title);
2018
2050
  if (existingTokens.size >= MIN_TOKENS_TO_QUALIFY) {
2019
2051
  if (jaccardScore(newTokens, existingTokens) >= FUZZY_THRESHOLD) {
@@ -2043,8 +2075,14 @@ var MaintenanceService = class {
2043
2075
  };
2044
2076
  await this.entryRepo.upsert(factObj, tx);
2045
2077
  insertedFacts.push({ id, entity_id: entityId, title: fact.title, body: fact.body, tags: JSON.stringify(fact.tags) });
2046
- healFactsForDedupe.push(factObj);
2078
+ healFactsForDedupe.push({ id, title: fact.title });
2047
2079
  }
2080
+ await this.entryRepo.markHealChecked(
2081
+ [...healCandidates.map((f) => f.id), ...insertedFacts.map((f) => f.id)],
2082
+ entityId,
2083
+ now,
2084
+ tx
2085
+ );
2048
2086
  });
2049
2087
  await this.searchService.sync(entityId);
2050
2088
  for (const factId of uniqueDeletedFactIds) {
@@ -2058,6 +2096,20 @@ var MaintenanceService = class {
2058
2096
  await this.embeddingService.embedFact(fact);
2059
2097
  }
2060
2098
  this.searchService.evictCache(entityId);
2099
+ let scanned = outcome.skipped.length;
2100
+ for (const batchResult of outcome.results) scanned += batchResult.batch.length;
2101
+ const allDowngraded = /* @__PURE__ */ new Set([...staleDowngradedIds, ...safeDowngraded]);
2102
+ const allDeleted = /* @__PURE__ */ new Set([...orphanedIds, ...uniqueDeletedFactIds]);
2103
+ const counts = await this.entryRepo.countHealCandidatesByEntityId(entityId, recheckCutoff);
2104
+ return {
2105
+ scanned,
2106
+ downgraded: allDowngraded.size,
2107
+ deleted: allDeleted.size,
2108
+ newFactsCreated: insertedFacts.length,
2109
+ skipped: outcome.skipped.length,
2110
+ remaining: counts.eligible,
2111
+ deferred: counts.deferred
2112
+ };
2061
2113
  }
2062
2114
  /** Core ontology backfill pass (locks handled by {@link runOntologyBackfill}). Package-internal orchestration hook. */
2063
2115
  async doRunOntologyBackfill(entityId, options) {
@@ -3433,6 +3485,7 @@ var WriteService = class {
3433
3485
  let shouldRunLibrarian = false;
3434
3486
  let librarianCount = 0;
3435
3487
  let prevMemoryCheckpoint = 0;
3488
+ let eventCount = 0;
3436
3489
  await this.db.withTransactionAsync(async (tx) => {
3437
3490
  await this.eventRepo.add(newEvent, tx);
3438
3491
  const threshold = this.options.config?.autoLibrarianThreshold || 20;
@@ -3440,6 +3493,7 @@ var WriteService = class {
3440
3493
  this.eventRepo.count(entityId, tx),
3441
3494
  this.metadataRepo.getCheckpoint(entityId, tx)
3442
3495
  ]);
3496
+ eventCount = count;
3443
3497
  let memoryCheckpoint = cp.memory ?? 0;
3444
3498
  if (memoryCheckpoint > count) memoryCheckpoint = 0;
3445
3499
  if (count - memoryCheckpoint >= threshold) {
@@ -3461,6 +3515,8 @@ var WriteService = class {
3461
3515
  if (!(e instanceof WikiBusyError)) throw e;
3462
3516
  await this.metadataRepo.updateCheckpoint(entityId, { memory: prevMemoryCheckpoint }, this.db);
3463
3517
  }
3518
+ } else if (!this.jobManager.isBlocked("librarian", entityId)) {
3519
+ this.maybeRunHeal(entityId, eventCount).catch(console.error);
3464
3520
  }
3465
3521
  }
3466
3522
  async runLibrarianThenMaybeHeal(entityId, currentEventCount, prevCheckpoint) {
@@ -3471,6 +3527,14 @@ var WriteService = class {
3471
3527
  await this.metadataRepo.updateCheckpoint(entityId, { memory: prevCheckpoint }, this.db);
3472
3528
  throw e;
3473
3529
  }
3530
+ await this.maybeRunHeal(entityId, currentEventCount);
3531
+ }
3532
+ /**
3533
+ * Run one bounded auto-heal pass if the heal checkpoint has fallen
3534
+ * `autoHealThreshold` events behind. Called after every write (see
3535
+ * {@link write}) so a partial pass retries on the next write.
3536
+ */
3537
+ async maybeRunHeal(entityId, currentEventCount) {
3474
3538
  const autoHealThreshold = this.options.config?.autoHealThreshold || 100;
3475
3539
  const cp = await this.metadataRepo.getCheckpoint(entityId, this.db);
3476
3540
  let healCheckpoint = cp.heal ?? 0;
@@ -3478,8 +3542,10 @@ var WriteService = class {
3478
3542
  const shouldRunHeal = currentEventCount - healCheckpoint >= autoHealThreshold;
3479
3543
  if (shouldRunHeal && this.jobManager.tryAcquireAutoHealLock(entityId)) {
3480
3544
  try {
3481
- await this.maintenanceService.doRunHeal(entityId);
3482
- await this.metadataRepo.updateCheckpoint(entityId, { heal: currentEventCount }, this.db);
3545
+ const result = await this.maintenanceService.doRunHeal(entityId);
3546
+ if (result.remaining === 0) {
3547
+ await this.metadataRepo.updateCheckpoint(entityId, { heal: currentEventCount }, this.db);
3548
+ }
3483
3549
  } finally {
3484
3550
  this.jobManager.releaseLock("heal", entityId);
3485
3551
  }
@@ -3487,6 +3553,6 @@ var WriteService = class {
3487
3553
  }
3488
3554
  };
3489
3555
 
3490
- export { BaseRepository, EmbeddingService, HOOK_TIMEOUT_MARKER, ImportExportService, IngestionService, JobManager, MaintenanceService, MetadataRepository, ONTOLOGY_BACKFILL_BATCH_SIZE, ONTOLOGY_BACKFILL_MAX_PROMPT_CHARS, ONTOLOGY_BACKFILL_RECHECK_MS, ONTOLOGY_BACKFILL_SYSTEM_PROMPT, PromptService, PrunePartialFailureError, RetrievalService, SearchService, WikiBusyError, WikiTransactionError, WriteService, __privateAdd, __privateGet, __privateSet, configureRandomSource, emptyManifest, entitySummaryMetaKey, extractSqliteCode, generateId, normalizeSourceHash, normalizeSourceRef, normalizeTitleKey, parseEmbedding, resolveEdgeDefinitions, resolveNodeType, validateInlineEdges, validateManifest };
3491
- //# sourceMappingURL=chunk-MYZJLVX4.mjs.map
3492
- //# sourceMappingURL=chunk-MYZJLVX4.mjs.map
3556
+ export { BaseRepository, DEFAULT_CHUNK_OVERLAP, DEFAULT_MAX_CHUNK_LENGTH, EmbeddingService, HEAL_BATCH_SIZE, HEAL_RECHECK_MS, HOOK_TIMEOUT_MARKER, ImportExportService, IngestionService, JobManager, MaintenanceService, MetadataRepository, ONTOLOGY_BACKFILL_BATCH_SIZE, ONTOLOGY_BACKFILL_MAX_PROMPT_CHARS, ONTOLOGY_BACKFILL_RECHECK_MS, ONTOLOGY_BACKFILL_SYSTEM_PROMPT, PromptService, PrunePartialFailureError, RetrievalService, SearchService, WikiBusyError, WikiTransactionError, WriteService, __privateAdd, __privateGet, __privateSet, chunkText, configureRandomSource, emptyManifest, entitySummaryMetaKey, extractSqliteCode, generateId, normalizeSourceHash, normalizeSourceRef, normalizeTitleKey, parseEmbedding, resolveEdgeDefinitions, resolveNodeType, safeSlice, validateInlineEdges, validateManifest };
3557
+ //# sourceMappingURL=chunk-DM7BKDCV.mjs.map
3558
+ //# sourceMappingURL=chunk-DM7BKDCV.mjs.map