@equationalapplications/core-llm-wiki 7.6.0 → 7.7.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +64 -0
- package/dist/{chunk-YBWE26XR.mjs → chunk-NWODQRSM.mjs} +52 -18
- package/dist/chunk-NWODQRSM.mjs.map +1 -0
- package/dist/index.d.mts +2 -2
- package/dist/index.d.ts +2 -2
- package/dist/index.js +250 -16
- package/dist/index.js.map +1 -1
- package/dist/index.mjs +203 -3
- package/dist/index.mjs.map +1 -1
- package/dist/{testing-d1SrpDvg.d.mts → testing-A4Sofxyo.d.mts} +85 -2
- package/dist/{testing-d1SrpDvg.d.ts → testing-A4Sofxyo.d.ts} +85 -2
- package/dist/testing.d.mts +1 -1
- package/dist/testing.d.ts +1 -1
- package/dist/testing.js +45 -15
- package/dist/testing.js.map +1 -1
- package/dist/testing.mjs +1 -1
- package/package.json +2 -2
- package/dist/chunk-YBWE26XR.mjs.map +0 -1
package/README.md
CHANGED
|
@@ -23,6 +23,10 @@ Platform-agnostic TypeScript engine for hybrid LLM memory. Features episodic fac
|
|
|
23
23
|
- **Type-safe** — Built with TypeScript, full type exports
|
|
24
24
|
- **Interoperability:** Supports [Open Knowledge Format (OKF)](https://github.com/GoogleCloudPlatform/knowledge-catalog/tree/main/okf) v0.1 + v0.2 import and export via the [llm-wiki OKF profiles](https://github.com/equationalapplications/expo-llm-wiki/blob/main/docs/okf-profile.md) (default `llm-wiki/2`, back-compat `llm-wiki/1`).
|
|
25
25
|
- **Per-entity seeded ontology** — Optional Strict, Emergent, or Off modes govern LLM graph extraction; seed taxonomies per entity and persist typed facts with inline edges.
|
|
26
|
+
- **Diagnostics** — Optional `onDiagnostic` hook with typed, content-free reports of dropped chunks, facts, edges, embedding failures and background-job failures ([Diagnostics](#diagnostics))
|
|
27
|
+
- **Draft review** — `excludeDrafts` on reads and traversal, plus `listDrafts` / `promoteDraft` ([Draft Review](#draft-review))
|
|
28
|
+
- **Evidence grounding** — Opt-in `grounding` check; facts that don't quote their source are stored as drafts ([Grounding](#grounding))
|
|
29
|
+
- **Optional classifier** — `LLMProvider.classify` types facts during ontology backfill when you opt in with `ontology.backfillClassifier: 'auto'` ([Ontology backfill](#ontology-backfill))
|
|
26
30
|
|
|
27
31
|
## GraphRAG & Multi-Modal Retrieval
|
|
28
32
|
|
|
@@ -143,6 +147,13 @@ const wikiMemory = new WikiMemory(db, {
|
|
|
143
147
|
preFilterLimit: 50, // default: undefined — MiniSearch pre-filter before cosine scan; recommended for >500 facts
|
|
144
148
|
hybridWeight: 0.7, // default: undefined — blend semantic (1.0) ↔ keyword (0.0); pure semantic when unset
|
|
145
149
|
enableOutbox: false, // default: false — when true, entry/task mutations write to an internal SQLite outbox table for external sync (e.g. via @equationalapplications/prisma-outbox)
|
|
150
|
+
excludeDrafts: false, // default: false — engine default for read()/traverseGraph() excludeDrafts; see Draft Review
|
|
151
|
+
grounding: { mode: 'off' }, // default: off — 'draft' checks evidence quotes; see Grounding
|
|
152
|
+
ontology: {
|
|
153
|
+
// mode, seedManifests: see Per-Entity Seeded Ontology
|
|
154
|
+
backfillClassifier: 'llm', // default: 'llm' — 'auto' uses llmProvider.classify when present; see Ontology backfill
|
|
155
|
+
classifyMinConfidence: 0.5, // default: 0.5 — classifier answers below this are left untyped
|
|
156
|
+
},
|
|
146
157
|
|
|
147
158
|
// Global prompt overrides — librarianSystemPrompt and healSystemPrompt apply to write() auto-runs;
|
|
148
159
|
// ingestSystemPrompt applies only to explicit ingestDocument() calls.
|
|
@@ -154,6 +165,8 @@ const wikiMemory = new WikiMemory(db, {
|
|
|
154
165
|
healSystemPrompt: `Fix the memory graph based on these candidates: {{healCandidates}}\n\nReturn ONLY valid JSON: { "downgraded": ["factId"], "deleted": ["factId"], "newFacts": [{ "title": "string", "body": "string", "tags": ["string"], "confidence": "certain|inferred|tentative" }] }. No markdown.`,
|
|
155
166
|
},
|
|
156
167
|
},
|
|
168
|
+
// Host callbacks sit beside llmProvider, not inside config:
|
|
169
|
+
// onDiagnostic: (d) => { ... }, // see Diagnostics
|
|
157
170
|
});
|
|
158
171
|
```
|
|
159
172
|
|
|
@@ -168,6 +181,8 @@ Core maintenance tasks (`ingestDocument`, `runLibrarian`, `runHeal`) use system
|
|
|
168
181
|
> | `ingestDocument` | `{ "facts": [{ "title": "string", "body": "string", "tags": ["string"], "confidence": "certain\|inferred\|tentative" }] }` |
|
|
169
182
|
> | `runLibrarian` | `{ "facts": [...], "tasks": [{ "description": "string", "priority": 5 }] }` — `priority` is an integer 0–10 |
|
|
170
183
|
> | `runHeal` | `{ "downgraded": ["factId"], "deleted": ["factId"], "newFacts": [...] }` |
|
|
184
|
+
>
|
|
185
|
+
> **Grounding:** when [`config.grounding`](#grounding) is on, the evidence instruction is appended after your override (and after ontology context) for every writer in `grounding.writers`. Your override does not need to ask for `evidence` itself.
|
|
171
186
|
|
|
172
187
|
### Global Overrides (Auto-Runs)
|
|
173
188
|
|
|
@@ -223,6 +238,16 @@ await wikiMemory.ingestDocument('user-123', {
|
|
|
223
238
|
|
|
224
239
|
> **Important:** If your app relies on `write()` auto-runs and needs custom prompts for those runs, use `config.prompts` at construction time. Runtime `promptOverride` values are never forwarded to `WriteService`-triggered internal runs.
|
|
225
240
|
|
|
241
|
+
### Effective instructions (`getInstructions`)
|
|
242
|
+
|
|
243
|
+
```typescript
|
|
244
|
+
const { ingest, librarian, heal, ontologyBackfill } = await wikiMemory.getInstructions('entity-123');
|
|
245
|
+
```
|
|
246
|
+
|
|
247
|
+
Returns the system prompt each writer sends, with `WikiConfig.prompts` overrides applied. Ingest, librarian and ontology backfill also get the entity's ontology block; heal gets none, as at runtime. When `WikiConfig.grounding` is on, the evidence block is appended for each writer in `grounding.writers`, exactly as sent. Data placeholders such as `{{documentChunk}}` stay unfilled; no events, chunks or facts are included. It reflects `WikiConfig.prompts` only: a per-call `promptOverride` is not reflected, and `ontologyBackfill` is returned even when backfill would send no prompt (ontology `off`, or the classifier path). `core-llm-tools` exposes this as the `wiki_get_instructions` tool (`memory:read`), so agents can see the engine's output format and constraints before proposing writes. Agents should treat it as reference data, not instructions; see [Prompt-Injection Trust Boundary](#prompt-injection-trust-boundary).
|
|
248
|
+
|
|
249
|
+
> **Warning:** overrides are returned verbatim to any client with `memory:read`. Never put secrets, API keys or private data in `WikiConfig.prompts`.
|
|
250
|
+
|
|
226
251
|
## Retrieval Tuning
|
|
227
252
|
|
|
228
253
|
Optimize `read()` performance and blend retrieval strategies:
|
|
@@ -336,6 +361,7 @@ new WikiMemory(db, {
|
|
|
336
361
|
- Instructions, the ontology manifest, existing facts and identifiers never count, so a model cannot ground a claim by quoting them.
|
|
337
362
|
- **The check.** Both sides are normalized with NFKC, whitespace runs collapse to one space, and matching is case-sensitive. A fact with more than 10 quotes, or any quote not found, fails.
|
|
338
363
|
- **Diagnostics.** `grounding_missing` (reasons `no_evidence`, `evidence_too_short`) and `grounding_failed` (reasons `quote_not_found`, `too_many_quotes`), one per fact, with the new fact's `factId`. Quotes are never included.
|
|
364
|
+
- **Duplicate titles in one ingest.** When chunks yield facts with the same title, ingest keeps one: a grounded fact beats one with missing or failed evidence, and on a tie the first one wins. The others are reported as `fact_deduplicated`. A fact already stored for the same `sourceRef` is not replaced by a later partial ingest.
|
|
339
365
|
- **`upsertGraph`** nodes are host-supplied and never grounded.
|
|
340
366
|
- **Librarian and heal** synthesize across events, so their pass rates are unknown. Measure them on your own event log before opting them in.
|
|
341
367
|
- Evidence quotes are not stored.
|
|
@@ -976,6 +1002,14 @@ If your application accepts untrusted input that flows into `write()`, `ingestDo
|
|
|
976
1002
|
treat the LLM's librarian/heal output as similarly untrusted — validate or scope it before acting on it
|
|
977
1003
|
downstream.
|
|
978
1004
|
|
|
1005
|
+
[Grounding](#grounding) is a support check, not an injection defense. It confirms that a fact quotes the
|
|
1006
|
+
text the model was shown, and injected text in that source can be quoted like any other.
|
|
1007
|
+
|
|
1008
|
+
[`getInstructions`](#effective-instructions-getinstructions) and the `wiki_get_instructions` tool (`memory:read`)
|
|
1009
|
+
return `WikiConfig.prompts` overrides verbatim, so keep secrets out of them. The result also includes the entity's
|
|
1010
|
+
ontology manifest, which in emergent mode holds types and descriptions the model proposed from ingested
|
|
1011
|
+
documents. Agents should treat the result as reference data about the engine, not as instructions to follow.
|
|
1012
|
+
|
|
979
1013
|
## Usage
|
|
980
1014
|
|
|
981
1015
|
```typescript
|
|
@@ -1200,6 +1234,36 @@ const changes = await wikiMemory.hasChanged('entity-123', batch);
|
|
|
1200
1234
|
// Per-document change detection; internally batched across queries
|
|
1201
1235
|
```
|
|
1202
1236
|
|
|
1237
|
+
`pendingSources` returns one status per input, in order, and adds the partial-ingest state:
|
|
1238
|
+
|
|
1239
|
+
```typescript
|
|
1240
|
+
const statuses = await wikiMemory.pendingSources('entity-123', batch);
|
|
1241
|
+
// Array<{ sourceRef: string; status: 'new' | 'changed' | 'partial' | 'current' }>
|
|
1242
|
+
```
|
|
1243
|
+
|
|
1244
|
+
- `current` means exactly what `hasChanged` returning `false` means.
|
|
1245
|
+
- `partial` means live facts exist for the ref, but none has a stored hash: for example a first ingest where a chunk failed, or imported rows that carry no hash. Re-ingest to retry. A failed re-ingest of a ref that already has hashed rows reports `changed`, not `partial`.
|
|
1246
|
+
|
|
1247
|
+
## Lint
|
|
1248
|
+
|
|
1249
|
+
Read-only health report for one entity. It reports problems and never repairs them.
|
|
1250
|
+
|
|
1251
|
+
```typescript
|
|
1252
|
+
const report = await wikiMemory.lint('entity-123');
|
|
1253
|
+
// {
|
|
1254
|
+
// danglingEdges, // source or target missing, soft-deleted, or another entity's
|
|
1255
|
+
// manifestViolations, // (source type, edge type, target type) not in the effective manifest
|
|
1256
|
+
// untypedFacts, // okf_type is null
|
|
1257
|
+
// drafts, // lifecycle_status = 'draft' (see Draft Review)
|
|
1258
|
+
// unverifiedInferred, // librarian_inferred facts with no okf_verified entry
|
|
1259
|
+
// sample: { danglingEdgeIds, manifestViolationEdgeIds }, // up to 20 each
|
|
1260
|
+
// }
|
|
1261
|
+
```
|
|
1262
|
+
|
|
1263
|
+
- Manifest violations are 0 when ontology is off or the manifest is empty.
|
|
1264
|
+
- An edge with an untyped endpoint counts as a violation.
|
|
1265
|
+
- Partial-ingest rows are not reported here; use `pendingSources`.
|
|
1266
|
+
|
|
1203
1267
|
## Dry-Run Deletion
|
|
1204
1268
|
|
|
1205
1269
|
Preview deletion impact without writing:
|
|
@@ -233,6 +233,10 @@ function typeSatisfies(declaredType, concreteType, manifest) {
|
|
|
233
233
|
const parent = typeof def?.parent_type === "string" ? def.parent_type.trim().toLowerCase() : "";
|
|
234
234
|
return parent !== "" && parent === declared;
|
|
235
235
|
}
|
|
236
|
+
function edgeTripleAllowed(manifest, edgeType, sourceType, targetType) {
|
|
237
|
+
const wanted = edgeType.trim().toLowerCase();
|
|
238
|
+
return (manifest.edge_types ?? []).some((d) => typeof d?.type === "string" && d.type.trim().toLowerCase() === wanted && typeof d.source_type === "string" && typeSatisfies(d.source_type, sourceType, manifest) && typeof d.target_type === "string" && typeSatisfies(d.target_type, targetType, manifest));
|
|
239
|
+
}
|
|
236
240
|
function validateManifest(manifest) {
|
|
237
241
|
const nodeSlugs = /* @__PURE__ */ new Set();
|
|
238
242
|
for (const node of manifest.node_types ?? []) {
|
|
@@ -1867,6 +1871,21 @@ ${JSON.stringify(facts, null, 2)}`
|
|
|
1867
1871
|
${JSON.stringify(facts, null, 2)}`
|
|
1868
1872
|
};
|
|
1869
1873
|
}
|
|
1874
|
+
/**
|
|
1875
|
+
* The system prompt each writer would send, without hydrating data (spec
|
|
1876
|
+
* §8.3). `buildSystemPrompt` with no variables hydrates only ontology
|
|
1877
|
+
* placeholders and leaves data placeholders verbatim; heal never receives
|
|
1878
|
+
* ontology context, matching `buildHealPrompt`.
|
|
1879
|
+
*/
|
|
1880
|
+
buildInstructionTemplates(ontologyContext) {
|
|
1881
|
+
const o = this.globalOverrides;
|
|
1882
|
+
return {
|
|
1883
|
+
ingest: this.appendGrounding(this.buildSystemPrompt(o?.ingestSystemPrompt ?? INGEST_SYSTEM_PROMPT, {}, ontologyContext), "ingest"),
|
|
1884
|
+
librarian: this.appendGrounding(this.buildSystemPrompt(o?.librarianSystemPrompt ?? LIBRARIAN_SYSTEM_PROMPT, {}, ontologyContext), "librarian"),
|
|
1885
|
+
heal: this.appendGrounding(o?.healSystemPrompt ?? HEAL_SYSTEM_PROMPT, "heal"),
|
|
1886
|
+
ontologyBackfill: this.buildSystemPrompt(o?.ontologyBackfillSystemPrompt ?? ONTOLOGY_BACKFILL_SYSTEM_PROMPT, {}, ontologyContext)
|
|
1887
|
+
};
|
|
1888
|
+
}
|
|
1870
1889
|
};
|
|
1871
1890
|
function applyBodyTruncation(candidates, attemptLevel, bodyTruncationChars) {
|
|
1872
1891
|
if (attemptLevel < 3) {
|
|
@@ -2147,11 +2166,16 @@ var IngestionService = class {
|
|
|
2147
2166
|
let ingestedChunks = 0;
|
|
2148
2167
|
let failedChunks = 0;
|
|
2149
2168
|
const failures = [];
|
|
2150
|
-
const
|
|
2151
|
-
const orderedChunkFacts = [];
|
|
2152
|
-
const groundingLedger = /* @__PURE__ */ new Map();
|
|
2169
|
+
const winners = /* @__PURE__ */ new Map();
|
|
2153
2170
|
const diagBuffer = new DiagnosticBuffer();
|
|
2154
2171
|
const diagBase = { entityId, operation: "ingest", trigger: "call" };
|
|
2172
|
+
const pushDeduplicated = (chunkIndex, itemIndex) => {
|
|
2173
|
+
diagBuffer.push({
|
|
2174
|
+
...diagBase,
|
|
2175
|
+
code: "fact_deduplicated",
|
|
2176
|
+
detail: { sourceRef, chunkIndex, itemIndex, reason: "exact_title" }
|
|
2177
|
+
});
|
|
2178
|
+
};
|
|
2155
2179
|
for (const [chunkIndex, slot] of chunkResults.entries()) {
|
|
2156
2180
|
if (slot.status === "failed") {
|
|
2157
2181
|
failedChunks++;
|
|
@@ -2162,22 +2186,32 @@ var IngestionService = class {
|
|
|
2162
2186
|
for (const r of slot.rejected) {
|
|
2163
2187
|
diagBuffer.push({ ...diagBase, code: "fact_rejected", detail: { sourceRef, chunkIndex, itemIndex: r.itemIndex, reason: r.reason } });
|
|
2164
2188
|
}
|
|
2165
|
-
const dedupedFacts = [];
|
|
2166
2189
|
slot.facts.forEach((fact, k) => {
|
|
2167
|
-
const
|
|
2168
|
-
|
|
2169
|
-
|
|
2170
|
-
|
|
2171
|
-
|
|
2190
|
+
const key = normalizeTitleKey(fact.title);
|
|
2191
|
+
const candidate = { chunkIndex, k, itemIndex: slot.itemIndexes[k], rank: slot.verdicts[k]?.status === "grounded" ? 1 : 0 };
|
|
2192
|
+
const current = winners.get(key);
|
|
2193
|
+
if (!current) {
|
|
2194
|
+
winners.set(key, candidate);
|
|
2195
|
+
} else if (candidate.rank > current.rank) {
|
|
2196
|
+
pushDeduplicated(current.chunkIndex, current.itemIndex);
|
|
2197
|
+
winners.set(key, candidate);
|
|
2172
2198
|
} else {
|
|
2173
|
-
|
|
2174
|
-
...diagBase,
|
|
2175
|
-
code: "fact_deduplicated",
|
|
2176
|
-
detail: { sourceRef, chunkIndex, itemIndex: slot.itemIndexes[k], reason: "exact_title" }
|
|
2177
|
-
});
|
|
2199
|
+
pushDeduplicated(candidate.chunkIndex, candidate.itemIndex);
|
|
2178
2200
|
}
|
|
2179
2201
|
});
|
|
2180
|
-
|
|
2202
|
+
}
|
|
2203
|
+
const orderedChunkFacts = [];
|
|
2204
|
+
const groundingLedger = /* @__PURE__ */ new Map();
|
|
2205
|
+
for (const [chunkIndex, slot] of chunkResults.entries()) {
|
|
2206
|
+
if (slot.status === "failed") continue;
|
|
2207
|
+
const keptFacts = [];
|
|
2208
|
+
slot.facts.forEach((fact, k) => {
|
|
2209
|
+
const winner = winners.get(normalizeTitleKey(fact.title));
|
|
2210
|
+
if (winner?.chunkIndex !== chunkIndex || winner.k !== k) return;
|
|
2211
|
+
keptFacts.push(fact);
|
|
2212
|
+
if (ingestGrounding) groundingLedger.set(fact, { verdict: slot.verdicts[k], chunkIndex, itemIndex: winner.itemIndex });
|
|
2213
|
+
});
|
|
2214
|
+
orderedChunkFacts.push({ facts: keptFacts, ontology_updates: slot.ontology_updates });
|
|
2181
2215
|
}
|
|
2182
2216
|
for (const reason of ["parse", "llm"]) {
|
|
2183
2217
|
const ofReason = failures.filter((f) => f.source === reason);
|
|
@@ -5606,6 +5640,6 @@ var WriteService = class {
|
|
|
5606
5640
|
}
|
|
5607
5641
|
};
|
|
5608
5642
|
|
|
5609
|
-
export { BaseRepository, DEFAULT_CHUNK_OVERLAP, DEFAULT_MAX_CHUNK_LENGTH, DEFAULT_MAX_EMBED_CHARS, DiagnosticBuffer, EMBED_CHARS_CEILING, EmbeddingService, HEAL_ANCHORS_PER_CANDIDATE, HEAL_BATCH_SIZE, HEAL_MAX_FACT_BODY_CHARS_L3, HEAL_MAX_TASKS, HEAL_RECHECK_MS, HOOK_TIMEOUT_MARKER, ImportExportService, IngestionService, JobManager, MaintenanceService, MetadataRepository, ONTOLOGY_BACKFILL_BATCH_SIZE, ONTOLOGY_BACKFILL_MAX_PROMPT_CHARS, ONTOLOGY_BACKFILL_RECHECK_MS, ONTOLOGY_BACKFILL_SYSTEM_PROMPT, PromptService, PrunePartialFailureError, RetrievalService, SearchService, WikiBusyError, WikiDraftNotFound, WikiDuplicateHashError, WikiGraphNodeOwnershipConflict, WikiIngestEmptyError, WikiInvalidReadOptions, WikiParseError, WikiSourceRefHashCollision, WikiStrictOntologyViolation, WikiTransactionError, WriteService, __privateAdd, __privateGet, __privateSet, chunkText, configureRandomSource, emptyManifest, entitySummaryMetaKey, extractSqliteCode, generateId, normalizeSourceHash, normalizeSourceRef, normalizeTitleKey, parseEmbedding, resolveEdgeDefinitions, resolveGrounding, resolveNodeType, safeSlice, typeSatisfies, validateInlineEdges, validateManifest };
|
|
5610
|
-
//# sourceMappingURL=chunk-
|
|
5611
|
-
//# sourceMappingURL=chunk-
|
|
5643
|
+
export { BaseRepository, DEFAULT_CHUNK_OVERLAP, DEFAULT_MAX_CHUNK_LENGTH, DEFAULT_MAX_EMBED_CHARS, DiagnosticBuffer, EMBED_CHARS_CEILING, EmbeddingService, HEAL_ANCHORS_PER_CANDIDATE, HEAL_BATCH_SIZE, HEAL_MAX_FACT_BODY_CHARS_L3, HEAL_MAX_TASKS, HEAL_RECHECK_MS, HOOK_TIMEOUT_MARKER, ImportExportService, IngestionService, JobManager, MaintenanceService, MetadataRepository, ONTOLOGY_BACKFILL_BATCH_SIZE, ONTOLOGY_BACKFILL_MAX_PROMPT_CHARS, ONTOLOGY_BACKFILL_RECHECK_MS, ONTOLOGY_BACKFILL_SYSTEM_PROMPT, PromptService, PrunePartialFailureError, RetrievalService, SearchService, WikiBusyError, WikiDraftNotFound, WikiDuplicateHashError, WikiGraphNodeOwnershipConflict, WikiIngestEmptyError, WikiInvalidReadOptions, WikiParseError, WikiSourceRefHashCollision, WikiStrictOntologyViolation, WikiTransactionError, WriteService, __privateAdd, __privateGet, __privateSet, chunkText, configureRandomSource, edgeTripleAllowed, emptyManifest, entitySummaryMetaKey, extractSqliteCode, generateId, normalizeSourceHash, normalizeSourceRef, normalizeTitleKey, parseEmbedding, resolveEdgeDefinitions, resolveGrounding, resolveNodeType, safeSlice, typeSatisfies, validateInlineEdges, validateManifest };
|
|
5644
|
+
//# sourceMappingURL=chunk-NWODQRSM.mjs.map
|
|
5645
|
+
//# sourceMappingURL=chunk-NWODQRSM.mjs.map
|