@equationalapplications/core-llm-wiki 7.5.0 → 7.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +93 -0
- package/dist/{chunk-3P7FAKJA.mjs → chunk-VOSMYISQ.mjs} +243 -41
- package/dist/chunk-VOSMYISQ.mjs.map +1 -0
- package/dist/index.d.mts +2 -2
- package/dist/index.d.ts +2 -2
- package/dist/index.js +434 -41
- package/dist/index.js.map +1 -1
- package/dist/index.mjs +196 -5
- package/dist/index.mjs.map +1 -1
- package/dist/{testing-ClpOUiLI.d.mts → testing-A4Sofxyo.d.mts} +136 -4
- package/dist/{testing-ClpOUiLI.d.ts → testing-A4Sofxyo.d.ts} +136 -4
- package/dist/testing.d.mts +1 -1
- package/dist/testing.d.ts +1 -1
- package/dist/testing.js +236 -38
- package/dist/testing.js.map +1 -1
- package/dist/testing.mjs +1 -1
- package/package.json +2 -2
- package/dist/chunk-3P7FAKJA.mjs.map +0 -1
package/README.md
CHANGED
|
@@ -23,6 +23,10 @@ Platform-agnostic TypeScript engine for hybrid LLM memory. Features episodic fac
|
|
|
23
23
|
- **Type-safe** — Built with TypeScript, full type exports
|
|
24
24
|
- **Interoperability:** Supports [Open Knowledge Format (OKF)](https://github.com/GoogleCloudPlatform/knowledge-catalog/tree/main/okf) v0.1 + v0.2 import and export via the [llm-wiki OKF profiles](https://github.com/equationalapplications/expo-llm-wiki/blob/main/docs/okf-profile.md) (default `llm-wiki/2`, back-compat `llm-wiki/1`).
|
|
25
25
|
- **Per-entity seeded ontology** — Optional Strict, Emergent, or Off modes govern LLM graph extraction; seed taxonomies per entity and persist typed facts with inline edges.
|
|
26
|
+
- **Diagnostics** — Optional `onDiagnostic` hook with typed, content-free reports of dropped chunks, facts, edges, embedding failures and background-job failures ([Diagnostics](#diagnostics))
|
|
27
|
+
- **Draft review** — `excludeDrafts` on reads and traversal, plus `listDrafts` / `promoteDraft` ([Draft Review](#draft-review))
|
|
28
|
+
- **Evidence grounding** — Opt-in `grounding` check; facts that don't quote their source are stored as drafts ([Grounding](#grounding))
|
|
29
|
+
- **Optional classifier** — `LLMProvider.classify` types facts during ontology backfill when you opt in with `ontology.backfillClassifier: 'auto'` ([Ontology backfill](#ontology-backfill))
|
|
26
30
|
|
|
27
31
|
## GraphRAG & Multi-Modal Retrieval
|
|
28
32
|
|
|
@@ -143,6 +147,13 @@ const wikiMemory = new WikiMemory(db, {
|
|
|
143
147
|
preFilterLimit: 50, // default: undefined — MiniSearch pre-filter before cosine scan; recommended for >500 facts
|
|
144
148
|
hybridWeight: 0.7, // default: undefined — blend semantic (1.0) ↔ keyword (0.0); pure semantic when unset
|
|
145
149
|
enableOutbox: false, // default: false — when true, entry/task mutations write to an internal SQLite outbox table for external sync (e.g. via @equationalapplications/prisma-outbox)
|
|
150
|
+
excludeDrafts: false, // default: false — engine default for read()/traverseGraph() excludeDrafts; see Draft Review
|
|
151
|
+
grounding: { mode: 'off' }, // default: off — 'draft' checks evidence quotes; see Grounding
|
|
152
|
+
ontology: {
|
|
153
|
+
// mode, seedManifests: see Per-Entity Seeded Ontology
|
|
154
|
+
backfillClassifier: 'llm', // default: 'llm' — 'auto' uses llmProvider.classify when present; see Ontology backfill
|
|
155
|
+
classifyMinConfidence: 0.5, // default: 0.5 — classifier answers below this are left untyped
|
|
156
|
+
},
|
|
146
157
|
|
|
147
158
|
// Global prompt overrides — librarianSystemPrompt and healSystemPrompt apply to write() auto-runs;
|
|
148
159
|
// ingestSystemPrompt applies only to explicit ingestDocument() calls.
|
|
@@ -154,6 +165,8 @@ const wikiMemory = new WikiMemory(db, {
|
|
|
154
165
|
healSystemPrompt: `Fix the memory graph based on these candidates: {{healCandidates}}\n\nReturn ONLY valid JSON: { "downgraded": ["factId"], "deleted": ["factId"], "newFacts": [{ "title": "string", "body": "string", "tags": ["string"], "confidence": "certain|inferred|tentative" }] }. No markdown.`,
|
|
155
166
|
},
|
|
156
167
|
},
|
|
168
|
+
// Host callbacks sit beside llmProvider, not inside config:
|
|
169
|
+
// onDiagnostic: (d) => { ... }, // see Diagnostics
|
|
157
170
|
});
|
|
158
171
|
```
|
|
159
172
|
|
|
@@ -168,6 +181,8 @@ Core maintenance tasks (`ingestDocument`, `runLibrarian`, `runHeal`) use system
|
|
|
168
181
|
> | `ingestDocument` | `{ "facts": [{ "title": "string", "body": "string", "tags": ["string"], "confidence": "certain\|inferred\|tentative" }] }` |
|
|
169
182
|
> | `runLibrarian` | `{ "facts": [...], "tasks": [{ "description": "string", "priority": 5 }] }` — `priority` is an integer 0–10 |
|
|
170
183
|
> | `runHeal` | `{ "downgraded": ["factId"], "deleted": ["factId"], "newFacts": [...] }` |
|
|
184
|
+
>
|
|
185
|
+
> **Grounding:** when [`config.grounding`](#grounding) is on, the evidence instruction is appended after your override (and after ontology context) for every writer in `grounding.writers`. Your override does not need to ask for `evidence` itself.
|
|
171
186
|
|
|
172
187
|
### Global Overrides (Auto-Runs)
|
|
173
188
|
|
|
@@ -223,6 +238,16 @@ await wikiMemory.ingestDocument('user-123', {
|
|
|
223
238
|
|
|
224
239
|
> **Important:** If your app relies on `write()` auto-runs and needs custom prompts for those runs, use `config.prompts` at construction time. Runtime `promptOverride` values are never forwarded to `WriteService`-triggered internal runs.
|
|
225
240
|
|
|
241
|
+
### Effective instructions (`getInstructions`)
|
|
242
|
+
|
|
243
|
+
```typescript
|
|
244
|
+
const { ingest, librarian, heal, ontologyBackfill } = await wikiMemory.getInstructions('entity-123');
|
|
245
|
+
```
|
|
246
|
+
|
|
247
|
+
Returns the system prompt each writer sends, with `WikiConfig.prompts` overrides applied. Ingest, librarian and ontology backfill also get the entity's ontology block; heal gets none, as at runtime. When `WikiConfig.grounding` is on, the evidence block is appended for each writer in `grounding.writers`, exactly as sent. Data placeholders such as `{{documentChunk}}` stay unfilled; no events, chunks or facts are included. It reflects `WikiConfig.prompts` only: a per-call `promptOverride` is not reflected, and `ontologyBackfill` is returned even when backfill would send no prompt (ontology `off`, or the classifier path). `core-llm-tools` exposes this as the `wiki_get_instructions` tool (`memory:read`), so agents can see the engine's output format and constraints before proposing writes. Agents should treat it as reference data, not instructions; see [Prompt-Injection Trust Boundary](#prompt-injection-trust-boundary).
|
|
248
|
+
|
|
249
|
+
> **Warning:** overrides are returned verbatim to any client with `memory:read`. Never put secrets, API keys or private data in `WikiConfig.prompts`.
|
|
250
|
+
|
|
226
251
|
## Retrieval Tuning
|
|
227
252
|
|
|
228
253
|
Optimize `read()` performance and blend retrieval strategies:
|
|
@@ -310,6 +335,36 @@ await wiki.promoteDraft(facts[0].id, 'user-1', { by: 'human:alice' }); // → st
|
|
|
310
335
|
|
|
311
336
|
`promoteDraft` throws `WikiDraftNotFound` when no live draft with that id exists for the entity. The error is contextless by design. Promotion does not change `updated_at`, so a promoted fact keeps its recency position.
|
|
312
337
|
|
|
338
|
+
## Grounding
|
|
339
|
+
|
|
340
|
+
Opt-in, deterministic evidence check for LLM-authored facts. When on, writers you choose must quote the source they were shown. A fact whose quotes are missing or not found is stored as a `draft` (see [Draft Review](#draft-review)), not rejected. A fact whose quotes all check out is stored `stable` with a `process:grounding-check` verifier, so its `trustTier` is `machine-confirmed`. Default off: 7.x write behavior is unchanged.
|
|
341
|
+
|
|
342
|
+
```ts
|
|
343
|
+
new WikiMemory(db, {
|
|
344
|
+
llmProvider,
|
|
345
|
+
config: {
|
|
346
|
+
grounding: {
|
|
347
|
+
mode: 'draft', // 'off' (default) | 'draft'
|
|
348
|
+
writers: ['ingest'], // default ['ingest']; also 'librarian', 'heal'
|
|
349
|
+
minEvidenceChars: 20, // shorter quotes count as absent
|
|
350
|
+
maxEvidence: 3, // quotes asked for per fact
|
|
351
|
+
maxEvidenceChars: 300,
|
|
352
|
+
},
|
|
353
|
+
},
|
|
354
|
+
});
|
|
355
|
+
```
|
|
356
|
+
|
|
357
|
+
- **What counts as source.**
|
|
358
|
+
- Ingest: the chunk text.
|
|
359
|
+
- Librarian: the `summary` of each event in the prompt.
|
|
360
|
+
- Heal: the `summary` of each recent event in the prompt, plus the bodies of non-draft document anchors. When heal is a writer, anchors are shown with their body clipped to 800 characters.
|
|
361
|
+
- Instructions, the ontology manifest, existing facts and identifiers never count, so a model cannot ground a claim by quoting them.
|
|
362
|
+
- **The check.** Both sides are normalized with NFKC, whitespace runs collapse to one space, and matching is case-sensitive. A fact with more than 10 quotes, or any quote not found, fails.
|
|
363
|
+
- **Diagnostics.** `grounding_missing` (reasons `no_evidence`, `evidence_too_short`) and `grounding_failed` (reasons `quote_not_found`, `too_many_quotes`), one per fact, with the new fact's `factId`. Quotes are never included.
|
|
364
|
+
- **`upsertGraph`** nodes are host-supplied and never grounded.
|
|
365
|
+
- **Librarian and heal** synthesize across events, so their pass rates are unknown. Measure them on your own event log before opting them in.
|
|
366
|
+
- Evidence quotes are not stored.
|
|
367
|
+
|
|
313
368
|
## Pluggable Vector Retrieval
|
|
314
369
|
|
|
315
370
|
When your entity corpus grows, in-process cosine similarity scoring becomes a bottleneck. The optional **`VectorRanker`** interface lets you delegate semantic ranking to [**sqlite-vec**](https://github.com/asg017/sqlite-vec), [**sqlite-vss**](https://github.com/asg017/sqlite-vss), or an external vector database while `WikiMemory` handles embedding validation, hybrid scoring, and tier-2 row hydration.
|
|
@@ -946,6 +1001,14 @@ If your application accepts untrusted input that flows into `write()`, `ingestDo
|
|
|
946
1001
|
treat the LLM's librarian/heal output as similarly untrusted — validate or scope it before acting on it
|
|
947
1002
|
downstream.
|
|
948
1003
|
|
|
1004
|
+
[Grounding](#grounding) is a support check, not an injection defense. It confirms that a fact quotes the
|
|
1005
|
+
text the model was shown, and injected text in that source can be quoted like any other.
|
|
1006
|
+
|
|
1007
|
+
[`getInstructions`](#effective-instructions-getinstructions) and the `wiki_get_instructions` tool (`memory:read`)
|
|
1008
|
+
return `WikiConfig.prompts` overrides verbatim, so keep secrets out of them. The result also includes the entity's
|
|
1009
|
+
ontology manifest, which in emergent mode holds types and descriptions the model proposed from ingested
|
|
1010
|
+
documents. Agents should treat the result as reference data about the engine, not as instructions to follow.
|
|
1011
|
+
|
|
949
1012
|
## Usage
|
|
950
1013
|
|
|
951
1014
|
```typescript
|
|
@@ -1170,6 +1233,36 @@ const changes = await wikiMemory.hasChanged('entity-123', batch);
|
|
|
1170
1233
|
// Per-document change detection; internally batched across queries
|
|
1171
1234
|
```
|
|
1172
1235
|
|
|
1236
|
+
`pendingSources` returns one status per input, in order, and adds the partial-ingest state:
|
|
1237
|
+
|
|
1238
|
+
```typescript
|
|
1239
|
+
const statuses = await wikiMemory.pendingSources('entity-123', batch);
|
|
1240
|
+
// Array<{ sourceRef: string; status: 'new' | 'changed' | 'partial' | 'current' }>
|
|
1241
|
+
```
|
|
1242
|
+
|
|
1243
|
+
- `current` means exactly what `hasChanged` returning `false` means.
|
|
1244
|
+
- `partial` means live facts exist for the ref, but none has a stored hash: for example a first ingest where a chunk failed, or imported rows that carry no hash. Re-ingest to retry. A failed re-ingest of a ref that already has hashed rows reports `changed`, not `partial`.
|
|
1245
|
+
|
|
1246
|
+
## Lint
|
|
1247
|
+
|
|
1248
|
+
Read-only health report for one entity. It reports problems and never repairs them.
|
|
1249
|
+
|
|
1250
|
+
```typescript
|
|
1251
|
+
const report = await wikiMemory.lint('entity-123');
|
|
1252
|
+
// {
|
|
1253
|
+
// danglingEdges, // source or target missing, soft-deleted, or another entity's
|
|
1254
|
+
// manifestViolations, // (source type, edge type, target type) not in the effective manifest
|
|
1255
|
+
// untypedFacts, // okf_type is null
|
|
1256
|
+
// drafts, // lifecycle_status = 'draft' (see Draft Review)
|
|
1257
|
+
// unverifiedInferred, // librarian_inferred facts with no okf_verified entry
|
|
1258
|
+
// sample: { danglingEdgeIds, manifestViolationEdgeIds }, // up to 20 each
|
|
1259
|
+
// }
|
|
1260
|
+
```
|
|
1261
|
+
|
|
1262
|
+
- Manifest violations are 0 when ontology is off or the manifest is empty.
|
|
1263
|
+
- An edge with an untyped endpoint counts as a violation.
|
|
1264
|
+
- Partial-ingest rows are not reported here; use `pendingSources`.
|
|
1265
|
+
|
|
1173
1266
|
## Dry-Run Deletion
|
|
1174
1267
|
|
|
1175
1268
|
Preview deletion impact without writing:
|