@equationalapplications/core-llm-wiki 7.4.0 → 7.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +41 -0
- package/dist/{chunk-G5OR2VWD.mjs → chunk-3P7FAKJA.mjs} +141 -9
- package/dist/chunk-3P7FAKJA.mjs.map +1 -0
- package/dist/index.d.mts +2 -2
- package/dist/index.d.ts +2 -2
- package/dist/index.js +139 -7
- package/dist/index.js.map +1 -1
- package/dist/index.mjs +2 -2
- package/dist/index.mjs.map +1 -1
- package/dist/{testing-Dmh1kfkd.d.mts → testing-ClpOUiLI.d.mts} +66 -1
- package/dist/{testing-Dmh1kfkd.d.ts → testing-ClpOUiLI.d.ts} +66 -1
- package/dist/testing.d.mts +1 -1
- package/dist/testing.d.ts +1 -1
- package/dist/testing.js +139 -7
- package/dist/testing.js.map +1 -1
- package/dist/testing.mjs +1 -1
- package/package.json +2 -2
- package/dist/chunk-G5OR2VWD.mjs.map +0 -1
package/README.md
CHANGED
|
@@ -799,6 +799,47 @@ const result = await wiki.runOntologyBackfill(entityId);
|
|
|
799
799
|
`config.prompts.ontologyBackfillSystemPrompt` (template may use `{{facts}}`
|
|
800
800
|
and the ontology placeholders).
|
|
801
801
|
|
|
802
|
+
#### Classifier mode (optional)
|
|
803
|
+
|
|
804
|
+
A System-One classifier (for example TypeSafe's Jev, an OpenJev-style open model, or a local ONNX classifier) can type facts during backfill without generating text. Add `classify` to your provider and opt in:
|
|
805
|
+
|
|
806
|
+
```ts
|
|
807
|
+
const wiki = createWiki(db, {
|
|
808
|
+
llmProvider: { generateText, classify },
|
|
809
|
+
config: { ontology: { backfillClassifier: 'auto', classifyMinConfidence: 0.5 } },
|
|
810
|
+
});
|
|
811
|
+
await wiki.runOntologyBackfill('user-1'); // uses classify
|
|
812
|
+
await wiki.runOntologyBackfill('user-1', { classifier: 'llm' }); // force the generative path
|
|
813
|
+
```
|
|
814
|
+
|
|
815
|
+
- Providing `classify` changes nothing by itself. The default is `'llm'`.
|
|
816
|
+
- Each untyped fact gets one `choice` question over the entity manifest's node types. Answers below `classifyMinConfidence` (default 0.5) are left untyped and retried after the cooldown.
|
|
817
|
+
- **No edges are proposed in classifier mode** (`edgesAdded: 0`): a classifier cannot extract edge targets. Run with `classifier: 'llm'` when you want edges.
|
|
818
|
+
- Manifests with more than 255 node types, or providers without `classify`, use the generative path.
|
|
819
|
+
- Answers are validated as untrusted. Off-list choices and out-of-range probabilities count toward `failedValidation`. A thrown `classify` counts toward `skipped` and is retried on the next pass.
|
|
820
|
+
|
|
821
|
+
Example adapter for Cloudflare Workers AI's `typesafe/jev`. This is illustrative, not a supported package; check the provider's current API reference before use.
|
|
822
|
+
|
|
823
|
+
```ts
|
|
824
|
+
const classify: LLMProvider['classify'] = async ({ state, questions }) => {
|
|
825
|
+
const jevQuestions = Object.fromEntries(Object.entries(questions).map(([key, q]) => [key,
|
|
826
|
+
q.kind === 'choice' ? { type: 'choice', instructions: q.instructions, criteria: Object.fromEntries(q.options.map((o) => [o, o])) }
|
|
827
|
+
: q.kind === 'score' ? { type: 'score', instructions: q.instructions, criteria: q.levels }
|
|
828
|
+
: { type: 'noul', instructions: q.instructions },
|
|
829
|
+
]));
|
|
830
|
+
const res = await env.AI.run('typesafe/jev', { state, questions: jevQuestions });
|
|
831
|
+
const answers = Object.fromEntries(Object.entries(res.answers).map(([key, a]: [string, any]) => [key,
|
|
832
|
+
a.type === 'choice' ? { kind: 'choice', choice: a.choice, confidence: a.confidence, probabilities: a.probabilities }
|
|
833
|
+
: a.type === 'score' ? {
|
|
834
|
+
kind: 'score', score: a.score, confidence: a.confidence,
|
|
835
|
+
probabilities: Object.keys(a.probabilities).sort((x, y) => Number(x) - Number(y)).map((k) => a.probabilities[k]),
|
|
836
|
+
}
|
|
837
|
+
: { kind: 'binary', probability: a.noul },
|
|
838
|
+
]));
|
|
839
|
+
return { answers };
|
|
840
|
+
};
|
|
841
|
+
```
|
|
842
|
+
|
|
802
843
|
## OKF Import/Export
|
|
803
844
|
|
|
804
845
|
The core package integrates with `@equationalapplications/core-okf` to seamlessly adapt wiki data dumps to and from Open Knowledge Format (OKF) bundles (v0.1 and v0.2; `formatOkfBundle` defaults to the v0.2 / `llm-wiki/2` profile).
|
|
@@ -2471,6 +2471,51 @@ var IngestionService = class {
|
|
|
2471
2471
|
}
|
|
2472
2472
|
};
|
|
2473
2473
|
|
|
2474
|
+
// src/utils/classifier.ts
|
|
2475
|
+
var isUnit = (v) => typeof v === "number" && Number.isFinite(v) && v >= 0 && v <= 1;
|
|
2476
|
+
function validateClassifierAnswer(response, key, question) {
|
|
2477
|
+
try {
|
|
2478
|
+
if (response === null || typeof response !== "object") return { ok: false, reason: "malformed" };
|
|
2479
|
+
const answers = response.answers;
|
|
2480
|
+
if (answers === null || typeof answers !== "object") return { ok: false, reason: "missing_answer" };
|
|
2481
|
+
if (!Object.prototype.hasOwnProperty.call(answers, key)) return { ok: false, reason: "missing_answer" };
|
|
2482
|
+
const raw = answers[key];
|
|
2483
|
+
if (raw === null || typeof raw !== "object") return { ok: false, reason: "missing_answer" };
|
|
2484
|
+
const a = raw;
|
|
2485
|
+
if (a.kind !== question.kind) return { ok: false, reason: "kind_mismatch" };
|
|
2486
|
+
if (question.kind === "choice") {
|
|
2487
|
+
if (typeof a.choice !== "string" || !question.options.includes(a.choice)) return { ok: false, reason: "choice_not_offered" };
|
|
2488
|
+
if (!isUnit(a.confidence)) return { ok: false, reason: "invalid_probability" };
|
|
2489
|
+
const probs = a.probabilities;
|
|
2490
|
+
if (probs === null || typeof probs !== "object" || Array.isArray(probs)) return { ok: false, reason: "invalid_probability" };
|
|
2491
|
+
const probabilities = {};
|
|
2492
|
+
for (const [k, v] of Object.entries(probs)) {
|
|
2493
|
+
if (!isUnit(v)) return { ok: false, reason: "invalid_probability" };
|
|
2494
|
+
probabilities[k] = v;
|
|
2495
|
+
}
|
|
2496
|
+
return { ok: true, answer: { kind: "choice", choice: a.choice, confidence: a.confidence, probabilities } };
|
|
2497
|
+
}
|
|
2498
|
+
if (question.kind === "binary") {
|
|
2499
|
+
if (!isUnit(a.probability)) return { ok: false, reason: "invalid_probability" };
|
|
2500
|
+
return { ok: true, answer: { kind: "binary", probability: a.probability } };
|
|
2501
|
+
}
|
|
2502
|
+
const maxScore = question.levels.length - 1;
|
|
2503
|
+
if (typeof a.score !== "number" || !Number.isFinite(a.score) || a.score < 0 || a.score > maxScore) {
|
|
2504
|
+
return { ok: false, reason: "score_out_of_range" };
|
|
2505
|
+
}
|
|
2506
|
+
if (!isUnit(a.confidence)) return { ok: false, reason: "invalid_probability" };
|
|
2507
|
+
if (!Array.isArray(a.probabilities) || !a.probabilities.every(isUnit)) return { ok: false, reason: "invalid_probability" };
|
|
2508
|
+
return { ok: true, answer: { kind: "score", score: a.score, confidence: a.confidence, probabilities: a.probabilities.slice() } };
|
|
2509
|
+
} catch {
|
|
2510
|
+
return { ok: false, reason: "malformed" };
|
|
2511
|
+
}
|
|
2512
|
+
}
|
|
2513
|
+
function classifierStateForFact(fact) {
|
|
2514
|
+
const parts = [fact.title, fact.body];
|
|
2515
|
+
if (Array.isArray(fact.tags) && fact.tags.length > 0) parts.push(`Tags: ${fact.tags.join(", ")}`);
|
|
2516
|
+
return parts.join("\n\n");
|
|
2517
|
+
}
|
|
2518
|
+
|
|
2474
2519
|
// src/repositories/BaseRepository.ts
|
|
2475
2520
|
var BaseRepository = class {
|
|
2476
2521
|
constructor(db, prefix) {
|
|
@@ -3539,7 +3584,7 @@ var MaintenanceService = class {
|
|
|
3539
3584
|
if (!ontologyService) {
|
|
3540
3585
|
return { ...zeroed, remaining: 0, deferred: 0 };
|
|
3541
3586
|
}
|
|
3542
|
-
const { mode } = await ontologyService.getEffectiveState(entityId);
|
|
3587
|
+
const { mode, manifest: effectiveManifest } = await ontologyService.getEffectiveState(entityId);
|
|
3543
3588
|
if (mode === "off") {
|
|
3544
3589
|
const counts2 = await this.entryRepo.countUntypedByEntityId(entityId, recheckCutoff);
|
|
3545
3590
|
return { ...zeroed, remaining: 0, deferred: counts2.deferred };
|
|
@@ -3549,6 +3594,14 @@ var MaintenanceService = class {
|
|
|
3549
3594
|
const counts2 = await this.entryRepo.countUntypedByEntityId(entityId, recheckCutoff);
|
|
3550
3595
|
return { ...zeroed, remaining: counts2.eligible, deferred: counts2.deferred };
|
|
3551
3596
|
}
|
|
3597
|
+
const classifierMode = options?.classifier ?? this.options.config?.ontology?.backfillClassifier ?? "llm";
|
|
3598
|
+
const classify = this.options.llmProvider.classify;
|
|
3599
|
+
if (classifierMode === "auto" && typeof classify === "function") {
|
|
3600
|
+
const slugs = effectiveManifest.node_types.map((n) => n.type);
|
|
3601
|
+
if (slugs.length > 0 && slugs.length <= 255) {
|
|
3602
|
+
return this._runClassifierBackfill(entityId, candidates, effectiveManifest, now, recheckCutoff);
|
|
3603
|
+
}
|
|
3604
|
+
}
|
|
3552
3605
|
const ontologyContext = await ontologyService.buildPromptContext(entityId);
|
|
3553
3606
|
const toPromptShape = (f) => ({ id: f.id, title: f.title, body: f.body, tags: f.tags });
|
|
3554
3607
|
const buildPrompt = (facts) => this.promptService.buildOntologyBackfillPrompt(
|
|
@@ -3627,6 +3680,87 @@ var MaintenanceService = class {
|
|
|
3627
3680
|
deferred: counts.deferred
|
|
3628
3681
|
};
|
|
3629
3682
|
}
|
|
3683
|
+
/**
|
|
3684
|
+
* Classifier-mode backfill (spec §7.3): one `choice` question per untyped
|
|
3685
|
+
* fact over the effective manifest's node-type slugs. Accepted answers go
|
|
3686
|
+
* through `_applyOntologyBackfillBatch` exactly like LLM classifications.
|
|
3687
|
+
* No edges are proposed.
|
|
3688
|
+
*/
|
|
3689
|
+
async _runClassifierBackfill(entityId, candidates, manifest, now, recheckCutoff) {
|
|
3690
|
+
const provider = this.options.llmProvider;
|
|
3691
|
+
const rawMin = this.options.config?.ontology?.classifyMinConfidence;
|
|
3692
|
+
const minConfidence = typeof rawMin === "number" && Number.isFinite(rawMin) && rawMin >= 0 && rawMin <= 1 ? rawMin : 0.5;
|
|
3693
|
+
const rawConcurrency = this.options.config?.chunkConcurrency ?? 1;
|
|
3694
|
+
const concurrency = Number.isFinite(rawConcurrency) && rawConcurrency >= 1 ? Math.floor(rawConcurrency) : 1;
|
|
3695
|
+
const question = {
|
|
3696
|
+
kind: "choice",
|
|
3697
|
+
options: manifest.node_types.map((n) => n.type),
|
|
3698
|
+
instructions: "Choose the ontology node type that best describes this fact.\n" + manifest.node_types.map((n) => `- ${n.type}: ${n.description}`).join("\n")
|
|
3699
|
+
};
|
|
3700
|
+
const outcomes = await withConcurrency(
|
|
3701
|
+
candidates.map((fact) => async () => {
|
|
3702
|
+
let response;
|
|
3703
|
+
try {
|
|
3704
|
+
response = await provider.classify({ state: classifierStateForFact(fact), questions: { okf_type: question } });
|
|
3705
|
+
} catch {
|
|
3706
|
+
return { fact, kind: "threw" };
|
|
3707
|
+
}
|
|
3708
|
+
const checked = validateClassifierAnswer(response, "okf_type", question);
|
|
3709
|
+
if (!checked.ok) return { fact, kind: "invalid", reason: checked.reason };
|
|
3710
|
+
if (checked.answer.kind !== "choice") return { fact, kind: "invalid", reason: "kind_mismatch" };
|
|
3711
|
+
if (checked.answer.confidence < minConfidence) return { fact, kind: "low_confidence" };
|
|
3712
|
+
return { fact, kind: "accepted", okfType: checked.answer.choice };
|
|
3713
|
+
}),
|
|
3714
|
+
concurrency
|
|
3715
|
+
);
|
|
3716
|
+
const attempted = outcomes.filter((o) => o.kind !== "threw").map((o) => o.fact);
|
|
3717
|
+
const skipped = outcomes.length - attempted.length;
|
|
3718
|
+
const classifications = outcomes.filter((o) => o.kind === "accepted").map((o) => ({ id: o.fact.id, okf_type: o.okfType }));
|
|
3719
|
+
const invalidCount = outcomes.filter((o) => o.kind === "invalid").length;
|
|
3720
|
+
let typed = 0;
|
|
3721
|
+
let failedValidation = 0;
|
|
3722
|
+
let edgesAdded = 0;
|
|
3723
|
+
let aborted = false;
|
|
3724
|
+
if (attempted.length > 0) {
|
|
3725
|
+
const applied = await this._applyOntologyBackfillBatch(
|
|
3726
|
+
entityId,
|
|
3727
|
+
{ batch: attempted, classifications, ontologyUpdates: void 0 },
|
|
3728
|
+
now
|
|
3729
|
+
);
|
|
3730
|
+
aborted = applied.abortedOntologyOff;
|
|
3731
|
+
if (!aborted) {
|
|
3732
|
+
typed = applied.typed;
|
|
3733
|
+
failedValidation = invalidCount + applied.failedValidation;
|
|
3734
|
+
edgesAdded = applied.edgesAdded;
|
|
3735
|
+
}
|
|
3736
|
+
}
|
|
3737
|
+
const diagBuffer = new DiagnosticBuffer();
|
|
3738
|
+
const diagBase = { entityId, operation: "ontologyBackfill", trigger: "call" };
|
|
3739
|
+
for (const o of outcomes) {
|
|
3740
|
+
if (o.kind === "low_confidence") {
|
|
3741
|
+
diagBuffer.push({ ...diagBase, code: "classification_low_confidence", detail: { factId: o.fact.id, reason: "below_threshold" } });
|
|
3742
|
+
} else if (o.kind === "invalid") {
|
|
3743
|
+
diagBuffer.push({ ...diagBase, code: "classification_invalid", detail: { factId: o.fact.id, reason: o.reason } });
|
|
3744
|
+
} else if (o.kind === "threw") {
|
|
3745
|
+
diagBuffer.push({ ...diagBase, code: "classification_invalid", detail: { factId: o.fact.id, reason: "classify_threw" } });
|
|
3746
|
+
}
|
|
3747
|
+
}
|
|
3748
|
+
diagBuffer.flush(this.options);
|
|
3749
|
+
const counts = await this.entryRepo.countUntypedByEntityId(entityId, recheckCutoff);
|
|
3750
|
+
if (aborted) {
|
|
3751
|
+
return { scanned: 0, typed: 0, failedValidation: 0, edgesAdded: 0, skipped, remaining: 0, deferred: counts.deferred };
|
|
3752
|
+
}
|
|
3753
|
+
this.searchService.evictCache(entityId);
|
|
3754
|
+
return {
|
|
3755
|
+
scanned: candidates.length,
|
|
3756
|
+
typed,
|
|
3757
|
+
failedValidation,
|
|
3758
|
+
edgesAdded,
|
|
3759
|
+
skipped,
|
|
3760
|
+
remaining: counts.eligible,
|
|
3761
|
+
deferred: counts.deferred
|
|
3762
|
+
};
|
|
3763
|
+
}
|
|
3630
3764
|
/**
|
|
3631
3765
|
* Applies one parsed backfill batch in its own transaction. Per-batch rather
|
|
3632
3766
|
* than one transaction for the pass, so mergeEmergentUpdates semantics and
|
|
@@ -4330,8 +4464,7 @@ var EmbeddingService = class {
|
|
|
4330
4464
|
}
|
|
4331
4465
|
}
|
|
4332
4466
|
async tryEmbedFact(fact, ctx) {
|
|
4333
|
-
|
|
4334
|
-
if (typeof embedFn !== "function") return { ok: false, kind: "no_provider" };
|
|
4467
|
+
if (typeof this.options.llmProvider.embed !== "function") return { ok: false, kind: "no_provider" };
|
|
4335
4468
|
let tagsStr;
|
|
4336
4469
|
if (Array.isArray(fact.tags)) {
|
|
4337
4470
|
tagsStr = fact.tags.join(" ");
|
|
@@ -4348,7 +4481,7 @@ var EmbeddingService = class {
|
|
|
4348
4481
|
const text = clip(`${fact.title} ${fact.body} ${tagsStr}`.trim(), maxEmbedChars);
|
|
4349
4482
|
let float32Vector;
|
|
4350
4483
|
try {
|
|
4351
|
-
const vector = await
|
|
4484
|
+
const vector = await this.options.llmProvider.embed(text);
|
|
4352
4485
|
if (vector.length === 0 || !vector.every((v) => typeof v === "number" && isFinite(v))) {
|
|
4353
4486
|
console.warn(`[WikiMemory] embedFact: embed() returned an invalid vector for ${fact.id}; skipping.`);
|
|
4354
4487
|
this.reportEmbed(ctx, fact, "embedding_failed", "invalid_vector");
|
|
@@ -4607,7 +4740,6 @@ var RetrievalService = class {
|
|
|
4607
4740
|
const hybridWeight = options?.hybridWeight ?? config?.hybridWeight;
|
|
4608
4741
|
const weight = hybridWeight !== void 0 && !Number.isNaN(hybridWeight) ? Math.max(0, Math.min(1, hybridWeight)) : void 0;
|
|
4609
4742
|
const skipEmbed = weight === 0;
|
|
4610
|
-
const embedFn = this.options.llmProvider.embed;
|
|
4611
4743
|
let facts = [];
|
|
4612
4744
|
let scoreByFactId;
|
|
4613
4745
|
if (maxResults === 0) ; else if (trimmedQuery) {
|
|
@@ -4618,11 +4750,11 @@ var RetrievalService = class {
|
|
|
4618
4750
|
const padLimit = (n) => n >= Number.MAX_SAFE_INTEGER ? n : n + draftPad;
|
|
4619
4751
|
if (scoredEntityIds.length === 0) {
|
|
4620
4752
|
usedEmbed = true;
|
|
4621
|
-
} else if (!skipEmbed &&
|
|
4753
|
+
} else if (!skipEmbed && typeof this.options.llmProvider.embed === "function") {
|
|
4622
4754
|
let rankerShouldRethrow = false;
|
|
4623
4755
|
let pendingRankerFallbackError;
|
|
4624
4756
|
try {
|
|
4625
|
-
const queryVec = await
|
|
4757
|
+
const queryVec = await this.options.llmProvider.embed(trimmedQuery);
|
|
4626
4758
|
if (queryVec.length === 0 || !queryVec.every((v) => typeof v === "number" && isFinite(v))) {
|
|
4627
4759
|
throw new Error(
|
|
4628
4760
|
"embed() returned an empty or non-finite vector. Falling back to keyword search."
|
|
@@ -5292,5 +5424,5 @@ var WriteService = class {
|
|
|
5292
5424
|
};
|
|
5293
5425
|
|
|
5294
5426
|
export { BaseRepository, DEFAULT_CHUNK_OVERLAP, DEFAULT_MAX_CHUNK_LENGTH, DEFAULT_MAX_EMBED_CHARS, DiagnosticBuffer, EMBED_CHARS_CEILING, EmbeddingService, HEAL_ANCHORS_PER_CANDIDATE, HEAL_BATCH_SIZE, HEAL_MAX_FACT_BODY_CHARS_L3, HEAL_MAX_TASKS, HEAL_RECHECK_MS, HOOK_TIMEOUT_MARKER, ImportExportService, IngestionService, JobManager, MaintenanceService, MetadataRepository, ONTOLOGY_BACKFILL_BATCH_SIZE, ONTOLOGY_BACKFILL_MAX_PROMPT_CHARS, ONTOLOGY_BACKFILL_RECHECK_MS, ONTOLOGY_BACKFILL_SYSTEM_PROMPT, PromptService, PrunePartialFailureError, RetrievalService, SearchService, WikiBusyError, WikiDraftNotFound, WikiDuplicateHashError, WikiGraphNodeOwnershipConflict, WikiIngestEmptyError, WikiInvalidReadOptions, WikiParseError, WikiSourceRefHashCollision, WikiStrictOntologyViolation, WikiTransactionError, WriteService, __privateAdd, __privateGet, __privateSet, chunkText, configureRandomSource, emptyManifest, entitySummaryMetaKey, extractSqliteCode, generateId, normalizeSourceHash, normalizeSourceRef, normalizeTitleKey, parseEmbedding, resolveEdgeDefinitions, resolveNodeType, safeSlice, typeSatisfies, validateInlineEdges, validateManifest };
|
|
5295
|
-
//# sourceMappingURL=chunk-
|
|
5296
|
-
//# sourceMappingURL=chunk-
|
|
5427
|
+
//# sourceMappingURL=chunk-3P7FAKJA.mjs.map
|
|
5428
|
+
//# sourceMappingURL=chunk-3P7FAKJA.mjs.map
|