@equationalapplications/core-llm-wiki 7.4.0 → 7.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -799,6 +799,47 @@ const result = await wiki.runOntologyBackfill(entityId);
799
799
  `config.prompts.ontologyBackfillSystemPrompt` (template may use `{{facts}}`
800
800
  and the ontology placeholders).
801
801
 
802
+ #### Classifier mode (optional)
803
+
804
+ A System-One classifier (for example TypeSafe's Jev, an OpenJev-style open model, or a local ONNX classifier) can type facts during backfill without generating text. Add `classify` to your provider and opt in:
805
+
806
+ ```ts
807
+ const wiki = createWiki(db, {
808
+ llmProvider: { generateText, classify },
809
+ config: { ontology: { backfillClassifier: 'auto', classifyMinConfidence: 0.5 } },
810
+ });
811
+ await wiki.runOntologyBackfill('user-1'); // uses classify
812
+ await wiki.runOntologyBackfill('user-1', { classifier: 'llm' }); // force the generative path
813
+ ```
814
+
815
+ - Providing `classify` changes nothing by itself. The default is `'llm'`.
816
+ - Each untyped fact gets one `choice` question over the entity manifest's node types. Answers below `classifyMinConfidence` (default 0.5) are left untyped and retried after the cooldown.
817
+ - **No edges are proposed in classifier mode** (`edgesAdded: 0`): a classifier cannot extract edge targets. Run with `classifier: 'llm'` when you want edges.
818
+ - Manifests with more than 255 node types, or providers without `classify`, use the generative path.
819
+ - Answers are validated as untrusted. Off-list choices and out-of-range probabilities count toward `failedValidation`. A thrown `classify` counts toward `skipped` and is retried on the next pass.
820
+
821
+ Example adapter for Cloudflare Workers AI's `typesafe/jev`. This is illustrative, not a supported package; check the provider's current API reference before use.
822
+
823
+ ```ts
824
+ const classify: LLMProvider['classify'] = async ({ state, questions }) => {
825
+ const jevQuestions = Object.fromEntries(Object.entries(questions).map(([key, q]) => [key,
826
+ q.kind === 'choice' ? { type: 'choice', instructions: q.instructions, criteria: Object.fromEntries(q.options.map((o) => [o, o])) }
827
+ : q.kind === 'score' ? { type: 'score', instructions: q.instructions, criteria: q.levels }
828
+ : { type: 'noul', instructions: q.instructions },
829
+ ]));
830
+ const res = await env.AI.run('typesafe/jev', { state, questions: jevQuestions });
831
+ const answers = Object.fromEntries(Object.entries(res.answers).map(([key, a]: [string, any]) => [key,
832
+ a.type === 'choice' ? { kind: 'choice', choice: a.choice, confidence: a.confidence, probabilities: a.probabilities }
833
+ : a.type === 'score' ? {
834
+ kind: 'score', score: a.score, confidence: a.confidence,
835
+ probabilities: Object.keys(a.probabilities).sort((x, y) => Number(x) - Number(y)).map((k) => a.probabilities[k]),
836
+ }
837
+ : { kind: 'binary', probability: a.noul },
838
+ ]));
839
+ return { answers };
840
+ };
841
+ ```
842
+
802
843
  ## OKF Import/Export
803
844
 
804
845
  The core package integrates with `@equationalapplications/core-okf` to seamlessly adapt wiki data dumps to and from Open Knowledge Format (OKF) bundles (v0.1 and v0.2; `formatOkfBundle` defaults to the v0.2 / `llm-wiki/2` profile).
@@ -2471,6 +2471,51 @@ var IngestionService = class {
2471
2471
  }
2472
2472
  };
2473
2473
 
2474
+ // src/utils/classifier.ts
2475
+ var isUnit = (v) => typeof v === "number" && Number.isFinite(v) && v >= 0 && v <= 1;
2476
+ function validateClassifierAnswer(response, key, question) {
2477
+ try {
2478
+ if (response === null || typeof response !== "object") return { ok: false, reason: "malformed" };
2479
+ const answers = response.answers;
2480
+ if (answers === null || typeof answers !== "object") return { ok: false, reason: "missing_answer" };
2481
+ if (!Object.prototype.hasOwnProperty.call(answers, key)) return { ok: false, reason: "missing_answer" };
2482
+ const raw = answers[key];
2483
+ if (raw === null || typeof raw !== "object") return { ok: false, reason: "missing_answer" };
2484
+ const a = raw;
2485
+ if (a.kind !== question.kind) return { ok: false, reason: "kind_mismatch" };
2486
+ if (question.kind === "choice") {
2487
+ if (typeof a.choice !== "string" || !question.options.includes(a.choice)) return { ok: false, reason: "choice_not_offered" };
2488
+ if (!isUnit(a.confidence)) return { ok: false, reason: "invalid_probability" };
2489
+ const probs = a.probabilities;
2490
+ if (probs === null || typeof probs !== "object" || Array.isArray(probs)) return { ok: false, reason: "invalid_probability" };
2491
+ const probabilities = {};
2492
+ for (const [k, v] of Object.entries(probs)) {
2493
+ if (!isUnit(v)) return { ok: false, reason: "invalid_probability" };
2494
+ probabilities[k] = v;
2495
+ }
2496
+ return { ok: true, answer: { kind: "choice", choice: a.choice, confidence: a.confidence, probabilities } };
2497
+ }
2498
+ if (question.kind === "binary") {
2499
+ if (!isUnit(a.probability)) return { ok: false, reason: "invalid_probability" };
2500
+ return { ok: true, answer: { kind: "binary", probability: a.probability } };
2501
+ }
2502
+ const maxScore = question.levels.length - 1;
2503
+ if (typeof a.score !== "number" || !Number.isFinite(a.score) || a.score < 0 || a.score > maxScore) {
2504
+ return { ok: false, reason: "score_out_of_range" };
2505
+ }
2506
+ if (!isUnit(a.confidence)) return { ok: false, reason: "invalid_probability" };
2507
+ if (!Array.isArray(a.probabilities) || !a.probabilities.every(isUnit)) return { ok: false, reason: "invalid_probability" };
2508
+ return { ok: true, answer: { kind: "score", score: a.score, confidence: a.confidence, probabilities: a.probabilities.slice() } };
2509
+ } catch {
2510
+ return { ok: false, reason: "malformed" };
2511
+ }
2512
+ }
2513
+ function classifierStateForFact(fact) {
2514
+ const parts = [fact.title, fact.body];
2515
+ if (Array.isArray(fact.tags) && fact.tags.length > 0) parts.push(`Tags: ${fact.tags.join(", ")}`);
2516
+ return parts.join("\n\n");
2517
+ }
2518
+
2474
2519
  // src/repositories/BaseRepository.ts
2475
2520
  var BaseRepository = class {
2476
2521
  constructor(db, prefix) {
@@ -3539,7 +3584,7 @@ var MaintenanceService = class {
3539
3584
  if (!ontologyService) {
3540
3585
  return { ...zeroed, remaining: 0, deferred: 0 };
3541
3586
  }
3542
- const { mode } = await ontologyService.getEffectiveState(entityId);
3587
+ const { mode, manifest: effectiveManifest } = await ontologyService.getEffectiveState(entityId);
3543
3588
  if (mode === "off") {
3544
3589
  const counts2 = await this.entryRepo.countUntypedByEntityId(entityId, recheckCutoff);
3545
3590
  return { ...zeroed, remaining: 0, deferred: counts2.deferred };
@@ -3549,6 +3594,14 @@ var MaintenanceService = class {
3549
3594
  const counts2 = await this.entryRepo.countUntypedByEntityId(entityId, recheckCutoff);
3550
3595
  return { ...zeroed, remaining: counts2.eligible, deferred: counts2.deferred };
3551
3596
  }
3597
+ const classifierMode = options?.classifier ?? this.options.config?.ontology?.backfillClassifier ?? "llm";
3598
+ const classify = this.options.llmProvider.classify;
3599
+ if (classifierMode === "auto" && typeof classify === "function") {
3600
+ const slugs = effectiveManifest.node_types.map((n) => n.type);
3601
+ if (slugs.length > 0 && slugs.length <= 255) {
3602
+ return this._runClassifierBackfill(entityId, candidates, effectiveManifest, now, recheckCutoff);
3603
+ }
3604
+ }
3552
3605
  const ontologyContext = await ontologyService.buildPromptContext(entityId);
3553
3606
  const toPromptShape = (f) => ({ id: f.id, title: f.title, body: f.body, tags: f.tags });
3554
3607
  const buildPrompt = (facts) => this.promptService.buildOntologyBackfillPrompt(
@@ -3627,6 +3680,87 @@ var MaintenanceService = class {
3627
3680
  deferred: counts.deferred
3628
3681
  };
3629
3682
  }
3683
+ /**
3684
+ * Classifier-mode backfill (spec §7.3): one `choice` question per untyped
3685
+ * fact over the effective manifest's node-type slugs. Accepted answers go
3686
+ * through `_applyOntologyBackfillBatch` exactly like LLM classifications.
3687
+ * No edges are proposed.
3688
+ */
3689
+ async _runClassifierBackfill(entityId, candidates, manifest, now, recheckCutoff) {
3690
+ const provider = this.options.llmProvider;
3691
+ const rawMin = this.options.config?.ontology?.classifyMinConfidence;
3692
+ const minConfidence = typeof rawMin === "number" && Number.isFinite(rawMin) && rawMin >= 0 && rawMin <= 1 ? rawMin : 0.5;
3693
+ const rawConcurrency = this.options.config?.chunkConcurrency ?? 1;
3694
+ const concurrency = Number.isFinite(rawConcurrency) && rawConcurrency >= 1 ? Math.floor(rawConcurrency) : 1;
3695
+ const question = {
3696
+ kind: "choice",
3697
+ options: manifest.node_types.map((n) => n.type),
3698
+ instructions: "Choose the ontology node type that best describes this fact.\n" + manifest.node_types.map((n) => `- ${n.type}: ${n.description}`).join("\n")
3699
+ };
3700
+ const outcomes = await withConcurrency(
3701
+ candidates.map((fact) => async () => {
3702
+ let response;
3703
+ try {
3704
+ response = await provider.classify({ state: classifierStateForFact(fact), questions: { okf_type: question } });
3705
+ } catch {
3706
+ return { fact, kind: "threw" };
3707
+ }
3708
+ const checked = validateClassifierAnswer(response, "okf_type", question);
3709
+ if (!checked.ok) return { fact, kind: "invalid", reason: checked.reason };
3710
+ if (checked.answer.kind !== "choice") return { fact, kind: "invalid", reason: "kind_mismatch" };
3711
+ if (checked.answer.confidence < minConfidence) return { fact, kind: "low_confidence" };
3712
+ return { fact, kind: "accepted", okfType: checked.answer.choice };
3713
+ }),
3714
+ concurrency
3715
+ );
3716
+ const attempted = outcomes.filter((o) => o.kind !== "threw").map((o) => o.fact);
3717
+ const skipped = outcomes.length - attempted.length;
3718
+ const classifications = outcomes.filter((o) => o.kind === "accepted").map((o) => ({ id: o.fact.id, okf_type: o.okfType }));
3719
+ const invalidCount = outcomes.filter((o) => o.kind === "invalid").length;
3720
+ let typed = 0;
3721
+ let failedValidation = 0;
3722
+ let edgesAdded = 0;
3723
+ let aborted = false;
3724
+ if (attempted.length > 0) {
3725
+ const applied = await this._applyOntologyBackfillBatch(
3726
+ entityId,
3727
+ { batch: attempted, classifications, ontologyUpdates: void 0 },
3728
+ now
3729
+ );
3730
+ aborted = applied.abortedOntologyOff;
3731
+ if (!aborted) {
3732
+ typed = applied.typed;
3733
+ failedValidation = invalidCount + applied.failedValidation;
3734
+ edgesAdded = applied.edgesAdded;
3735
+ }
3736
+ }
3737
+ const diagBuffer = new DiagnosticBuffer();
3738
+ const diagBase = { entityId, operation: "ontologyBackfill", trigger: "call" };
3739
+ for (const o of outcomes) {
3740
+ if (o.kind === "low_confidence") {
3741
+ diagBuffer.push({ ...diagBase, code: "classification_low_confidence", detail: { factId: o.fact.id, reason: "below_threshold" } });
3742
+ } else if (o.kind === "invalid") {
3743
+ diagBuffer.push({ ...diagBase, code: "classification_invalid", detail: { factId: o.fact.id, reason: o.reason } });
3744
+ } else if (o.kind === "threw") {
3745
+ diagBuffer.push({ ...diagBase, code: "classification_invalid", detail: { factId: o.fact.id, reason: "classify_threw" } });
3746
+ }
3747
+ }
3748
+ diagBuffer.flush(this.options);
3749
+ const counts = await this.entryRepo.countUntypedByEntityId(entityId, recheckCutoff);
3750
+ if (aborted) {
3751
+ return { scanned: 0, typed: 0, failedValidation: 0, edgesAdded: 0, skipped, remaining: 0, deferred: counts.deferred };
3752
+ }
3753
+ this.searchService.evictCache(entityId);
3754
+ return {
3755
+ scanned: candidates.length,
3756
+ typed,
3757
+ failedValidation,
3758
+ edgesAdded,
3759
+ skipped,
3760
+ remaining: counts.eligible,
3761
+ deferred: counts.deferred
3762
+ };
3763
+ }
3630
3764
  /**
3631
3765
  * Applies one parsed backfill batch in its own transaction. Per-batch rather
3632
3766
  * than one transaction for the pass, so mergeEmergentUpdates semantics and
@@ -4330,8 +4464,7 @@ var EmbeddingService = class {
4330
4464
  }
4331
4465
  }
4332
4466
  async tryEmbedFact(fact, ctx) {
4333
- const embedFn = this.options.llmProvider.embed;
4334
- if (typeof embedFn !== "function") return { ok: false, kind: "no_provider" };
4467
+ if (typeof this.options.llmProvider.embed !== "function") return { ok: false, kind: "no_provider" };
4335
4468
  let tagsStr;
4336
4469
  if (Array.isArray(fact.tags)) {
4337
4470
  tagsStr = fact.tags.join(" ");
@@ -4348,7 +4481,7 @@ var EmbeddingService = class {
4348
4481
  const text = clip(`${fact.title} ${fact.body} ${tagsStr}`.trim(), maxEmbedChars);
4349
4482
  let float32Vector;
4350
4483
  try {
4351
- const vector = await embedFn(text);
4484
+ const vector = await this.options.llmProvider.embed(text);
4352
4485
  if (vector.length === 0 || !vector.every((v) => typeof v === "number" && isFinite(v))) {
4353
4486
  console.warn(`[WikiMemory] embedFact: embed() returned an invalid vector for ${fact.id}; skipping.`);
4354
4487
  this.reportEmbed(ctx, fact, "embedding_failed", "invalid_vector");
@@ -4607,7 +4740,6 @@ var RetrievalService = class {
4607
4740
  const hybridWeight = options?.hybridWeight ?? config?.hybridWeight;
4608
4741
  const weight = hybridWeight !== void 0 && !Number.isNaN(hybridWeight) ? Math.max(0, Math.min(1, hybridWeight)) : void 0;
4609
4742
  const skipEmbed = weight === 0;
4610
- const embedFn = this.options.llmProvider.embed;
4611
4743
  let facts = [];
4612
4744
  let scoreByFactId;
4613
4745
  if (maxResults === 0) ; else if (trimmedQuery) {
@@ -4618,11 +4750,11 @@ var RetrievalService = class {
4618
4750
  const padLimit = (n) => n >= Number.MAX_SAFE_INTEGER ? n : n + draftPad;
4619
4751
  if (scoredEntityIds.length === 0) {
4620
4752
  usedEmbed = true;
4621
- } else if (!skipEmbed && embedFn) {
4753
+ } else if (!skipEmbed && typeof this.options.llmProvider.embed === "function") {
4622
4754
  let rankerShouldRethrow = false;
4623
4755
  let pendingRankerFallbackError;
4624
4756
  try {
4625
- const queryVec = await embedFn(trimmedQuery);
4757
+ const queryVec = await this.options.llmProvider.embed(trimmedQuery);
4626
4758
  if (queryVec.length === 0 || !queryVec.every((v) => typeof v === "number" && isFinite(v))) {
4627
4759
  throw new Error(
4628
4760
  "embed() returned an empty or non-finite vector. Falling back to keyword search."
@@ -5292,5 +5424,5 @@ var WriteService = class {
5292
5424
  };
5293
5425
 
5294
5426
  export { BaseRepository, DEFAULT_CHUNK_OVERLAP, DEFAULT_MAX_CHUNK_LENGTH, DEFAULT_MAX_EMBED_CHARS, DiagnosticBuffer, EMBED_CHARS_CEILING, EmbeddingService, HEAL_ANCHORS_PER_CANDIDATE, HEAL_BATCH_SIZE, HEAL_MAX_FACT_BODY_CHARS_L3, HEAL_MAX_TASKS, HEAL_RECHECK_MS, HOOK_TIMEOUT_MARKER, ImportExportService, IngestionService, JobManager, MaintenanceService, MetadataRepository, ONTOLOGY_BACKFILL_BATCH_SIZE, ONTOLOGY_BACKFILL_MAX_PROMPT_CHARS, ONTOLOGY_BACKFILL_RECHECK_MS, ONTOLOGY_BACKFILL_SYSTEM_PROMPT, PromptService, PrunePartialFailureError, RetrievalService, SearchService, WikiBusyError, WikiDraftNotFound, WikiDuplicateHashError, WikiGraphNodeOwnershipConflict, WikiIngestEmptyError, WikiInvalidReadOptions, WikiParseError, WikiSourceRefHashCollision, WikiStrictOntologyViolation, WikiTransactionError, WriteService, __privateAdd, __privateGet, __privateSet, chunkText, configureRandomSource, emptyManifest, entitySummaryMetaKey, extractSqliteCode, generateId, normalizeSourceHash, normalizeSourceRef, normalizeTitleKey, parseEmbedding, resolveEdgeDefinitions, resolveNodeType, safeSlice, typeSatisfies, validateInlineEdges, validateManifest };
5295
- //# sourceMappingURL=chunk-G5OR2VWD.mjs.map
5296
- //# sourceMappingURL=chunk-G5OR2VWD.mjs.map
5427
+ //# sourceMappingURL=chunk-3P7FAKJA.mjs.map
5428
+ //# sourceMappingURL=chunk-3P7FAKJA.mjs.map