@equationalapplications/core-llm-wiki 7.3.0 → 7.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/testing.js CHANGED
@@ -817,8 +817,7 @@ var EmbeddingService = class {
817
817
  }
818
818
  }
819
819
  async tryEmbedFact(fact, ctx) {
820
- const embedFn = this.options.llmProvider.embed;
821
- if (typeof embedFn !== "function") return { ok: false, kind: "no_provider" };
820
+ if (typeof this.options.llmProvider.embed !== "function") return { ok: false, kind: "no_provider" };
822
821
  let tagsStr;
823
822
  if (Array.isArray(fact.tags)) {
824
823
  tagsStr = fact.tags.join(" ");
@@ -835,7 +834,7 @@ var EmbeddingService = class {
835
834
  const text = clip(`${fact.title} ${fact.body} ${tagsStr}`.trim(), maxEmbedChars);
836
835
  let float32Vector;
837
836
  try {
838
- const vector = await embedFn(text);
837
+ const vector = await this.options.llmProvider.embed(text);
839
838
  if (vector.length === 0 || !vector.every((v) => typeof v === "number" && isFinite(v))) {
840
839
  console.warn(`[WikiMemory] embedFact: embed() returned an invalid vector for ${fact.id}; skipping.`);
841
840
  this.reportEmbed(ctx, fact, "embedding_failed", "invalid_vector");
@@ -2281,6 +2280,51 @@ var IngestionService = class {
2281
2280
  }
2282
2281
  };
2283
2282
 
2283
+ // src/utils/classifier.ts
2284
+ var isUnit = (v) => typeof v === "number" && Number.isFinite(v) && v >= 0 && v <= 1;
2285
+ function validateClassifierAnswer(response, key, question) {
2286
+ try {
2287
+ if (response === null || typeof response !== "object") return { ok: false, reason: "malformed" };
2288
+ const answers = response.answers;
2289
+ if (answers === null || typeof answers !== "object") return { ok: false, reason: "missing_answer" };
2290
+ if (!Object.prototype.hasOwnProperty.call(answers, key)) return { ok: false, reason: "missing_answer" };
2291
+ const raw = answers[key];
2292
+ if (raw === null || typeof raw !== "object") return { ok: false, reason: "missing_answer" };
2293
+ const a = raw;
2294
+ if (a.kind !== question.kind) return { ok: false, reason: "kind_mismatch" };
2295
+ if (question.kind === "choice") {
2296
+ if (typeof a.choice !== "string" || !question.options.includes(a.choice)) return { ok: false, reason: "choice_not_offered" };
2297
+ if (!isUnit(a.confidence)) return { ok: false, reason: "invalid_probability" };
2298
+ const probs = a.probabilities;
2299
+ if (probs === null || typeof probs !== "object" || Array.isArray(probs)) return { ok: false, reason: "invalid_probability" };
2300
+ const probabilities = {};
2301
+ for (const [k, v] of Object.entries(probs)) {
2302
+ if (!isUnit(v)) return { ok: false, reason: "invalid_probability" };
2303
+ probabilities[k] = v;
2304
+ }
2305
+ return { ok: true, answer: { kind: "choice", choice: a.choice, confidence: a.confidence, probabilities } };
2306
+ }
2307
+ if (question.kind === "binary") {
2308
+ if (!isUnit(a.probability)) return { ok: false, reason: "invalid_probability" };
2309
+ return { ok: true, answer: { kind: "binary", probability: a.probability } };
2310
+ }
2311
+ const maxScore = question.levels.length - 1;
2312
+ if (typeof a.score !== "number" || !Number.isFinite(a.score) || a.score < 0 || a.score > maxScore) {
2313
+ return { ok: false, reason: "score_out_of_range" };
2314
+ }
2315
+ if (!isUnit(a.confidence)) return { ok: false, reason: "invalid_probability" };
2316
+ if (!Array.isArray(a.probabilities) || !a.probabilities.every(isUnit)) return { ok: false, reason: "invalid_probability" };
2317
+ return { ok: true, answer: { kind: "score", score: a.score, confidence: a.confidence, probabilities: a.probabilities.slice() } };
2318
+ } catch {
2319
+ return { ok: false, reason: "malformed" };
2320
+ }
2321
+ }
2322
+ function classifierStateForFact(fact) {
2323
+ const parts = [fact.title, fact.body];
2324
+ if (Array.isArray(fact.tags) && fact.tags.length > 0) parts.push(`Tags: ${fact.tags.join(", ")}`);
2325
+ return parts.join("\n\n");
2326
+ }
2327
+
2284
2328
  // src/utils/embedding.ts
2285
2329
  function parseEmbedding(blob, text) {
2286
2330
  if (blob && blob.byteLength > 0) {
@@ -3191,7 +3235,7 @@ var MaintenanceService = class {
3191
3235
  if (!ontologyService) {
3192
3236
  return { ...zeroed, remaining: 0, deferred: 0 };
3193
3237
  }
3194
- const { mode } = await ontologyService.getEffectiveState(entityId);
3238
+ const { mode, manifest: effectiveManifest } = await ontologyService.getEffectiveState(entityId);
3195
3239
  if (mode === "off") {
3196
3240
  const counts2 = await this.entryRepo.countUntypedByEntityId(entityId, recheckCutoff);
3197
3241
  return { ...zeroed, remaining: 0, deferred: counts2.deferred };
@@ -3201,6 +3245,14 @@ var MaintenanceService = class {
3201
3245
  const counts2 = await this.entryRepo.countUntypedByEntityId(entityId, recheckCutoff);
3202
3246
  return { ...zeroed, remaining: counts2.eligible, deferred: counts2.deferred };
3203
3247
  }
3248
+ const classifierMode = options?.classifier ?? this.options.config?.ontology?.backfillClassifier ?? "llm";
3249
+ const classify = this.options.llmProvider.classify;
3250
+ if (classifierMode === "auto" && typeof classify === "function") {
3251
+ const slugs = effectiveManifest.node_types.map((n) => n.type);
3252
+ if (slugs.length > 0 && slugs.length <= 255) {
3253
+ return this._runClassifierBackfill(entityId, candidates, effectiveManifest, now, recheckCutoff);
3254
+ }
3255
+ }
3204
3256
  const ontologyContext = await ontologyService.buildPromptContext(entityId);
3205
3257
  const toPromptShape = (f) => ({ id: f.id, title: f.title, body: f.body, tags: f.tags });
3206
3258
  const buildPrompt = (facts) => this.promptService.buildOntologyBackfillPrompt(
@@ -3279,6 +3331,87 @@ var MaintenanceService = class {
3279
3331
  deferred: counts.deferred
3280
3332
  };
3281
3333
  }
3334
+ /**
3335
+ * Classifier-mode backfill (spec §7.3): one `choice` question per untyped
3336
+ * fact over the effective manifest's node-type slugs. Accepted answers go
3337
+ * through `_applyOntologyBackfillBatch` exactly like LLM classifications.
3338
+ * No edges are proposed.
3339
+ */
3340
+ async _runClassifierBackfill(entityId, candidates, manifest, now, recheckCutoff) {
3341
+ const provider = this.options.llmProvider;
3342
+ const rawMin = this.options.config?.ontology?.classifyMinConfidence;
3343
+ const minConfidence = typeof rawMin === "number" && Number.isFinite(rawMin) && rawMin >= 0 && rawMin <= 1 ? rawMin : 0.5;
3344
+ const rawConcurrency = this.options.config?.chunkConcurrency ?? 1;
3345
+ const concurrency = Number.isFinite(rawConcurrency) && rawConcurrency >= 1 ? Math.floor(rawConcurrency) : 1;
3346
+ const question = {
3347
+ kind: "choice",
3348
+ options: manifest.node_types.map((n) => n.type),
3349
+ instructions: "Choose the ontology node type that best describes this fact.\n" + manifest.node_types.map((n) => `- ${n.type}: ${n.description}`).join("\n")
3350
+ };
3351
+ const outcomes = await withConcurrency(
3352
+ candidates.map((fact) => async () => {
3353
+ let response;
3354
+ try {
3355
+ response = await provider.classify({ state: classifierStateForFact(fact), questions: { okf_type: question } });
3356
+ } catch {
3357
+ return { fact, kind: "threw" };
3358
+ }
3359
+ const checked = validateClassifierAnswer(response, "okf_type", question);
3360
+ if (!checked.ok) return { fact, kind: "invalid", reason: checked.reason };
3361
+ if (checked.answer.kind !== "choice") return { fact, kind: "invalid", reason: "kind_mismatch" };
3362
+ if (checked.answer.confidence < minConfidence) return { fact, kind: "low_confidence" };
3363
+ return { fact, kind: "accepted", okfType: checked.answer.choice };
3364
+ }),
3365
+ concurrency
3366
+ );
3367
+ const attempted = outcomes.filter((o) => o.kind !== "threw").map((o) => o.fact);
3368
+ const skipped = outcomes.length - attempted.length;
3369
+ const classifications = outcomes.filter((o) => o.kind === "accepted").map((o) => ({ id: o.fact.id, okf_type: o.okfType }));
3370
+ const invalidCount = outcomes.filter((o) => o.kind === "invalid").length;
3371
+ let typed = 0;
3372
+ let failedValidation = 0;
3373
+ let edgesAdded = 0;
3374
+ let aborted = false;
3375
+ if (attempted.length > 0) {
3376
+ const applied = await this._applyOntologyBackfillBatch(
3377
+ entityId,
3378
+ { batch: attempted, classifications, ontologyUpdates: void 0 },
3379
+ now
3380
+ );
3381
+ aborted = applied.abortedOntologyOff;
3382
+ if (!aborted) {
3383
+ typed = applied.typed;
3384
+ failedValidation = invalidCount + applied.failedValidation;
3385
+ edgesAdded = applied.edgesAdded;
3386
+ }
3387
+ }
3388
+ const diagBuffer = new DiagnosticBuffer();
3389
+ const diagBase = { entityId, operation: "ontologyBackfill", trigger: "call" };
3390
+ for (const o of outcomes) {
3391
+ if (o.kind === "low_confidence") {
3392
+ diagBuffer.push({ ...diagBase, code: "classification_low_confidence", detail: { factId: o.fact.id, reason: "below_threshold" } });
3393
+ } else if (o.kind === "invalid") {
3394
+ diagBuffer.push({ ...diagBase, code: "classification_invalid", detail: { factId: o.fact.id, reason: o.reason } });
3395
+ } else if (o.kind === "threw") {
3396
+ diagBuffer.push({ ...diagBase, code: "classification_invalid", detail: { factId: o.fact.id, reason: "classify_threw" } });
3397
+ }
3398
+ }
3399
+ diagBuffer.flush(this.options);
3400
+ const counts = await this.entryRepo.countUntypedByEntityId(entityId, recheckCutoff);
3401
+ if (aborted) {
3402
+ return { scanned: 0, typed: 0, failedValidation: 0, edgesAdded: 0, skipped, remaining: 0, deferred: counts.deferred };
3403
+ }
3404
+ this.searchService.evictCache(entityId);
3405
+ return {
3406
+ scanned: candidates.length,
3407
+ typed,
3408
+ failedValidation,
3409
+ edgesAdded,
3410
+ skipped,
3411
+ remaining: counts.eligible,
3412
+ deferred: counts.deferred
3413
+ };
3414
+ }
3282
3415
  /**
3283
3416
  * Applies one parsed backfill batch in its own transaction. Per-batch rather
3284
3417
  * than one transaction for the pass, so mergeEmergentUpdates semantics and
@@ -3516,6 +3649,7 @@ function selectWithFloors(sortedRows, floors, maxResults) {
3516
3649
  }
3517
3650
 
3518
3651
  // src/services/RetrievalService.ts
3652
+ var EMPTY_ID_SET = /* @__PURE__ */ new Set();
3519
3653
  var RetrievalService = class {
3520
3654
  constructor(options, entryRepo, taskRepo, eventRepo, metadataRepo, searchService) {
3521
3655
  this.options = options;
@@ -3548,6 +3682,7 @@ var RetrievalService = class {
3548
3682
  }
3549
3683
  const rawMaxResults = options?.maxResults ?? config?.maxResults ?? config?.maxFtsResults ?? 10;
3550
3684
  const maxResults = Number.isFinite(rawMaxResults) ? Math.max(0, Math.trunc(rawMaxResults)) : 10;
3685
+ const excludeDrafts = (options?.excludeDrafts ?? config?.excludeDrafts ?? false) === true;
3551
3686
  const trimmedQuery = query.trim();
3552
3687
  const sanitizedTierFloors = trimmedQuery && exposeMetadata ? validateTierFloors(
3553
3688
  entityIds,
@@ -3561,19 +3696,21 @@ var RetrievalService = class {
3561
3696
  const hybridWeight = options?.hybridWeight ?? config?.hybridWeight;
3562
3697
  const weight = hybridWeight !== void 0 && !Number.isNaN(hybridWeight) ? Math.max(0, Math.min(1, hybridWeight)) : void 0;
3563
3698
  const skipEmbed = weight === 0;
3564
- const embedFn = this.options.llmProvider.embed;
3565
3699
  let facts = [];
3566
3700
  let scoreByFactId;
3567
3701
  if (maxResults === 0) ; else if (trimmedQuery) {
3568
3702
  let usedEmbed = false;
3569
3703
  const scoredEntityIds = this._filterScoredEntities(entityIds, sanitizedTierWeights, options?.includeZeroWeightEntities);
3704
+ const draftIds = excludeDrafts && scoredEntityIds.length > 0 ? await this.entryRepo.findDraftIdsByEntityIds(scoredEntityIds) : EMPTY_ID_SET;
3705
+ const draftPad = draftIds.size;
3706
+ const padLimit = (n) => n >= Number.MAX_SAFE_INTEGER ? n : n + draftPad;
3570
3707
  if (scoredEntityIds.length === 0) {
3571
3708
  usedEmbed = true;
3572
- } else if (!skipEmbed && embedFn) {
3709
+ } else if (!skipEmbed && typeof this.options.llmProvider.embed === "function") {
3573
3710
  let rankerShouldRethrow = false;
3574
3711
  let pendingRankerFallbackError;
3575
3712
  try {
3576
- const queryVec = await embedFn(trimmedQuery);
3713
+ const queryVec = await this.options.llmProvider.embed(trimmedQuery);
3577
3714
  if (queryVec.length === 0 || !queryVec.every((v) => typeof v === "number" && isFinite(v))) {
3578
3715
  throw new Error(
3579
3716
  "embed() returned an empty or non-finite vector. Falling back to keyword search."
@@ -3600,7 +3737,7 @@ var RetrievalService = class {
3600
3737
  let miniSearchScores;
3601
3738
  if (effectivePreFilterLimit !== void 0) {
3602
3739
  populateCache = false;
3603
- const preResults = this.searchService.searchKeyword(trimmedQuery, scoredEntityIds, Number.MAX_SAFE_INTEGER);
3740
+ const preResults = this.searchService.searchKeyword(trimmedQuery, scoredEntityIds, Number.MAX_SAFE_INTEGER).filter((r) => !draftIds.has(r.id));
3604
3741
  if (preResults.length === 0) {
3605
3742
  candidateRows = null;
3606
3743
  } else {
@@ -3622,9 +3759,9 @@ var RetrievalService = class {
3622
3759
  }
3623
3760
  } else {
3624
3761
  if (useRanker) {
3625
- candidateRows = await this.entryRepo.findMetadataByEntityIds(scoredEntityIds);
3762
+ candidateRows = this._withoutDrafts(await this.entryRepo.findMetadataByEntityIds(scoredEntityIds), draftIds);
3626
3763
  } else {
3627
- candidateRows = await this.entryRepo.findWithEmbeddingsByEntityIds(scoredEntityIds);
3764
+ candidateRows = this._withoutDrafts(await this.entryRepo.findWithEmbeddingsByEntityIds(scoredEntityIds), draftIds);
3628
3765
  }
3629
3766
  if (weight !== void 0 && weight < 1) {
3630
3767
  miniSearchScores = this.searchService.getMiniSearchScores(trimmedQuery, scoredEntityIds);
@@ -3654,7 +3791,7 @@ var RetrievalService = class {
3654
3791
  candidateRows: rowsForEntity,
3655
3792
  weight,
3656
3793
  miniSearchScores,
3657
- limit: Math.max(maxResults * 2, maxResults + 50)
3794
+ limit: padLimit(Math.max(maxResults * 2, maxResults + 50))
3658
3795
  });
3659
3796
  return ranked.map((row) => ({ ...row, entity_id: scopedEntityId }));
3660
3797
  })
@@ -3868,7 +4005,7 @@ var RetrievalService = class {
3868
4005
  });
3869
4006
  } else if (policy === "keyword") {
3870
4007
  const hasActiveFloorsKeywordFallback = sanitizedTierFloors !== void 0 && Object.values(sanitizedTierFloors).some((f) => f > 0);
3871
- const keywordOversampledLimit = hasActiveFloorsKeywordFallback ? Number.MAX_SAFE_INTEGER : Math.max(maxResults * 2, maxResults + 50);
4008
+ const keywordOversampledLimit = hasActiveFloorsKeywordFallback ? Number.MAX_SAFE_INTEGER : padLimit(Math.max(maxResults * 2, maxResults + 50));
3872
4009
  const preFilteredIds = effectivePreFilterLimit !== void 0 ? new Set(candidateRows.map((r) => r.id)) : void 0;
3873
4010
  const keywordResults = this.searchService.searchKeyword(trimmedQuery, scoredEntityIds, keywordOversampledLimit);
3874
4011
  const topResults = preFilteredIds === void 0 ? keywordResults : keywordResults.filter((r) => preFilteredIds.has(r.id));
@@ -3909,6 +4046,7 @@ var RetrievalService = class {
3909
4046
  // read() re-sorts after applying tier weights
3910
4047
  });
3911
4048
  }
4049
+ if (draftPad > 0) scored = scored.filter((s) => !draftIds.has(s.id));
3912
4050
  if (scored.length > 0) {
3913
4051
  scored = scored.map((row) => ({
3914
4052
  ...row,
@@ -3971,8 +4109,8 @@ var RetrievalService = class {
3971
4109
  }
3972
4110
  if (!usedEmbed && scoredEntityIds.length > 0) {
3973
4111
  const hasActiveFloors = sanitizedTierFloors !== void 0 && Object.values(sanitizedTierFloors).some((f) => f > 0);
3974
- const fallbackOversampledLimit = hasActiveFloors ? Number.MAX_SAFE_INTEGER : Math.max(maxResults * 2, maxResults + 50);
3975
- const results = this.searchService.searchKeyword(trimmedQuery, scoredEntityIds, fallbackOversampledLimit);
4112
+ const fallbackOversampledLimit = hasActiveFloors ? Number.MAX_SAFE_INTEGER : padLimit(Math.max(maxResults * 2, maxResults + 50));
4113
+ const results = this.searchService.searchKeyword(trimmedQuery, scoredEntityIds, fallbackOversampledLimit).filter((r) => !draftIds.has(r.id));
3976
4114
  const candidates = results.map((r) => ({
3977
4115
  id: r.id,
3978
4116
  entity_id: r.entity_id,
@@ -3996,7 +4134,7 @@ var RetrievalService = class {
3996
4134
  await this.entryRepo.trackAccess(ids, now);
3997
4135
  }
3998
4136
  } else {
3999
- facts = await this.entryRepo.findRecentByEntityIds(entityIds, maxResults);
4137
+ facts = excludeDrafts ? await this.entryRepo.findRecentByEntityIds(entityIds, maxResults, void 0, { excludeDrafts: true }) : await this.entryRepo.findRecentByEntityIds(entityIds, maxResults);
4000
4138
  }
4001
4139
  const eventsLimit = Math.min(10 * entityIds.length, 100);
4002
4140
  const [tasks, events] = await Promise.all([
@@ -4032,6 +4170,9 @@ var RetrievalService = class {
4032
4170
  _tieBreakSort(items) {
4033
4171
  items.sort((a, b) => this._compareScoredRows(a, b));
4034
4172
  }
4173
+ _withoutDrafts(rows, draftIds) {
4174
+ return draftIds.size === 0 ? rows : rows.filter((row) => !draftIds.has(row.id));
4175
+ }
4035
4176
  /**
4036
4177
  * Comparator for score + deterministic tie-break fields.
4037
4178
  * Negative return means "a ranks ahead of b" for descending score order.