@equationalapplications/core-llm-wiki 7.4.0 → 7.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +41 -0
- package/dist/{chunk-G5OR2VWD.mjs → chunk-3P7FAKJA.mjs} +141 -9
- package/dist/chunk-3P7FAKJA.mjs.map +1 -0
- package/dist/index.d.mts +2 -2
- package/dist/index.d.ts +2 -2
- package/dist/index.js +139 -7
- package/dist/index.js.map +1 -1
- package/dist/index.mjs +2 -2
- package/dist/index.mjs.map +1 -1
- package/dist/{testing-Dmh1kfkd.d.mts → testing-ClpOUiLI.d.mts} +66 -1
- package/dist/{testing-Dmh1kfkd.d.ts → testing-ClpOUiLI.d.ts} +66 -1
- package/dist/testing.d.mts +1 -1
- package/dist/testing.d.ts +1 -1
- package/dist/testing.js +139 -7
- package/dist/testing.js.map +1 -1
- package/dist/testing.mjs +1 -1
- package/package.json +2 -2
- package/dist/chunk-G5OR2VWD.mjs.map +0 -1
|
@@ -82,6 +82,14 @@ interface OntologyConfig {
|
|
|
82
82
|
manifest: OntologyManifest;
|
|
83
83
|
mode?: OntologyMode;
|
|
84
84
|
}>;
|
|
85
|
+
/**
|
|
86
|
+
* Engine default for `runOntologyBackfill`'s `classifier` option.
|
|
87
|
+
* `'auto'` uses `llmProvider.classify` for node typing when present
|
|
88
|
+
* (no edges are proposed); `'llm'` always uses `generateText`. Default `'llm'`.
|
|
89
|
+
*/
|
|
90
|
+
backfillClassifier?: 'auto' | 'llm';
|
|
91
|
+
/** Minimum classifier confidence to apply a node type. Finite, in [0, 1]. Default 0.5. */
|
|
92
|
+
classifyMinConfidence?: number;
|
|
85
93
|
}
|
|
86
94
|
interface ExtractedFactEdge {
|
|
87
95
|
edge_type: string;
|
|
@@ -498,6 +506,8 @@ interface LLMProvider {
|
|
|
498
506
|
/**
|
|
499
507
|
* Generates text using the developer's LLM of choice.
|
|
500
508
|
* Expected to return the raw text response (typically a JSON string).
|
|
509
|
+
* Called with the provider as `this`, so class-based adapters may keep
|
|
510
|
+
* SDK handles and config on the instance.
|
|
501
511
|
*/
|
|
502
512
|
generateText: (params: {
|
|
503
513
|
systemPrompt: string;
|
|
@@ -508,6 +518,8 @@ interface LLMProvider {
|
|
|
508
518
|
* Must return a stable-dimension float array for any input text.
|
|
509
519
|
* Called once per fact on creation/update, and once per `read()` query.
|
|
510
520
|
* When absent or throws, `read()` falls back to MiniSearch.
|
|
521
|
+
* Called with the provider as `this` (never detached), same as
|
|
522
|
+
* `generateText`, so class-based adapters may read their own state.
|
|
511
523
|
*/
|
|
512
524
|
embed?: (text: string) => Promise<number[]>;
|
|
513
525
|
/**
|
|
@@ -521,6 +533,49 @@ interface LLMProvider {
|
|
|
521
533
|
* never trusted as a guarantee.
|
|
522
534
|
*/
|
|
523
535
|
maxOutputTokens?: number;
|
|
536
|
+
/**
|
|
537
|
+
* Optional non-generative classifier (System-One models such as Jev,
|
|
538
|
+
* OpenJev, or a local ONNX classifier). Core never calls it unless the host
|
|
539
|
+
* opts in (e.g. `WikiConfig.ontology.backfillClassifier: 'auto'`); merely
|
|
540
|
+
* providing it changes nothing (REQ-COMPAT-01.5). Output is validated as
|
|
541
|
+
* untrusted.
|
|
542
|
+
*/
|
|
543
|
+
classify?: (request: ClassifyRequest) => Promise<ClassifyResponse>;
|
|
544
|
+
}
|
|
545
|
+
/** One typed question for a classifier. Vendor-neutral (Jev `noul` ↔ `binary`). */
|
|
546
|
+
type ClassifierQuestion = {
|
|
547
|
+
kind: 'choice';
|
|
548
|
+
options: string[];
|
|
549
|
+
instructions?: string;
|
|
550
|
+
} | {
|
|
551
|
+
kind: 'binary';
|
|
552
|
+
instructions: string;
|
|
553
|
+
} | {
|
|
554
|
+
kind: 'score';
|
|
555
|
+
levels: string[];
|
|
556
|
+
instructions?: string;
|
|
557
|
+
};
|
|
558
|
+
/** One state evaluated against a map of questions (many questions per state, one state per call). */
|
|
559
|
+
interface ClassifyRequest {
|
|
560
|
+
state: string;
|
|
561
|
+
questions: Record<string, ClassifierQuestion>;
|
|
562
|
+
}
|
|
563
|
+
type ClassifierAnswer = {
|
|
564
|
+
kind: 'choice';
|
|
565
|
+
choice: string;
|
|
566
|
+
confidence: number;
|
|
567
|
+
probabilities: Record<string, number>;
|
|
568
|
+
} | {
|
|
569
|
+
kind: 'binary';
|
|
570
|
+
probability: number;
|
|
571
|
+
} | {
|
|
572
|
+
kind: 'score';
|
|
573
|
+
score: number;
|
|
574
|
+
confidence: number;
|
|
575
|
+
probabilities: number[];
|
|
576
|
+
};
|
|
577
|
+
interface ClassifyResponse {
|
|
578
|
+
answers: Record<string, ClassifierAnswer>;
|
|
524
579
|
}
|
|
525
580
|
/**
|
|
526
581
|
* Result of semantic ranking for a single fact.
|
|
@@ -2275,6 +2330,7 @@ declare class MaintenanceService {
|
|
|
2275
2330
|
runOntologyBackfill(entityId: string, options?: {
|
|
2276
2331
|
promptOverride?: string;
|
|
2277
2332
|
batchSize?: number;
|
|
2333
|
+
classifier?: 'auto' | 'llm';
|
|
2278
2334
|
}): Promise<OntologyBackfillResult>;
|
|
2279
2335
|
/**
|
|
2280
2336
|
* Re-embed facts, honouring per-row failure markers and exponential backoff.
|
|
@@ -2349,7 +2405,15 @@ declare class MaintenanceService {
|
|
|
2349
2405
|
doRunOntologyBackfill(entityId: string, options?: {
|
|
2350
2406
|
promptOverride?: string;
|
|
2351
2407
|
batchSize?: number;
|
|
2408
|
+
classifier?: 'auto' | 'llm';
|
|
2352
2409
|
}): Promise<OntologyBackfillResult>;
|
|
2410
|
+
/**
|
|
2411
|
+
* Classifier-mode backfill (spec §7.3): one `choice` question per untyped
|
|
2412
|
+
* fact over the effective manifest's node-type slugs. Accepted answers go
|
|
2413
|
+
* through `_applyOntologyBackfillBatch` exactly like LLM classifications.
|
|
2414
|
+
* No edges are proposed.
|
|
2415
|
+
*/
|
|
2416
|
+
private _runClassifierBackfill;
|
|
2353
2417
|
/**
|
|
2354
2418
|
* Applies one parsed backfill batch in its own transaction. Per-batch rather
|
|
2355
2419
|
* than one transaction for the pass, so mergeEmergentUpdates semantics and
|
|
@@ -2657,6 +2721,7 @@ declare class WikiMemory {
|
|
|
2657
2721
|
runOntologyBackfill(entityId: string, options?: {
|
|
2658
2722
|
promptOverride?: string;
|
|
2659
2723
|
batchSize?: number;
|
|
2724
|
+
classifier?: 'auto' | 'llm';
|
|
2660
2725
|
}): Promise<OntologyBackfillResult>;
|
|
2661
2726
|
runReembed(entityId?: string, opts?: {
|
|
2662
2727
|
force?: boolean;
|
|
@@ -2877,4 +2942,4 @@ declare class WikiMemory {
|
|
|
2877
2942
|
setGeneratedByTask(taskId: string, entityId: string, actor: string): Promise<void>;
|
|
2878
2943
|
}
|
|
2879
2944
|
|
|
2880
|
-
export { type
|
|
2945
|
+
export { type WikiConfig as $, type OntologyNodeType as A, type OntologyPromptContext as B, type ChunkFailure as C, type DegradedRecord as D, type EmbedFactResult as E, type FormatContextOptions as F, type GraphNeighborhood as G, HEAL_BATCH_SIZE as H, type IngestDocumentResult as I, type OntologyUpdates as J, PromptService as K, type LLMProvider as L, type MemoryBundle as M, PrunePartialFailureError as N, ONTOLOGY_BACKFILL_BATCH_SIZE as O, type PromptOverrides as P, type ReembedResult as Q, type ReadOptions as R, type SQLiteAdapter as S, type VectorRankerFallback as T, type VectorRankerRankArgs as U, type VectorRanker as V, type WikiOptions as W, type VectorRankerSemanticResult as X, WikiBusyError as Y, type WikiBusyOperation as Z, type WikiCheckpoint as _, type MemoryDump as a, type WikiDiagnostic as a0, type WikiDiagnosticCode as a1, type WikiDiagnosticDetail as a2, type WikiDiagnosticOperation as a3, type WikiDiagnosticSeverity as a4, type WikiDiagnosticTrigger as a5, WikiDraftNotFound as a6, WikiDuplicateHashError as a7, type WikiEdge as a8, type WikiEvent as a9, type WikiFact as aa, WikiGraphNodeOwnershipConflict as ab, WikiIngestEmptyError as ac, WikiInvalidReadOptions as ad, type WikiMemoryTestAccess as ae, type WikiOutboxEvent as af, WikiParseError as ag, WikiSourceRefHashCollision as ah, WikiStrictOntologyViolation as ai, type WikiTask as aj, WikiTransactionError as ak, validateManifest as al, EmbeddingService as am, ImportExportService as an, IngestionService as ao, JobManager as ap, MaintenanceService as aq, RetrievalService as ar, SearchService as as, WriteService as at, type FormattedMemoryDump as b, WikiMemory as c, type ClassifierAnswer as d, type ClassifierQuestion as e, type ClassifyRequest as f, type ClassifyResponse as g, type DraftPage as h, type EmbedFailureKind as i, type EmbeddingMarkerKind as j, type EntityStatus as k, type ExtractedFact as l, type ExtractedFactEdge as m, type ExtractedFactWithOntology as n, type ExtractedTask as o, type GraphTraversalOptions as p, HEAL_RECHECK_MS as q, HOOK_TIMEOUT_MARKER as r, type HealResult as s, ONTOLOGY_BACKFILL_MAX_PROMPT_CHARS as t, ONTOLOGY_BACKFILL_RECHECK_MS as u, type OntologyBackfillResult as v, type OntologyConfig as w, type OntologyEdgeType as x, type OntologyManifest as y, type OntologyMode as z };
|
|
@@ -82,6 +82,14 @@ interface OntologyConfig {
|
|
|
82
82
|
manifest: OntologyManifest;
|
|
83
83
|
mode?: OntologyMode;
|
|
84
84
|
}>;
|
|
85
|
+
/**
|
|
86
|
+
* Engine default for `runOntologyBackfill`'s `classifier` option.
|
|
87
|
+
* `'auto'` uses `llmProvider.classify` for node typing when present
|
|
88
|
+
* (no edges are proposed); `'llm'` always uses `generateText`. Default `'llm'`.
|
|
89
|
+
*/
|
|
90
|
+
backfillClassifier?: 'auto' | 'llm';
|
|
91
|
+
/** Minimum classifier confidence to apply a node type. Finite, in [0, 1]. Default 0.5. */
|
|
92
|
+
classifyMinConfidence?: number;
|
|
85
93
|
}
|
|
86
94
|
interface ExtractedFactEdge {
|
|
87
95
|
edge_type: string;
|
|
@@ -498,6 +506,8 @@ interface LLMProvider {
|
|
|
498
506
|
/**
|
|
499
507
|
* Generates text using the developer's LLM of choice.
|
|
500
508
|
* Expected to return the raw text response (typically a JSON string).
|
|
509
|
+
* Called with the provider as `this`, so class-based adapters may keep
|
|
510
|
+
* SDK handles and config on the instance.
|
|
501
511
|
*/
|
|
502
512
|
generateText: (params: {
|
|
503
513
|
systemPrompt: string;
|
|
@@ -508,6 +518,8 @@ interface LLMProvider {
|
|
|
508
518
|
* Must return a stable-dimension float array for any input text.
|
|
509
519
|
* Called once per fact on creation/update, and once per `read()` query.
|
|
510
520
|
* When absent or throws, `read()` falls back to MiniSearch.
|
|
521
|
+
* Called with the provider as `this` (never detached), same as
|
|
522
|
+
* `generateText`, so class-based adapters may read their own state.
|
|
511
523
|
*/
|
|
512
524
|
embed?: (text: string) => Promise<number[]>;
|
|
513
525
|
/**
|
|
@@ -521,6 +533,49 @@ interface LLMProvider {
|
|
|
521
533
|
* never trusted as a guarantee.
|
|
522
534
|
*/
|
|
523
535
|
maxOutputTokens?: number;
|
|
536
|
+
/**
|
|
537
|
+
* Optional non-generative classifier (System-One models such as Jev,
|
|
538
|
+
* OpenJev, or a local ONNX classifier). Core never calls it unless the host
|
|
539
|
+
* opts in (e.g. `WikiConfig.ontology.backfillClassifier: 'auto'`); merely
|
|
540
|
+
* providing it changes nothing (REQ-COMPAT-01.5). Output is validated as
|
|
541
|
+
* untrusted.
|
|
542
|
+
*/
|
|
543
|
+
classify?: (request: ClassifyRequest) => Promise<ClassifyResponse>;
|
|
544
|
+
}
|
|
545
|
+
/** One typed question for a classifier. Vendor-neutral (Jev `noul` ↔ `binary`). */
|
|
546
|
+
type ClassifierQuestion = {
|
|
547
|
+
kind: 'choice';
|
|
548
|
+
options: string[];
|
|
549
|
+
instructions?: string;
|
|
550
|
+
} | {
|
|
551
|
+
kind: 'binary';
|
|
552
|
+
instructions: string;
|
|
553
|
+
} | {
|
|
554
|
+
kind: 'score';
|
|
555
|
+
levels: string[];
|
|
556
|
+
instructions?: string;
|
|
557
|
+
};
|
|
558
|
+
/** One state evaluated against a map of questions (many questions per state, one state per call). */
|
|
559
|
+
interface ClassifyRequest {
|
|
560
|
+
state: string;
|
|
561
|
+
questions: Record<string, ClassifierQuestion>;
|
|
562
|
+
}
|
|
563
|
+
type ClassifierAnswer = {
|
|
564
|
+
kind: 'choice';
|
|
565
|
+
choice: string;
|
|
566
|
+
confidence: number;
|
|
567
|
+
probabilities: Record<string, number>;
|
|
568
|
+
} | {
|
|
569
|
+
kind: 'binary';
|
|
570
|
+
probability: number;
|
|
571
|
+
} | {
|
|
572
|
+
kind: 'score';
|
|
573
|
+
score: number;
|
|
574
|
+
confidence: number;
|
|
575
|
+
probabilities: number[];
|
|
576
|
+
};
|
|
577
|
+
interface ClassifyResponse {
|
|
578
|
+
answers: Record<string, ClassifierAnswer>;
|
|
524
579
|
}
|
|
525
580
|
/**
|
|
526
581
|
* Result of semantic ranking for a single fact.
|
|
@@ -2275,6 +2330,7 @@ declare class MaintenanceService {
|
|
|
2275
2330
|
runOntologyBackfill(entityId: string, options?: {
|
|
2276
2331
|
promptOverride?: string;
|
|
2277
2332
|
batchSize?: number;
|
|
2333
|
+
classifier?: 'auto' | 'llm';
|
|
2278
2334
|
}): Promise<OntologyBackfillResult>;
|
|
2279
2335
|
/**
|
|
2280
2336
|
* Re-embed facts, honouring per-row failure markers and exponential backoff.
|
|
@@ -2349,7 +2405,15 @@ declare class MaintenanceService {
|
|
|
2349
2405
|
doRunOntologyBackfill(entityId: string, options?: {
|
|
2350
2406
|
promptOverride?: string;
|
|
2351
2407
|
batchSize?: number;
|
|
2408
|
+
classifier?: 'auto' | 'llm';
|
|
2352
2409
|
}): Promise<OntologyBackfillResult>;
|
|
2410
|
+
/**
|
|
2411
|
+
* Classifier-mode backfill (spec §7.3): one `choice` question per untyped
|
|
2412
|
+
* fact over the effective manifest's node-type slugs. Accepted answers go
|
|
2413
|
+
* through `_applyOntologyBackfillBatch` exactly like LLM classifications.
|
|
2414
|
+
* No edges are proposed.
|
|
2415
|
+
*/
|
|
2416
|
+
private _runClassifierBackfill;
|
|
2353
2417
|
/**
|
|
2354
2418
|
* Applies one parsed backfill batch in its own transaction. Per-batch rather
|
|
2355
2419
|
* than one transaction for the pass, so mergeEmergentUpdates semantics and
|
|
@@ -2657,6 +2721,7 @@ declare class WikiMemory {
|
|
|
2657
2721
|
runOntologyBackfill(entityId: string, options?: {
|
|
2658
2722
|
promptOverride?: string;
|
|
2659
2723
|
batchSize?: number;
|
|
2724
|
+
classifier?: 'auto' | 'llm';
|
|
2660
2725
|
}): Promise<OntologyBackfillResult>;
|
|
2661
2726
|
runReembed(entityId?: string, opts?: {
|
|
2662
2727
|
force?: boolean;
|
|
@@ -2877,4 +2942,4 @@ declare class WikiMemory {
|
|
|
2877
2942
|
setGeneratedByTask(taskId: string, entityId: string, actor: string): Promise<void>;
|
|
2878
2943
|
}
|
|
2879
2944
|
|
|
2880
|
-
export { type
|
|
2945
|
+
export { type WikiConfig as $, type OntologyNodeType as A, type OntologyPromptContext as B, type ChunkFailure as C, type DegradedRecord as D, type EmbedFactResult as E, type FormatContextOptions as F, type GraphNeighborhood as G, HEAL_BATCH_SIZE as H, type IngestDocumentResult as I, type OntologyUpdates as J, PromptService as K, type LLMProvider as L, type MemoryBundle as M, PrunePartialFailureError as N, ONTOLOGY_BACKFILL_BATCH_SIZE as O, type PromptOverrides as P, type ReembedResult as Q, type ReadOptions as R, type SQLiteAdapter as S, type VectorRankerFallback as T, type VectorRankerRankArgs as U, type VectorRanker as V, type WikiOptions as W, type VectorRankerSemanticResult as X, WikiBusyError as Y, type WikiBusyOperation as Z, type WikiCheckpoint as _, type MemoryDump as a, type WikiDiagnostic as a0, type WikiDiagnosticCode as a1, type WikiDiagnosticDetail as a2, type WikiDiagnosticOperation as a3, type WikiDiagnosticSeverity as a4, type WikiDiagnosticTrigger as a5, WikiDraftNotFound as a6, WikiDuplicateHashError as a7, type WikiEdge as a8, type WikiEvent as a9, type WikiFact as aa, WikiGraphNodeOwnershipConflict as ab, WikiIngestEmptyError as ac, WikiInvalidReadOptions as ad, type WikiMemoryTestAccess as ae, type WikiOutboxEvent as af, WikiParseError as ag, WikiSourceRefHashCollision as ah, WikiStrictOntologyViolation as ai, type WikiTask as aj, WikiTransactionError as ak, validateManifest as al, EmbeddingService as am, ImportExportService as an, IngestionService as ao, JobManager as ap, MaintenanceService as aq, RetrievalService as ar, SearchService as as, WriteService as at, type FormattedMemoryDump as b, WikiMemory as c, type ClassifierAnswer as d, type ClassifierQuestion as e, type ClassifyRequest as f, type ClassifyResponse as g, type DraftPage as h, type EmbedFailureKind as i, type EmbeddingMarkerKind as j, type EntityStatus as k, type ExtractedFact as l, type ExtractedFactEdge as m, type ExtractedFactWithOntology as n, type ExtractedTask as o, type GraphTraversalOptions as p, HEAL_RECHECK_MS as q, HOOK_TIMEOUT_MARKER as r, type HealResult as s, ONTOLOGY_BACKFILL_MAX_PROMPT_CHARS as t, ONTOLOGY_BACKFILL_RECHECK_MS as u, type OntologyBackfillResult as v, type OntologyConfig as w, type OntologyEdgeType as x, type OntologyManifest as y, type OntologyMode as z };
|
package/dist/testing.d.mts
CHANGED
|
@@ -1,3 +1,3 @@
|
|
|
1
|
-
export {
|
|
1
|
+
export { am as EmbeddingService, an as ImportExportService, ao as IngestionService, ap as JobManager, ap as JobManagerType, aq as MaintenanceService, ar as RetrievalService, as as SearchService, as as SearchServiceType, ae as WikiMemoryTestAccess, at as WriteService } from './testing-ClpOUiLI.mjs';
|
|
2
2
|
import '@equationalapplications/core-okf';
|
|
3
3
|
import 'minisearch';
|
package/dist/testing.d.ts
CHANGED
|
@@ -1,3 +1,3 @@
|
|
|
1
|
-
export {
|
|
1
|
+
export { am as EmbeddingService, an as ImportExportService, ao as IngestionService, ap as JobManager, ap as JobManagerType, aq as MaintenanceService, ar as RetrievalService, as as SearchService, as as SearchServiceType, ae as WikiMemoryTestAccess, at as WriteService } from './testing-ClpOUiLI.js';
|
|
2
2
|
import '@equationalapplications/core-okf';
|
|
3
3
|
import 'minisearch';
|
package/dist/testing.js
CHANGED
|
@@ -817,8 +817,7 @@ var EmbeddingService = class {
|
|
|
817
817
|
}
|
|
818
818
|
}
|
|
819
819
|
async tryEmbedFact(fact, ctx) {
|
|
820
|
-
|
|
821
|
-
if (typeof embedFn !== "function") return { ok: false, kind: "no_provider" };
|
|
820
|
+
if (typeof this.options.llmProvider.embed !== "function") return { ok: false, kind: "no_provider" };
|
|
822
821
|
let tagsStr;
|
|
823
822
|
if (Array.isArray(fact.tags)) {
|
|
824
823
|
tagsStr = fact.tags.join(" ");
|
|
@@ -835,7 +834,7 @@ var EmbeddingService = class {
|
|
|
835
834
|
const text = clip(`${fact.title} ${fact.body} ${tagsStr}`.trim(), maxEmbedChars);
|
|
836
835
|
let float32Vector;
|
|
837
836
|
try {
|
|
838
|
-
const vector = await
|
|
837
|
+
const vector = await this.options.llmProvider.embed(text);
|
|
839
838
|
if (vector.length === 0 || !vector.every((v) => typeof v === "number" && isFinite(v))) {
|
|
840
839
|
console.warn(`[WikiMemory] embedFact: embed() returned an invalid vector for ${fact.id}; skipping.`);
|
|
841
840
|
this.reportEmbed(ctx, fact, "embedding_failed", "invalid_vector");
|
|
@@ -2281,6 +2280,51 @@ var IngestionService = class {
|
|
|
2281
2280
|
}
|
|
2282
2281
|
};
|
|
2283
2282
|
|
|
2283
|
+
// src/utils/classifier.ts
|
|
2284
|
+
var isUnit = (v) => typeof v === "number" && Number.isFinite(v) && v >= 0 && v <= 1;
|
|
2285
|
+
function validateClassifierAnswer(response, key, question) {
|
|
2286
|
+
try {
|
|
2287
|
+
if (response === null || typeof response !== "object") return { ok: false, reason: "malformed" };
|
|
2288
|
+
const answers = response.answers;
|
|
2289
|
+
if (answers === null || typeof answers !== "object") return { ok: false, reason: "missing_answer" };
|
|
2290
|
+
if (!Object.prototype.hasOwnProperty.call(answers, key)) return { ok: false, reason: "missing_answer" };
|
|
2291
|
+
const raw = answers[key];
|
|
2292
|
+
if (raw === null || typeof raw !== "object") return { ok: false, reason: "missing_answer" };
|
|
2293
|
+
const a = raw;
|
|
2294
|
+
if (a.kind !== question.kind) return { ok: false, reason: "kind_mismatch" };
|
|
2295
|
+
if (question.kind === "choice") {
|
|
2296
|
+
if (typeof a.choice !== "string" || !question.options.includes(a.choice)) return { ok: false, reason: "choice_not_offered" };
|
|
2297
|
+
if (!isUnit(a.confidence)) return { ok: false, reason: "invalid_probability" };
|
|
2298
|
+
const probs = a.probabilities;
|
|
2299
|
+
if (probs === null || typeof probs !== "object" || Array.isArray(probs)) return { ok: false, reason: "invalid_probability" };
|
|
2300
|
+
const probabilities = {};
|
|
2301
|
+
for (const [k, v] of Object.entries(probs)) {
|
|
2302
|
+
if (!isUnit(v)) return { ok: false, reason: "invalid_probability" };
|
|
2303
|
+
probabilities[k] = v;
|
|
2304
|
+
}
|
|
2305
|
+
return { ok: true, answer: { kind: "choice", choice: a.choice, confidence: a.confidence, probabilities } };
|
|
2306
|
+
}
|
|
2307
|
+
if (question.kind === "binary") {
|
|
2308
|
+
if (!isUnit(a.probability)) return { ok: false, reason: "invalid_probability" };
|
|
2309
|
+
return { ok: true, answer: { kind: "binary", probability: a.probability } };
|
|
2310
|
+
}
|
|
2311
|
+
const maxScore = question.levels.length - 1;
|
|
2312
|
+
if (typeof a.score !== "number" || !Number.isFinite(a.score) || a.score < 0 || a.score > maxScore) {
|
|
2313
|
+
return { ok: false, reason: "score_out_of_range" };
|
|
2314
|
+
}
|
|
2315
|
+
if (!isUnit(a.confidence)) return { ok: false, reason: "invalid_probability" };
|
|
2316
|
+
if (!Array.isArray(a.probabilities) || !a.probabilities.every(isUnit)) return { ok: false, reason: "invalid_probability" };
|
|
2317
|
+
return { ok: true, answer: { kind: "score", score: a.score, confidence: a.confidence, probabilities: a.probabilities.slice() } };
|
|
2318
|
+
} catch {
|
|
2319
|
+
return { ok: false, reason: "malformed" };
|
|
2320
|
+
}
|
|
2321
|
+
}
|
|
2322
|
+
function classifierStateForFact(fact) {
|
|
2323
|
+
const parts = [fact.title, fact.body];
|
|
2324
|
+
if (Array.isArray(fact.tags) && fact.tags.length > 0) parts.push(`Tags: ${fact.tags.join(", ")}`);
|
|
2325
|
+
return parts.join("\n\n");
|
|
2326
|
+
}
|
|
2327
|
+
|
|
2284
2328
|
// src/utils/embedding.ts
|
|
2285
2329
|
function parseEmbedding(blob, text) {
|
|
2286
2330
|
if (blob && blob.byteLength > 0) {
|
|
@@ -3191,7 +3235,7 @@ var MaintenanceService = class {
|
|
|
3191
3235
|
if (!ontologyService) {
|
|
3192
3236
|
return { ...zeroed, remaining: 0, deferred: 0 };
|
|
3193
3237
|
}
|
|
3194
|
-
const { mode } = await ontologyService.getEffectiveState(entityId);
|
|
3238
|
+
const { mode, manifest: effectiveManifest } = await ontologyService.getEffectiveState(entityId);
|
|
3195
3239
|
if (mode === "off") {
|
|
3196
3240
|
const counts2 = await this.entryRepo.countUntypedByEntityId(entityId, recheckCutoff);
|
|
3197
3241
|
return { ...zeroed, remaining: 0, deferred: counts2.deferred };
|
|
@@ -3201,6 +3245,14 @@ var MaintenanceService = class {
|
|
|
3201
3245
|
const counts2 = await this.entryRepo.countUntypedByEntityId(entityId, recheckCutoff);
|
|
3202
3246
|
return { ...zeroed, remaining: counts2.eligible, deferred: counts2.deferred };
|
|
3203
3247
|
}
|
|
3248
|
+
const classifierMode = options?.classifier ?? this.options.config?.ontology?.backfillClassifier ?? "llm";
|
|
3249
|
+
const classify = this.options.llmProvider.classify;
|
|
3250
|
+
if (classifierMode === "auto" && typeof classify === "function") {
|
|
3251
|
+
const slugs = effectiveManifest.node_types.map((n) => n.type);
|
|
3252
|
+
if (slugs.length > 0 && slugs.length <= 255) {
|
|
3253
|
+
return this._runClassifierBackfill(entityId, candidates, effectiveManifest, now, recheckCutoff);
|
|
3254
|
+
}
|
|
3255
|
+
}
|
|
3204
3256
|
const ontologyContext = await ontologyService.buildPromptContext(entityId);
|
|
3205
3257
|
const toPromptShape = (f) => ({ id: f.id, title: f.title, body: f.body, tags: f.tags });
|
|
3206
3258
|
const buildPrompt = (facts) => this.promptService.buildOntologyBackfillPrompt(
|
|
@@ -3279,6 +3331,87 @@ var MaintenanceService = class {
|
|
|
3279
3331
|
deferred: counts.deferred
|
|
3280
3332
|
};
|
|
3281
3333
|
}
|
|
3334
|
+
/**
|
|
3335
|
+
* Classifier-mode backfill (spec §7.3): one `choice` question per untyped
|
|
3336
|
+
* fact over the effective manifest's node-type slugs. Accepted answers go
|
|
3337
|
+
* through `_applyOntologyBackfillBatch` exactly like LLM classifications.
|
|
3338
|
+
* No edges are proposed.
|
|
3339
|
+
*/
|
|
3340
|
+
async _runClassifierBackfill(entityId, candidates, manifest, now, recheckCutoff) {
|
|
3341
|
+
const provider = this.options.llmProvider;
|
|
3342
|
+
const rawMin = this.options.config?.ontology?.classifyMinConfidence;
|
|
3343
|
+
const minConfidence = typeof rawMin === "number" && Number.isFinite(rawMin) && rawMin >= 0 && rawMin <= 1 ? rawMin : 0.5;
|
|
3344
|
+
const rawConcurrency = this.options.config?.chunkConcurrency ?? 1;
|
|
3345
|
+
const concurrency = Number.isFinite(rawConcurrency) && rawConcurrency >= 1 ? Math.floor(rawConcurrency) : 1;
|
|
3346
|
+
const question = {
|
|
3347
|
+
kind: "choice",
|
|
3348
|
+
options: manifest.node_types.map((n) => n.type),
|
|
3349
|
+
instructions: "Choose the ontology node type that best describes this fact.\n" + manifest.node_types.map((n) => `- ${n.type}: ${n.description}`).join("\n")
|
|
3350
|
+
};
|
|
3351
|
+
const outcomes = await withConcurrency(
|
|
3352
|
+
candidates.map((fact) => async () => {
|
|
3353
|
+
let response;
|
|
3354
|
+
try {
|
|
3355
|
+
response = await provider.classify({ state: classifierStateForFact(fact), questions: { okf_type: question } });
|
|
3356
|
+
} catch {
|
|
3357
|
+
return { fact, kind: "threw" };
|
|
3358
|
+
}
|
|
3359
|
+
const checked = validateClassifierAnswer(response, "okf_type", question);
|
|
3360
|
+
if (!checked.ok) return { fact, kind: "invalid", reason: checked.reason };
|
|
3361
|
+
if (checked.answer.kind !== "choice") return { fact, kind: "invalid", reason: "kind_mismatch" };
|
|
3362
|
+
if (checked.answer.confidence < minConfidence) return { fact, kind: "low_confidence" };
|
|
3363
|
+
return { fact, kind: "accepted", okfType: checked.answer.choice };
|
|
3364
|
+
}),
|
|
3365
|
+
concurrency
|
|
3366
|
+
);
|
|
3367
|
+
const attempted = outcomes.filter((o) => o.kind !== "threw").map((o) => o.fact);
|
|
3368
|
+
const skipped = outcomes.length - attempted.length;
|
|
3369
|
+
const classifications = outcomes.filter((o) => o.kind === "accepted").map((o) => ({ id: o.fact.id, okf_type: o.okfType }));
|
|
3370
|
+
const invalidCount = outcomes.filter((o) => o.kind === "invalid").length;
|
|
3371
|
+
let typed = 0;
|
|
3372
|
+
let failedValidation = 0;
|
|
3373
|
+
let edgesAdded = 0;
|
|
3374
|
+
let aborted = false;
|
|
3375
|
+
if (attempted.length > 0) {
|
|
3376
|
+
const applied = await this._applyOntologyBackfillBatch(
|
|
3377
|
+
entityId,
|
|
3378
|
+
{ batch: attempted, classifications, ontologyUpdates: void 0 },
|
|
3379
|
+
now
|
|
3380
|
+
);
|
|
3381
|
+
aborted = applied.abortedOntologyOff;
|
|
3382
|
+
if (!aborted) {
|
|
3383
|
+
typed = applied.typed;
|
|
3384
|
+
failedValidation = invalidCount + applied.failedValidation;
|
|
3385
|
+
edgesAdded = applied.edgesAdded;
|
|
3386
|
+
}
|
|
3387
|
+
}
|
|
3388
|
+
const diagBuffer = new DiagnosticBuffer();
|
|
3389
|
+
const diagBase = { entityId, operation: "ontologyBackfill", trigger: "call" };
|
|
3390
|
+
for (const o of outcomes) {
|
|
3391
|
+
if (o.kind === "low_confidence") {
|
|
3392
|
+
diagBuffer.push({ ...diagBase, code: "classification_low_confidence", detail: { factId: o.fact.id, reason: "below_threshold" } });
|
|
3393
|
+
} else if (o.kind === "invalid") {
|
|
3394
|
+
diagBuffer.push({ ...diagBase, code: "classification_invalid", detail: { factId: o.fact.id, reason: o.reason } });
|
|
3395
|
+
} else if (o.kind === "threw") {
|
|
3396
|
+
diagBuffer.push({ ...diagBase, code: "classification_invalid", detail: { factId: o.fact.id, reason: "classify_threw" } });
|
|
3397
|
+
}
|
|
3398
|
+
}
|
|
3399
|
+
diagBuffer.flush(this.options);
|
|
3400
|
+
const counts = await this.entryRepo.countUntypedByEntityId(entityId, recheckCutoff);
|
|
3401
|
+
if (aborted) {
|
|
3402
|
+
return { scanned: 0, typed: 0, failedValidation: 0, edgesAdded: 0, skipped, remaining: 0, deferred: counts.deferred };
|
|
3403
|
+
}
|
|
3404
|
+
this.searchService.evictCache(entityId);
|
|
3405
|
+
return {
|
|
3406
|
+
scanned: candidates.length,
|
|
3407
|
+
typed,
|
|
3408
|
+
failedValidation,
|
|
3409
|
+
edgesAdded,
|
|
3410
|
+
skipped,
|
|
3411
|
+
remaining: counts.eligible,
|
|
3412
|
+
deferred: counts.deferred
|
|
3413
|
+
};
|
|
3414
|
+
}
|
|
3282
3415
|
/**
|
|
3283
3416
|
* Applies one parsed backfill batch in its own transaction. Per-batch rather
|
|
3284
3417
|
* than one transaction for the pass, so mergeEmergentUpdates semantics and
|
|
@@ -3563,7 +3696,6 @@ var RetrievalService = class {
|
|
|
3563
3696
|
const hybridWeight = options?.hybridWeight ?? config?.hybridWeight;
|
|
3564
3697
|
const weight = hybridWeight !== void 0 && !Number.isNaN(hybridWeight) ? Math.max(0, Math.min(1, hybridWeight)) : void 0;
|
|
3565
3698
|
const skipEmbed = weight === 0;
|
|
3566
|
-
const embedFn = this.options.llmProvider.embed;
|
|
3567
3699
|
let facts = [];
|
|
3568
3700
|
let scoreByFactId;
|
|
3569
3701
|
if (maxResults === 0) ; else if (trimmedQuery) {
|
|
@@ -3574,11 +3706,11 @@ var RetrievalService = class {
|
|
|
3574
3706
|
const padLimit = (n) => n >= Number.MAX_SAFE_INTEGER ? n : n + draftPad;
|
|
3575
3707
|
if (scoredEntityIds.length === 0) {
|
|
3576
3708
|
usedEmbed = true;
|
|
3577
|
-
} else if (!skipEmbed &&
|
|
3709
|
+
} else if (!skipEmbed && typeof this.options.llmProvider.embed === "function") {
|
|
3578
3710
|
let rankerShouldRethrow = false;
|
|
3579
3711
|
let pendingRankerFallbackError;
|
|
3580
3712
|
try {
|
|
3581
|
-
const queryVec = await
|
|
3713
|
+
const queryVec = await this.options.llmProvider.embed(trimmedQuery);
|
|
3582
3714
|
if (queryVec.length === 0 || !queryVec.every((v) => typeof v === "number" && isFinite(v))) {
|
|
3583
3715
|
throw new Error(
|
|
3584
3716
|
"embed() returned an empty or non-finite vector. Falling back to keyword search."
|