@tangle-network/agent-knowledge 5.0.4 → 6.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +17 -0
- package/README.md +2 -2
- package/dist/benchmarks/index.d.ts +2 -53
- package/dist/benchmarks/index.js +2 -49
- package/dist/benchmarks-CmW6iORW.js +2718 -0
- package/dist/benchmarks-CmW6iORW.js.map +1 -0
- package/dist/cli.d.ts +1 -1
- package/dist/cli.js +180 -274
- package/dist/cli.js.map +1 -1
- package/dist/ids-DRqPZ42_.js +15 -0
- package/dist/ids-DRqPZ42_.js.map +1 -0
- package/dist/index-CGBctbit.d.ts +857 -0
- package/dist/index-CGBctbit.d.ts.map +1 -0
- package/dist/index-CIW3G4s_.d.ts +680 -0
- package/dist/index-CIW3G4s_.d.ts.map +1 -0
- package/dist/index.d.ts +1671 -1876
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +5836 -6530
- package/dist/index.js.map +1 -1
- package/dist/inspect-D5iarJc2.js +1864 -0
- package/dist/inspect-D5iarJc2.js.map +1 -0
- package/dist/memory/index.d.ts +3 -8
- package/dist/memory/index.js +3 -81
- package/dist/memory-C6KPRhoU.js +4494 -0
- package/dist/memory-C6KPRhoU.js.map +1 -0
- package/dist/search-CP0QtBJZ.js +113 -0
- package/dist/search-CP0QtBJZ.js.map +1 -0
- package/dist/sources/index.d.ts +212 -205
- package/dist/sources/index.d.ts.map +1 -0
- package/dist/sources/index.js +614 -33
- package/dist/sources/index.js.map +1 -1
- package/dist/types-DcCCzreS.d.ts +175 -0
- package/dist/types-DcCCzreS.d.ts.map +1 -0
- package/dist/viz/index.d.ts +23 -22
- package/dist/viz/index.d.ts.map +1 -0
- package/dist/viz/index.js +134 -10
- package/dist/viz/index.js.map +1 -1
- package/docs/results/investment-thesis.md +1 -1
- package/docs/results/research-driving.md +1 -1
- package/docs/{two-agent-research-ab.md → verified-research-ab.md} +2 -2
- package/package.json +22 -11
- package/dist/benchmarks/index.js.map +0 -1
- package/dist/chunk-4PNXQ2NT.js +0 -147
- package/dist/chunk-4PNXQ2NT.js.map +0 -1
- package/dist/chunk-AKYJG2MR.js +0 -2183
- package/dist/chunk-AKYJG2MR.js.map +0 -1
- package/dist/chunk-DQ3PDMDP.js +0 -115
- package/dist/chunk-DQ3PDMDP.js.map +0 -1
- package/dist/chunk-EYIA5PLQ.js +0 -3153
- package/dist/chunk-EYIA5PLQ.js.map +0 -1
- package/dist/chunk-LMR53POQ.js +0 -5437
- package/dist/chunk-LMR53POQ.js.map +0 -1
- package/dist/chunk-MYFM6LKH.js +0 -551
- package/dist/chunk-MYFM6LKH.js.map +0 -1
- package/dist/chunk-YMKHCTS2.js +0 -19
- package/dist/chunk-YMKHCTS2.js.map +0 -1
- package/dist/index-Cf7txrYP.d.ts +0 -790
- package/dist/memory/index.js.map +0 -1
- package/dist/types-6x0OpfW6.d.ts +0 -173
- package/dist/types-BY-xLVw-.d.ts +0 -622
package/dist/index.d.ts
CHANGED
|
@@ -1,54 +1,50 @@
|
|
|
1
|
-
import { S as
|
|
2
|
-
|
|
3
|
-
import {
|
|
4
|
-
import {
|
|
5
|
-
import {
|
|
6
|
-
import {
|
|
7
|
-
import {
|
|
8
|
-
|
|
9
|
-
import {
|
|
10
|
-
|
|
11
|
-
export { INDUSTRY_MEMORY_BENCHMARKS, INDUSTRY_RAG_BENCHMARKS, buildFirstPartyMemoryLifecycleBenchmarkCases, buildIndustryMemoryBenchmarkSmokeCases, buildIndustryRagBenchmarkSmokeCases, buildKnowledgeBenchmarkScenarios, buildRetrievalBenchmarkCasesFromQrels, createInMemoryBenchmarkAdapter, createNoopMemoryBenchmarkAdapter, isKnowledgeMemoryBenchmarkCase, knowledgeBenchmarkJudge, parseKnowledgeBenchmarkJsonl, parseKnowledgeBenchmarkQrels, renderKnowledgeBenchmarkReportMarkdown, respondToIndustryMemoryBenchmarkSmokeCase, respondToIndustryRagBenchmarkSmokeCase, runKnowledgeBenchmarkSuite, runMemoryAdapterBenchmark, scoreKnowledgeBenchmarkArtifact, scoreMemoryBenchmarkArtifact, summarizeKnowledgeBenchmarkCampaign } from './benchmarks/index.js';
|
|
12
|
-
import { KnowledgeFragment } from './sources/index.js';
|
|
13
|
-
export { CornellLiiSelector, CornellLiiSourceOptions, FetchOpts, FragmentProvenance, IRS_DIMENSION_HINTS, IrsPublicationsSourceOptions, KnowledgeSource, MAX_RESPONSE_BYTES, MIN_REQUEST_GAP_MS, POLITE_USER_AGENT, PoliteFetchOptions, PoliteFetchResult, StateSosEntity, StateSosSourceConfig, __resetHttpThrottle, createCornellLiiSource, createIrsPublicationsSource, createStateSosSource, extractLinks, firstMatch, htmlToText, innerHtmlById, looksLikeBlockPage, politeFetch } from './sources/index.js';
|
|
14
|
-
import '@tangle-network/agent-eval/rl';
|
|
15
|
-
|
|
1
|
+
import { S as SourceRegistry, _ as KnowledgeUnit, a as KnowledgeEventType, b as SourceAnchor, c as KnowledgeGraphNode, d as KnowledgeLintFinding, f as KnowledgePage, g as KnowledgeSearchResult, h as KnowledgeRelease, i as KnowledgeEvent, l as KnowledgeId, m as KnowledgeRelation, n as KnowledgeBaseCandidate, o as KnowledgeGraph, p as KnowledgePolicy, r as KnowledgeClaim, s as KnowledgeGraphEdge, t as ClaimRef, u as KnowledgeIndex, v as KnowledgeWriteBlock, x as SourceRecord, y as KnowledgeWriteParseResult } from "./types-DcCCzreS.js";
|
|
2
|
+
import { $ as RetrievalConfig, A as KnowledgeBenchmarkResponder, At as AgentMemoryScope, B as KnowledgeMemoryEvent, Bt as RetrievalHoldoutSessionState, C as KnowledgeBenchmarkArtifact, Ct as createInMemoryBenchmarkAdapter, D as KnowledgeBenchmarkEvaluation, Dt as AgentMemoryContext, E as KnowledgeBenchmarkDistribution, Et as AgentMemoryBranchIsolation, F as KnowledgeBenchmarkSplit, Ft as RetrievalHoldoutCallContext, G as MemoryAdapterBenchmarkCandidate, H as KnowledgeRetrievalBenchmarkCase, I as KnowledgeBenchmarkTaskKind, It as RetrievalHoldoutConfig, J as RunKnowledgeBenchmarkSuiteResult, K as MemoryAdapterBenchmarkRankingRow, L as KnowledgeClaimMatcher, Lt as RetrievalHoldoutEligibleItem, M as KnowledgeBenchmarkSliceSummary, Mt as AgentMemoryWriteInput, N as KnowledgeBenchmarkSource, Nt as AgentMemoryWriteResult, O as KnowledgeBenchmarkFamily, Ot as AgentMemoryHit, P as KnowledgeBenchmarkSpec, Pt as RetrievalHoldoutBypassReason, Q as PartitionRetrievalScenariosOptions, R as KnowledgeMemoryBenchmarkCase, Rt as RetrievalHoldoutEvent, S as KnowledgeAnswerBenchmarkTaskKind, St as acquireAgentMemoryRunLease, T as KnowledgeBenchmarkCaseBase, Tt as AgentMemoryAdapter, U as KnowledgeRetrievalBenchmarkQrel, V as KnowledgeMemoryFactMatcher, W as KnowledgeRetrievalBenchmarkQuery, X as RunMemoryAdapterBenchmarkResult, Y as RunMemoryAdapterBenchmarkOptions, Z as BuildRetrievalEvalDispatchOptions, _ as buildIndustryRagBenchmarkSmokeCases, _t as scoreRetrievalArtifact, a as runKnowledgeBenchmarkSuite, at as RetrievalGoldTarget, b as BuildRetrievalBenchmarkCasesFromQrelsOptions, bt as AgentMemoryRunLease, c as buildRetrievalBenchmarkCasesFromQrels, ct as RetrievalRecallJudgeOptions, d as summarizeKnowledgeBenchmarkCampaign, dt as RetrievedSourceSpan, et as RetrievalEvalArtifact, f as runMemoryAdapterBenchmark, ft as buildRetrievalEvalDispatch, g as buildIndustryMemoryBenchmarkSmokeCases, gt as retrievalRecallJudge, h as buildFirstPartyMemoryLifecycleBenchmarkCases, ht as retrievalConfigSurface, i as renderKnowledgeBenchmarkReportMarkdown, it as RetrievalEvalScenario, j as KnowledgeBenchmarkScenario, jt as AgentMemorySearchOptions, k as KnowledgeBenchmarkReport, kt as AgentMemoryKind, l as parseKnowledgeBenchmarkJsonl, lt as RetrievalScenarioPartitions, m as INDUSTRY_RAG_BENCHMARKS, mt as retrievalConfigFromSurface, n as buildKnowledgeBenchmarkScenarios, nt as RetrievalEvalRetrieverInput, o as scoreKnowledgeBenchmarkArtifact, ot as RetrievalMetricSummary, p as INDUSTRY_MEMORY_BENCHMARKS, pt as partitionRetrievalScenarios, q as RunKnowledgeBenchmarkSuiteOptions, r as knowledgeBenchmarkJudge, rt as RetrievalEvalRetrieverResult, s as scoreMemoryBenchmarkArtifact, st as RetrievalMetricWeights, t as isKnowledgeMemoryBenchmarkCase, tt as RetrievalEvalRetriever, u as parseKnowledgeBenchmarkQrels, ut as RetrievedKnowledgeHit, v as respondToIndustryMemoryBenchmarkSmokeCase, vt as AgentMemoryAcquireRunLease, w as KnowledgeBenchmarkCase, wt as createNoopMemoryBenchmarkAdapter, x as KnowledgeAnswerBenchmarkCase, xt as OwnedAgentMemoryRunLease, y as respondToIndustryRagBenchmarkSmokeCase, yt as AgentMemoryControllerMode, z as KnowledgeMemoryBenchmarkTaskKind, zt as RetrievalHoldoutResult } from "./index-CIW3G4s_.js";
|
|
3
|
+
import { $ as buildAgentMemorySequenceScenarios, A as AgentMemoryFinalPair, At as defaultGetMemoryContext, B as applySessionStickyRetrievalHoldout, C as runBoundedMemoryLifecycle, Ct as AgentMemoryJournalEntry, D as AgentMemoryActivationDriver, Dt as ForkAgentMemoryBranchSnapshotOptions, E as AgentMemoryActivation, Et as CreateAgentMemoryBranchOptions, F as RunAgentMemoryImprovementResult, Ft as SerializedCandidateCodec, G as toOffPolicyTrajectory, H as emitRetrievalHoldoutBypass, I as RetrievalHoldoutOffPolicyOptions, It as jsonCandidateCodec, J as GraphitiToolNames, K as GraphitiMcpClientLike, L as RetrievalHoldoutOffPolicyResult, Lt as jsonObjectCandidateCodec, M as AgentMemoryPromotionDecision, Mt as RunSerializedKnowledgeOptimizationOptions, N as MemoryConfigScenario, Nt as RunSerializedKnowledgeOptimizationResult, O as AgentMemoryDimensionComparison, Ot as createAgentMemoryBranch, P as RunAgentMemoryImprovementOptions, Pt as SerializedCandidate, Q as agentMemorySequenceJudge, R as RetrievalHoldoutSessionSummary, Rt as runSerializedKnowledgeOptimization, S as resolveMemoryCleanupTimeoutMs, St as AgentMemoryBranchSnapshot, T as runAgentMemoryImprovement, Tt as AgentMemoryVisibility, U as resetRetrievalHoldoutRegistry, V as deterministicRng, W as retrievalHoldoutConfigHash, X as graphitiMemoryAdapterIdentity, Y as createGraphitiMemoryAdapter, Z as runAgentMemoryExperiment, _ as AgentMemoryLifecycleTimeoutError, _t as BuildAgentMemorySequencesFromBenchmarkCasesOptions, a as AgentMemoryScopeSchema, at as AgentMemoryExecutionPaidCallInput, b as createMemoryExecutionPool, bt as AgentMemoryBranch, c as createNeo4jAgentMemoryAdapter, ct as AgentMemoryExperimentCandidate, d as Mem0HostedMemoryAdapterOptions, dt as AgentMemorySequence, et as buildAgentMemorySequencesFromBenchmarkCases, f as Mem0MemoryAdapterOptions, ft as AgentMemorySequenceArtifact, g as mem0MemoryAdapterIdentity, gt as AgentMemorySequenceStep, h as createMem0MemoryAdapter, ht as AgentMemorySequenceScenario, i as AgentMemoryKindSchema, it as AgentMemoryExecutionCostReceipt, j as AgentMemoryImprovementRunLease, jt as renderMemoryContext, k as AgentMemoryFinalEvaluation, kt as forkAgentMemoryBranchSnapshot, l as Mem0ClientMode, lt as AgentMemoryExperimentRankingRow, m as Mem0OssMemoryAdapterOptions, mt as AgentMemorySequenceProbeResult, n as memoryWriteResultToSourceRecord, nt as AgentMemoryExecutionContext, o as AgentMemoryWriteInputSchema, ot as AgentMemoryExecutionPaidCallResult, p as Mem0OssClient, pt as AgentMemorySequenceProbe, q as GraphitiMemoryAdapterOptions, r as AgentMemoryHitSchema, rt as AgentMemoryExecutionCostMeter, s as Neo4jAgentMemoryAdapterOptions, st as AgentMemoryExecutionStep, t as memoryHitToSourceRecord, tt as AgentMemoryAttemptEvent, u as Mem0HostedClient, ut as AgentMemoryExperimentRunLease, v as AgentMemoryLifecycleUnsafeError, vt as RunAgentMemoryExperimentOptions, w as sleepForMemoryRecovery, wt as AgentMemorySharingPolicy, x as memoryRecoveryDelayMs, xt as AgentMemoryBranchLifetime, y as DEFAULT_MEMORY_CLEANUP_TIMEOUT_MS, yt as RunAgentMemoryExperimentResult, z as applyRetrievalHoldout, zt as scenarioContentFingerprint } from "./index-CGBctbit.js";
|
|
4
|
+
import { CornellLiiSelector, CornellLiiSourceOptions, FetchOpts, FragmentProvenance, IRS_DIMENSION_HINTS, IrsPublicationsSourceOptions, KnowledgeFragment, KnowledgeSource, MAX_RESPONSE_BYTES, MIN_REQUEST_GAP_MS, POLITE_USER_AGENT, PoliteFetchOptions, PoliteFetchResult, StateSosEntity, StateSosSourceConfig, __resetHttpThrottle, createCornellLiiSource, createIrsPublicationsSource, createStateSosSource, extractLinks, firstMatch, htmlToText, innerHtmlById, looksLikeBlockPage, politeFetch } from "./sources/index.js";
|
|
5
|
+
import { AgentCandidateJsonValue, AgentCandidateKnowledgeRef, AgentImprovementActivation, AgentImprovementActivationResult } from "@tangle-network/agent-interface";
|
|
6
|
+
import { AnalystFinding, AnalystSeverity, ControlEvalResult, ControlRuntimeConfig, DataAcquisitionPlan, DatasetScenario, GateDecision, KnowledgeAcquisitionMode, KnowledgeBundle, KnowledgeFreshness, KnowledgeImportance, KnowledgeReadinessReport, KnowledgeRequirement, KnowledgeRequirementCategory, KnowledgeSensitivity, ReleaseConfidenceScorecard, ReleaseTraceEvidence, RunRecord, UserQuestion } from "@tangle-network/agent-eval";
|
|
7
|
+
import { z } from "zod";
|
|
8
|
+
import "proper-lockfile";
|
|
9
|
+
import { ComparisonCost, DispatchContext, JudgeConfig, OptimizationMethod, Scenario } from "@tangle-network/agent-eval/campaign";
|
|
10
|
+
//#region src/adapters.d.ts
|
|
16
11
|
interface SourceAdapterInput {
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
12
|
+
uri: string;
|
|
13
|
+
bytes?: Uint8Array;
|
|
14
|
+
text?: string;
|
|
15
|
+
metadata?: Record<string, unknown>;
|
|
21
16
|
}
|
|
22
17
|
interface SourceAdapterOutput {
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
18
|
+
title?: string;
|
|
19
|
+
mediaType?: string;
|
|
20
|
+
text?: string;
|
|
21
|
+
anchors?: SourceRecord['anchors'];
|
|
22
|
+
metadata?: Record<string, unknown>;
|
|
28
23
|
}
|
|
29
24
|
interface SourceAdapter {
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
25
|
+
id: string;
|
|
26
|
+
canLoad(input: SourceAdapterInput): boolean;
|
|
27
|
+
load(input: SourceAdapterInput): Promise<SourceAdapterOutput> | SourceAdapterOutput;
|
|
33
28
|
}
|
|
34
29
|
declare const textSourceAdapter: SourceAdapter;
|
|
35
30
|
declare function mediaTypeFor(uri: string): string;
|
|
36
|
-
|
|
31
|
+
//#endregion
|
|
32
|
+
//#region src/eval-readiness.d.ts
|
|
37
33
|
interface KnowledgeReadinessSpec {
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
34
|
+
id: string;
|
|
35
|
+
description: string;
|
|
36
|
+
query: string;
|
|
37
|
+
requiredFor: string[];
|
|
38
|
+
category: KnowledgeRequirementCategory;
|
|
39
|
+
acquisitionMode: KnowledgeAcquisitionMode;
|
|
40
|
+
importance: KnowledgeImportance;
|
|
41
|
+
freshness: KnowledgeFreshness;
|
|
42
|
+
sensitivity: KnowledgeSensitivity;
|
|
43
|
+
confidenceNeeded: number;
|
|
44
|
+
fallbackPolicy?: KnowledgeRequirement['fallbackPolicy'];
|
|
45
|
+
minSources?: number;
|
|
46
|
+
minHits?: number;
|
|
47
|
+
metadata?: Record<string, unknown>;
|
|
52
48
|
}
|
|
53
49
|
/**
|
|
54
50
|
* Defaults applied by `defineReadinessSpec` when the caller omits the field.
|
|
@@ -59,14 +55,14 @@ interface KnowledgeReadinessSpec {
|
|
|
59
55
|
* topic that must reflect today's regulatory state).
|
|
60
56
|
*/
|
|
61
57
|
declare const READINESS_SPEC_DEFAULTS: {
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
58
|
+
readonly category: 'domain_specific';
|
|
59
|
+
readonly acquisitionMode: 'search_web';
|
|
60
|
+
readonly importance: 'high';
|
|
61
|
+
readonly freshness: 'monthly';
|
|
62
|
+
readonly sensitivity: 'public';
|
|
63
|
+
readonly confidenceNeeded: 0.7;
|
|
64
|
+
readonly minSources: 1;
|
|
65
|
+
readonly minHits: 2;
|
|
70
66
|
};
|
|
71
67
|
/**
|
|
72
68
|
* Inputs accepted by `defineReadinessSpec`. The four fields the caller cannot
|
|
@@ -106,58 +102,60 @@ type DefineReadinessSpecInput = Pick<KnowledgeReadinessSpec, 'id' | 'description
|
|
|
106
102
|
*/
|
|
107
103
|
declare function defineReadinessSpec(input: DefineReadinessSpecInput): KnowledgeReadinessSpec;
|
|
108
104
|
interface BuildEvalKnowledgeBundleOptions {
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
105
|
+
taskId: string;
|
|
106
|
+
index: KnowledgeIndex;
|
|
107
|
+
specs: KnowledgeReadinessSpec[];
|
|
108
|
+
userAnswers?: Record<string, string>;
|
|
109
|
+
searchLimit?: number;
|
|
110
|
+
metadata?: Record<string, unknown>;
|
|
111
|
+
now?: Date;
|
|
116
112
|
}
|
|
117
113
|
interface EvalKnowledgeBundleBuildResult {
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
114
|
+
bundle: KnowledgeBundle;
|
|
115
|
+
report: KnowledgeReadinessReport;
|
|
116
|
+
requirements: KnowledgeRequirement[];
|
|
117
|
+
searchResultsByRequirement: Record<string, KnowledgeSearchResult[]>;
|
|
118
|
+
questions: UserQuestion[];
|
|
119
|
+
acquisitionPlans: DataAcquisitionPlan[];
|
|
124
120
|
}
|
|
125
121
|
declare function buildEvalKnowledgeBundle(options: BuildEvalKnowledgeBundleOptions): EvalKnowledgeBundleBuildResult;
|
|
126
|
-
|
|
122
|
+
//#endregion
|
|
123
|
+
//#region src/sources.d.ts
|
|
127
124
|
interface AddSourceOptions {
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
125
|
+
copyIntoRaw?: boolean;
|
|
126
|
+
adapters?: SourceAdapter[];
|
|
127
|
+
now?: () => Date;
|
|
131
128
|
}
|
|
132
129
|
interface AddSourceTextInput {
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
130
|
+
uri: string;
|
|
131
|
+
text: string;
|
|
132
|
+
title?: string;
|
|
133
|
+
mediaType?: string;
|
|
134
|
+
validUntil?: string;
|
|
135
|
+
lastVerifiedAt?: string;
|
|
136
|
+
metadata?: Record<string, unknown>;
|
|
140
137
|
}
|
|
141
138
|
declare function loadSourceRegistry(root: string): Promise<SourceRegistry>;
|
|
142
139
|
declare function writeSourceRegistry(root: string, registry: SourceRegistry): Promise<void>;
|
|
143
140
|
declare function addSourcePath(root: string, sourcePath: string, options?: AddSourceOptions): Promise<SourceRecord[]>;
|
|
144
141
|
declare function addSourceText(root: string, input: AddSourceTextInput, options?: Pick<AddSourceOptions, 'adapters' | 'now'>): Promise<SourceRecord>;
|
|
145
142
|
declare function sourceRegistryPath(root: string): string;
|
|
146
|
-
|
|
143
|
+
//#endregion
|
|
144
|
+
//#region src/verified-research-loop.d.ts
|
|
147
145
|
/**
|
|
148
146
|
* A knowledge gap the loop surfaces from `scoreKnowledgeReadiness`. The worker
|
|
149
147
|
* targets these; the driver folds the unfilled remainder into the worker's next
|
|
150
148
|
* prompt and runs its own gap-fill pass over them.
|
|
151
149
|
*/
|
|
152
150
|
interface KnowledgeGap {
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
151
|
+
/** Readiness-spec id this gap belongs to. */
|
|
152
|
+
id: string;
|
|
153
|
+
/** Human-readable description of what's missing. */
|
|
154
|
+
description: string;
|
|
155
|
+
/** The search query the readiness check ran for this requirement. */
|
|
156
|
+
query: string;
|
|
157
|
+
/** True when the gap blocks readiness (vs. a soft, non-blocking gap). */
|
|
158
|
+
blocking: boolean;
|
|
161
159
|
}
|
|
162
160
|
/** A new source the worker (or driver) discovered and wants to add to the KB. */
|
|
163
161
|
type ResearchSourceProposal = AddSourceTextInput;
|
|
@@ -171,62 +169,62 @@ type ResearchSourceProposal = AddSourceTextInput;
|
|
|
171
169
|
* sources, so a rejected source never reaches the curated pages.
|
|
172
170
|
*/
|
|
173
171
|
interface ResearchContribution {
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
172
|
+
/** Immutable sources to register (the raw evidence). */
|
|
173
|
+
sources?: ResearchSourceProposal[];
|
|
174
|
+
/** Safe write-protocol text producing curated `knowledge/*.md` pages. */
|
|
175
|
+
proposalText?: string;
|
|
176
|
+
/**
|
|
177
|
+
* Build the page write-protocol text FROM the sources the driver accepted —
|
|
178
|
+
* the curated, citing pages the readiness gate searches. Receives the
|
|
179
|
+
* registered `SourceRecord`s (with their assigned ids, so a page's frontmatter
|
|
180
|
+
* `sources:` can cite them). Returns `---FILE: knowledge/...---` block text or
|
|
181
|
+
* `undefined`. Runs after verification, so a page never cites a rejected
|
|
182
|
+
* source. Concatenated after any static `proposalText`.
|
|
183
|
+
*/
|
|
184
|
+
buildPages?: (acceptedSources: SourceRecord[]) => string | undefined;
|
|
185
|
+
/** Free-form research transcript — products can persist this. */
|
|
186
|
+
notes?: string;
|
|
187
|
+
metadata?: Record<string, unknown>;
|
|
190
188
|
}
|
|
191
189
|
/** Context handed to the worker each round. */
|
|
192
190
|
interface WorkerResearchContext {
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
|
|
202
|
-
|
|
191
|
+
root: string;
|
|
192
|
+
goal: string;
|
|
193
|
+
round: number;
|
|
194
|
+
index: KnowledgeIndex;
|
|
195
|
+
/** Gaps the readiness gate currently reports — what the worker should close. */
|
|
196
|
+
gaps: KnowledgeGap[];
|
|
197
|
+
/** Steer text the driver folded in from the previous round's remaining gaps. */
|
|
198
|
+
steer?: string;
|
|
199
|
+
readiness: EvalKnowledgeBundleBuildResult;
|
|
200
|
+
signal?: AbortSignal;
|
|
203
201
|
}
|
|
204
202
|
/** Context handed to the driver's verifier for one candidate source. */
|
|
205
203
|
interface SourceVerificationContext {
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
|
|
204
|
+
root: string;
|
|
205
|
+
goal: string;
|
|
206
|
+
round: number;
|
|
207
|
+
index: KnowledgeIndex;
|
|
208
|
+
gaps: KnowledgeGap[];
|
|
209
|
+
/** Sources already accepted earlier THIS round (in-round dedup). */
|
|
210
|
+
acceptedThisRound: ResearchSourceProposal[];
|
|
211
|
+
signal?: AbortSignal;
|
|
214
212
|
}
|
|
215
213
|
/** A single rejected source plus the reason the driver gave. */
|
|
216
214
|
interface RejectedSource {
|
|
217
|
-
|
|
218
|
-
|
|
215
|
+
source: ResearchSourceProposal;
|
|
216
|
+
reason: string;
|
|
219
217
|
}
|
|
220
218
|
/** Context handed to the driver's gap-fill pass (only when `driverResearches`). */
|
|
221
219
|
interface DriverResearchContext {
|
|
222
|
-
|
|
223
|
-
|
|
224
|
-
|
|
225
|
-
|
|
226
|
-
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
220
|
+
root: string;
|
|
221
|
+
goal: string;
|
|
222
|
+
round: number;
|
|
223
|
+
index: KnowledgeIndex;
|
|
224
|
+
/** Gaps STILL open after the worker's accepted contribution applied. */
|
|
225
|
+
remainingGaps: KnowledgeGap[];
|
|
226
|
+
readiness: EvalKnowledgeBundleBuildResult;
|
|
227
|
+
signal?: AbortSignal;
|
|
230
228
|
}
|
|
231
229
|
/**
|
|
232
230
|
* The differentiated driver role.
|
|
@@ -242,69 +240,69 @@ interface DriverResearchContext {
|
|
|
242
240
|
* next prompt. Defaults to a compact bulleted list when omitted.
|
|
243
241
|
*/
|
|
244
242
|
interface ResearchDriver {
|
|
245
|
-
|
|
246
|
-
|
|
247
|
-
|
|
243
|
+
verifySource(source: ResearchSourceProposal, ctx: SourceVerificationContext): Promise<SourceVerdict> | SourceVerdict;
|
|
244
|
+
research?(ctx: DriverResearchContext): Promise<ResearchContribution> | ResearchContribution;
|
|
245
|
+
foldGaps?(gaps: KnowledgeGap[]): string;
|
|
248
246
|
}
|
|
249
247
|
type SourceVerdict = {
|
|
250
|
-
|
|
248
|
+
accept: true;
|
|
251
249
|
} | {
|
|
252
|
-
|
|
253
|
-
|
|
250
|
+
accept: false;
|
|
251
|
+
reason: string;
|
|
254
252
|
};
|
|
255
253
|
/** The worker: primary research targeting the round's gaps. */
|
|
256
254
|
type ResearchWorker = (ctx: WorkerResearchContext) => Promise<ResearchContribution> | ResearchContribution;
|
|
257
255
|
interface VerifiedResearchLoopOptions {
|
|
258
|
-
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
|
|
264
|
-
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
|
|
269
|
-
|
|
270
|
-
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
|
|
275
|
-
|
|
276
|
-
|
|
277
|
-
|
|
256
|
+
root: string;
|
|
257
|
+
goal: string;
|
|
258
|
+
worker: ResearchWorker;
|
|
259
|
+
driver: ResearchDriver;
|
|
260
|
+
/**
|
|
261
|
+
* When false (default), the driver ONLY verifies + gates — a pure coordinator
|
|
262
|
+
* that contributes no research of its own (the "doesn't participate in the
|
|
263
|
+
* work" mode). When true, the driver also runs its `research` gap-fill pass
|
|
264
|
+
* each round over the gaps the worker left open.
|
|
265
|
+
*/
|
|
266
|
+
driverResearches?: boolean;
|
|
267
|
+
maxRounds?: number;
|
|
268
|
+
actor?: string;
|
|
269
|
+
/** Readiness specs define the gate; an empty list means the loop never gates. */
|
|
270
|
+
readinessSpecs?: KnowledgeReadinessSpec[];
|
|
271
|
+
readinessTaskId?: string;
|
|
272
|
+
readiness?: Omit<BuildEvalKnowledgeBundleOptions, 'taskId' | 'index' | 'specs'>;
|
|
273
|
+
sourceOptions?: Pick<AddSourceOptions, 'adapters' | 'now'>;
|
|
274
|
+
signal?: AbortSignal;
|
|
275
|
+
onRound?: (round: VerifiedResearchRound) => Promise<void> | void;
|
|
278
276
|
}
|
|
279
277
|
interface VerifiedResearchRound {
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
|
|
284
|
-
|
|
285
|
-
|
|
286
|
-
|
|
287
|
-
|
|
288
|
-
|
|
289
|
-
|
|
290
|
-
|
|
291
|
-
|
|
292
|
-
|
|
293
|
-
|
|
294
|
-
|
|
295
|
-
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
|
|
278
|
+
round: number;
|
|
279
|
+
/** Gaps reported at the START of the round (what the worker targeted). */
|
|
280
|
+
gaps: KnowledgeGap[];
|
|
281
|
+
/** Worker sources accepted by the driver and written to the KB. */
|
|
282
|
+
acceptedWorkerSources: SourceRecord[];
|
|
283
|
+
/** Worker sources the driver rejected (with reasons) — never written. */
|
|
284
|
+
rejectedWorkerSources: RejectedSource[];
|
|
285
|
+
/** Sources the driver itself added in its gap-fill pass. */
|
|
286
|
+
driverSources: SourceRecord[];
|
|
287
|
+
/** Curated pages written this round (worker proposal + driver proposal). */
|
|
288
|
+
writtenPages: string[];
|
|
289
|
+
readiness?: EvalKnowledgeBundleBuildResult;
|
|
290
|
+
/** True once the readiness gate reports no blocking gaps. */
|
|
291
|
+
ready: boolean;
|
|
292
|
+
event: KnowledgeEvent;
|
|
293
|
+
notes: {
|
|
294
|
+
worker?: string;
|
|
295
|
+
driver?: string;
|
|
296
|
+
};
|
|
299
297
|
}
|
|
300
298
|
interface VerifiedResearchLoopResult {
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
|
|
307
|
-
|
|
299
|
+
root: string;
|
|
300
|
+
goal: string;
|
|
301
|
+
rounds: number;
|
|
302
|
+
ready: boolean;
|
|
303
|
+
index: KnowledgeIndex;
|
|
304
|
+
readiness?: EvalKnowledgeBundleBuildResult;
|
|
305
|
+
steps: VerifiedResearchRound[];
|
|
308
306
|
}
|
|
309
307
|
/**
|
|
310
308
|
* Two-agent (driver + worker) sibling of `runKnowledgeResearchLoop`.
|
|
@@ -335,67 +333,30 @@ declare function runVerifiedResearchLoop(options: VerifiedResearchLoopOptions):
|
|
|
335
333
|
* driver can compose into `verifySource` (real verifiers can do more).
|
|
336
334
|
*/
|
|
337
335
|
declare function sourceMatchesGaps(source: ResearchSourceProposal, index: KnowledgeIndex, gaps: KnowledgeGap[]): KnowledgeSearchResult[];
|
|
338
|
-
|
|
339
|
-
|
|
340
|
-
/** @deprecated Renamed to {@link VerifiedResearchLoopOptions}. */
|
|
341
|
-
type TwoAgentResearchLoopOptions = VerifiedResearchLoopOptions;
|
|
342
|
-
/** @deprecated Renamed to {@link VerifiedResearchLoopResult}. */
|
|
343
|
-
type TwoAgentResearchLoopResult = VerifiedResearchLoopResult;
|
|
344
|
-
/** @deprecated Renamed to {@link VerifiedResearchRound}. */
|
|
345
|
-
type TwoAgentResearchRound = VerifiedResearchRound;
|
|
346
|
-
|
|
347
|
-
/**
|
|
348
|
-
* Real web-research worker + verifying driver for `runVerifiedResearchLoop`.
|
|
349
|
-
*
|
|
350
|
-
* This is the GENERAL, any-topic implementation behind the two-agent research
|
|
351
|
-
* loop's live arm. Given the open knowledge gaps the readiness gate surfaces,
|
|
352
|
-
* the worker:
|
|
353
|
-
*
|
|
354
|
-
* 1. asks an LLM (glm-5.2 by default) to turn each gap into focused web
|
|
355
|
-
* search queries,
|
|
356
|
-
* 2. runs a REAL web search over the Tangle router (`POST /v1/search` — the
|
|
357
|
-
* same endpoint `tcloud mcp`'s `web_search` tool forwards to), so there is
|
|
358
|
-
* no hardcoded corpus,
|
|
359
|
-
* 3. fetches the top results with the repo's polite, cached `politeFetch` and
|
|
360
|
-
* reduces each page to text with `htmlToText`,
|
|
361
|
-
* 4. proposes the readable, verifiable pages as `ResearchSourceProposal`s plus
|
|
362
|
-
* a `buildPages` that writes citing `knowledge/*.md` pages from the sources
|
|
363
|
-
* the driver accepts.
|
|
364
|
-
*
|
|
365
|
-
* The verifying DRIVER is the differentiated role from the two-agent loop: a
|
|
366
|
-
* second LLM pass that judges each fetched source's on-topic relevance to the
|
|
367
|
-
* goal + open gaps and rejects off-topic / spam / already-covered material. The
|
|
368
|
-
* worker ADDS; the driver GATES. Together they build a cleaner knowledge base
|
|
369
|
-
* than a single agent at the same compute budget.
|
|
370
|
-
*
|
|
371
|
-
* Dependency-free on purpose: it talks to the router over `fetch` directly with
|
|
372
|
-
* the published OpenAI-compatible chat shape and the `/v1/search` shape, so it
|
|
373
|
-
* works whether or not the `tcloud` CLI is installed. Point it at any router by
|
|
374
|
-
* passing `baseUrl`; supply the key via `apiKey` or `TANGLE_API_KEY`.
|
|
375
|
-
*/
|
|
376
|
-
|
|
336
|
+
//#endregion
|
|
337
|
+
//#region src/web-research-worker.d.ts
|
|
377
338
|
/** One live web result, as the router's `/v1/search` returns it. */
|
|
378
339
|
interface WebSearchHit {
|
|
379
|
-
|
|
380
|
-
|
|
381
|
-
|
|
340
|
+
title: string;
|
|
341
|
+
url: string;
|
|
342
|
+
snippet?: string;
|
|
382
343
|
}
|
|
383
344
|
/**
|
|
384
345
|
* The two router capabilities the worker/driver need. Injectable so tests can
|
|
385
346
|
* stub the network; the default talks to the live Tangle router over `fetch`.
|
|
386
347
|
*/
|
|
387
348
|
interface RouterClient {
|
|
388
|
-
|
|
389
|
-
|
|
390
|
-
|
|
391
|
-
|
|
392
|
-
|
|
393
|
-
|
|
394
|
-
|
|
395
|
-
|
|
396
|
-
|
|
397
|
-
|
|
398
|
-
|
|
349
|
+
/** Live web search — returns title/url/snippet hits. */
|
|
350
|
+
search(query: string, opts?: {
|
|
351
|
+
maxResults?: number;
|
|
352
|
+
}): Promise<WebSearchHit[]>;
|
|
353
|
+
/** Chat completion — returns the assistant message's visible text. */
|
|
354
|
+
chat(messages: {
|
|
355
|
+
role: 'system' | 'user';
|
|
356
|
+
content: string;
|
|
357
|
+
}[], maxTokens?: number): Promise<string>;
|
|
358
|
+
/** Cumulative cost (chat + search) since this client was created. */
|
|
359
|
+
usage(): RouterUsage;
|
|
399
360
|
}
|
|
400
361
|
/**
|
|
401
362
|
* Cumulative router cost — the per-arm signal the A/B reports ALONGSIDE quality,
|
|
@@ -405,38 +366,38 @@ interface RouterClient {
|
|
|
405
366
|
* than its "equal passes" budget implies.
|
|
406
367
|
*/
|
|
407
368
|
interface RouterUsage {
|
|
408
|
-
|
|
409
|
-
|
|
410
|
-
|
|
411
|
-
|
|
412
|
-
|
|
413
|
-
|
|
369
|
+
chatCalls: number;
|
|
370
|
+
searchCalls: number;
|
|
371
|
+
promptTokens: number;
|
|
372
|
+
completionTokens: number;
|
|
373
|
+
usd: number;
|
|
374
|
+
wallMs: number;
|
|
414
375
|
}
|
|
415
376
|
interface TangleRouterOptions {
|
|
416
|
-
|
|
417
|
-
|
|
418
|
-
|
|
419
|
-
|
|
420
|
-
|
|
421
|
-
|
|
422
|
-
|
|
423
|
-
|
|
424
|
-
|
|
425
|
-
|
|
426
|
-
|
|
427
|
-
|
|
428
|
-
|
|
429
|
-
|
|
430
|
-
|
|
431
|
-
|
|
432
|
-
|
|
433
|
-
|
|
434
|
-
|
|
377
|
+
/** Router base URL. Defaults to `https://router.tangle.tools/v1`. */
|
|
378
|
+
baseUrl?: string;
|
|
379
|
+
/** Bearer key. Defaults to `process.env.TANGLE_API_KEY`. */
|
|
380
|
+
apiKey?: string;
|
|
381
|
+
/** Chat model id. Defaults to `glm-5.2`. */
|
|
382
|
+
model?: string;
|
|
383
|
+
/** Optional preferred search provider (exa | you | perplexity | …). */
|
|
384
|
+
searchProvider?: string;
|
|
385
|
+
/**
|
|
386
|
+
* Retries on a TRANSIENT upstream status (502/503/504/429) with exponential
|
|
387
|
+
* backoff. Default 4. A 4xx that isn't 429, and a 401, are NOT retried — those
|
|
388
|
+
* are not transient. After the budget is exhausted the call still fails loud
|
|
389
|
+
* with the original `RouterError`, so the fail-closed contract holds; this only
|
|
390
|
+
* stops a single upstream-capacity blip from voiding a whole multi-topic run.
|
|
391
|
+
*/
|
|
392
|
+
maxRetries?: number;
|
|
393
|
+
/** Base backoff in ms (doubled each retry, ±25% jitter). Default 1500. */
|
|
394
|
+
retryBaseMs?: number;
|
|
395
|
+
signal?: AbortSignal;
|
|
435
396
|
}
|
|
436
397
|
/** A small error so a failed router call fails loud rather than returning junk. */
|
|
437
398
|
declare class RouterError extends Error {
|
|
438
|
-
|
|
439
|
-
|
|
399
|
+
readonly status: number;
|
|
400
|
+
constructor(status: number, message: string);
|
|
440
401
|
}
|
|
441
402
|
/**
|
|
442
403
|
* Build a dependency-free Tangle router client over `fetch`. This is the same
|
|
@@ -445,21 +406,21 @@ declare class RouterError extends Error {
|
|
|
445
406
|
*/
|
|
446
407
|
declare function createTangleRouterClient(options?: TangleRouterOptions): RouterClient;
|
|
447
408
|
interface WebResearchWorkerOptions {
|
|
448
|
-
|
|
449
|
-
|
|
450
|
-
|
|
451
|
-
|
|
452
|
-
|
|
453
|
-
|
|
454
|
-
|
|
455
|
-
|
|
456
|
-
|
|
457
|
-
|
|
458
|
-
|
|
459
|
-
|
|
460
|
-
|
|
461
|
-
|
|
462
|
-
|
|
409
|
+
/** Router client. Defaults to a live Tangle router client from env creds. */
|
|
410
|
+
router?: RouterClient;
|
|
411
|
+
router_options?: TangleRouterOptions;
|
|
412
|
+
/** Max search queries the LLM may form per gap. Default 2. */
|
|
413
|
+
queriesPerGap?: number;
|
|
414
|
+
/** Max web results fetched per query. Default 3. */
|
|
415
|
+
resultsPerQuery?: number;
|
|
416
|
+
/** Hard cap on sources proposed per round (across all gaps). Default 6. */
|
|
417
|
+
maxSourcesPerRound?: number;
|
|
418
|
+
/** Disk cache dir for `politeFetch`. Optional; speeds repeat runs. */
|
|
419
|
+
cacheDir?: string;
|
|
420
|
+
/** Minimum readable text length to keep a fetched page. Default 200. */
|
|
421
|
+
minTextChars?: number;
|
|
422
|
+
/** Max chars of page text stored per source (keeps pages bounded). Default 4000. */
|
|
423
|
+
maxTextChars?: number;
|
|
463
424
|
}
|
|
464
425
|
/**
|
|
465
426
|
* The real web-research worker. Conforms to the loop's `ResearchWorker`
|
|
@@ -468,14 +429,14 @@ interface WebResearchWorkerOptions {
|
|
|
468
429
|
*/
|
|
469
430
|
declare function createWebResearchWorker(options?: WebResearchWorkerOptions): ResearchWorker;
|
|
470
431
|
interface VerifyingDriverOptions {
|
|
471
|
-
|
|
472
|
-
|
|
473
|
-
|
|
474
|
-
|
|
475
|
-
|
|
476
|
-
|
|
477
|
-
|
|
478
|
-
|
|
432
|
+
router?: RouterClient;
|
|
433
|
+
router_options?: TangleRouterOptions;
|
|
434
|
+
/**
|
|
435
|
+
* When the LLM verdict can't be parsed, default to REJECT (fail-closed) so a
|
|
436
|
+
* model hiccup never poisons the KB with an unverified source. Set `true` to
|
|
437
|
+
* accept-on-parse-failure only if you have a reason to. Default false.
|
|
438
|
+
*/
|
|
439
|
+
acceptOnParseFailure?: boolean;
|
|
479
440
|
}
|
|
480
441
|
/**
|
|
481
442
|
* The verifying driver: a real LLM pass that judges each candidate source's
|
|
@@ -488,44 +449,8 @@ interface VerifyingDriverOptions {
|
|
|
488
449
|
* judgement, not bookkeeping.
|
|
489
450
|
*/
|
|
490
451
|
declare function createVerifyingResearchDriver(options?: VerifyingDriverOptions): ResearchDriver;
|
|
491
|
-
|
|
492
|
-
|
|
493
|
-
* Adaptive verifier mode for `runVerifiedResearchLoop`.
|
|
494
|
-
*
|
|
495
|
-
* The cost/quality A/B (`docs/results/cost-quality.md`) found the LLM relevance
|
|
496
|
-
* verifier's cleanliness win is dominated by DE-DUPLICATION — which a
|
|
497
|
-
* deterministic content-hash / canonical-URL check captures at ~none of the LLM
|
|
498
|
-
* premium — and that an LLM check only earns its dollar on the off-scope tail.
|
|
499
|
-
* The honest production move it names is: do the cheap deterministic work first,
|
|
500
|
-
* spend the LLM only where it pays. This module is that driver.
|
|
501
|
-
*
|
|
502
|
-
* Per candidate source the adaptive driver runs THREE stages, cheapest first,
|
|
503
|
-
* and stops at the first that decides:
|
|
504
|
-
*
|
|
505
|
-
* 1. DEDUP ($0, no LLM). Reject a source whose CONTENT (normalized-text hash)
|
|
506
|
-
* or whose CANONICAL URL matches one already accepted this round or already
|
|
507
|
-
* in the knowledge base. This is the de-dup the relevance judge was being
|
|
508
|
-
* paid to do; doing it deterministically is free and exact.
|
|
509
|
-
*
|
|
510
|
-
* 2. HEURISTIC TRIAGE ($0, no LLM). For a unique survivor, a cheap host /
|
|
511
|
-
* title / length signal classifies it as clearly-keep, clearly-drop, or
|
|
512
|
-
* AMBIGUOUS. Clear cases are resolved without a model: an authoritative host
|
|
513
|
-
* (arxiv, *.edu, *.gov, official docs) with a substantial readable body is
|
|
514
|
-
* kept; an obvious spam/listicle/marketing title or a too-thin body is
|
|
515
|
-
* dropped. Only genuinely ambiguous survivors fall through.
|
|
516
|
-
*
|
|
517
|
-
* 3. LLM ESCALATION ($, one call). ONLY the ambiguous survivors reach the LLM
|
|
518
|
-
* `verifySource` — the shipped `createVerifyingResearchDriver` relevance
|
|
519
|
-
* judge. This is where the verifier earns its premium: the off-scope tail a
|
|
520
|
-
* cheap rule can't adjudicate.
|
|
521
|
-
*
|
|
522
|
-
* The result is the cost/quality frontier point the doc predicted: most of the
|
|
523
|
-
* cleanliness (dedup + clear drops) at a fraction of the LLM $/calls (only the
|
|
524
|
-
* ambiguous tail pays). It is a real `ResearchDriver` — same contract the
|
|
525
|
-
* two-agent loop already gates on — and reuses `sha256`, the relevance verifier,
|
|
526
|
-
* and the index; it reinvents none of them.
|
|
527
|
-
*/
|
|
528
|
-
|
|
452
|
+
//#endregion
|
|
453
|
+
//#region src/adaptive-driver.d.ts
|
|
529
454
|
/**
|
|
530
455
|
* Canonicalize a URL for duplicate detection: lowercase host, strip a leading
|
|
531
456
|
* `www.`, drop the scheme, the fragment, a trailing slash, and tracking query
|
|
@@ -548,59 +473,59 @@ type DedupReason = 'duplicate-url' | 'duplicate-content';
|
|
|
548
473
|
type TriageClass = 'keep' | 'drop' | 'ambiguous';
|
|
549
474
|
/** One source's adaptive routing decision, for instrumentation and the doc. */
|
|
550
475
|
interface AdaptiveDecision {
|
|
551
|
-
|
|
552
|
-
|
|
553
|
-
|
|
554
|
-
|
|
555
|
-
|
|
556
|
-
|
|
557
|
-
|
|
476
|
+
uri: string;
|
|
477
|
+
/** The stage that decided this source: dedup | heuristic | llm. */
|
|
478
|
+
stage: 'dedup' | 'heuristic' | 'llm';
|
|
479
|
+
accepted: boolean;
|
|
480
|
+
/** The triage class assigned (set once past dedup). */
|
|
481
|
+
triage?: TriageClass;
|
|
482
|
+
reason?: string;
|
|
558
483
|
}
|
|
559
484
|
/** Running tally of where the adaptive driver spent its decisions. */
|
|
560
485
|
interface AdaptiveStats {
|
|
561
|
-
|
|
562
|
-
|
|
563
|
-
|
|
564
|
-
|
|
565
|
-
|
|
566
|
-
|
|
567
|
-
|
|
568
|
-
|
|
569
|
-
|
|
570
|
-
|
|
571
|
-
|
|
572
|
-
|
|
486
|
+
total: number;
|
|
487
|
+
/** Rejected by deterministic dedup (URL or content). $0. */
|
|
488
|
+
dedupRejected: number;
|
|
489
|
+
/** Kept by the cheap heuristic without an LLM call. $0. */
|
|
490
|
+
heuristicKept: number;
|
|
491
|
+
/** Dropped by the cheap heuristic without an LLM call. $0. */
|
|
492
|
+
heuristicDropped: number;
|
|
493
|
+
/** Escalated to the LLM relevance verifier ($ — the only paid stage). */
|
|
494
|
+
llmCalls: number;
|
|
495
|
+
/** Of the escalations, how many the LLM accepted. */
|
|
496
|
+
llmAccepted: number;
|
|
497
|
+
decisions: AdaptiveDecision[];
|
|
573
498
|
}
|
|
574
499
|
interface AdaptiveDriverOptions {
|
|
575
|
-
|
|
576
|
-
|
|
577
|
-
|
|
578
|
-
|
|
579
|
-
|
|
580
|
-
|
|
581
|
-
|
|
582
|
-
|
|
583
|
-
|
|
584
|
-
|
|
585
|
-
|
|
586
|
-
|
|
587
|
-
|
|
588
|
-
|
|
589
|
-
|
|
590
|
-
|
|
591
|
-
|
|
592
|
-
|
|
593
|
-
|
|
594
|
-
|
|
595
|
-
|
|
596
|
-
|
|
597
|
-
|
|
598
|
-
|
|
599
|
-
|
|
600
|
-
|
|
601
|
-
|
|
602
|
-
|
|
603
|
-
|
|
500
|
+
/** Router client for the LLM escalation. Defaults to a live client from env. */
|
|
501
|
+
router?: RouterClient;
|
|
502
|
+
router_options?: TangleRouterOptions;
|
|
503
|
+
/** Passed through to the escalation relevance verifier. */
|
|
504
|
+
verifying?: Pick<VerifyingDriverOptions, 'acceptOnParseFailure'>;
|
|
505
|
+
/**
|
|
506
|
+
* Hosts an authoritative source lives on. A unique survivor on one of these,
|
|
507
|
+
* with a substantial body, is KEPT deterministically (no LLM). Suffix-matched
|
|
508
|
+
* against the canonical host, so `arxiv.org` matches `export.arxiv.org`. The
|
|
509
|
+
* defaults cover papers, official docs, and standards bodies.
|
|
510
|
+
*/
|
|
511
|
+
authoritativeHosts?: string[];
|
|
512
|
+
/**
|
|
513
|
+
* Title/snippet patterns that mark obvious spam / listicle / marketing — a
|
|
514
|
+
* unique survivor matching one is DROPPED deterministically (no LLM).
|
|
515
|
+
*/
|
|
516
|
+
spamPatterns?: RegExp[];
|
|
517
|
+
/**
|
|
518
|
+
* Below this many readable chars a survivor is too thin to be a real reference
|
|
519
|
+
* and is dropped deterministically. Default 400.
|
|
520
|
+
*/
|
|
521
|
+
minBodyChars?: number;
|
|
522
|
+
/**
|
|
523
|
+
* A survivor whose body is at or above this many chars AND on an authoritative
|
|
524
|
+
* host is kept without an LLM call. Default 600.
|
|
525
|
+
*/
|
|
526
|
+
substantialBodyChars?: number;
|
|
527
|
+
/** Receives each routing decision as it is made (for live instrumentation). */
|
|
528
|
+
onDecision?: (decision: AdaptiveDecision) => void;
|
|
604
529
|
}
|
|
605
530
|
/**
|
|
606
531
|
* Classify a UNIQUE survivor (already past dedup) with cheap host/title/length
|
|
@@ -609,18 +534,18 @@ interface AdaptiveDriverOptions {
|
|
|
609
534
|
* with a plausible body, which a host/title rule cannot adjudicate.
|
|
610
535
|
*/
|
|
611
536
|
declare function triageSource(source: ResearchSourceProposal, options: {
|
|
612
|
-
|
|
613
|
-
|
|
614
|
-
|
|
615
|
-
|
|
537
|
+
authoritativeHosts: string[];
|
|
538
|
+
spamPatterns: RegExp[];
|
|
539
|
+
minBodyChars: number;
|
|
540
|
+
substantialBodyChars: number;
|
|
616
541
|
}): {
|
|
617
|
-
|
|
618
|
-
|
|
542
|
+
triage: TriageClass;
|
|
543
|
+
reason: string;
|
|
619
544
|
};
|
|
620
545
|
interface AdaptiveResearchDriver {
|
|
621
|
-
|
|
622
|
-
|
|
623
|
-
|
|
546
|
+
verifySource(source: ResearchSourceProposal, ctx: SourceVerificationContext): Promise<SourceVerdict>;
|
|
547
|
+
/** Live tally of where decisions were spent — the cost/quality instrumentation. */
|
|
548
|
+
stats(): AdaptiveStats;
|
|
624
549
|
}
|
|
625
550
|
/**
|
|
626
551
|
* Build the adaptive verifier. The deterministic stages (dedup + heuristic
|
|
@@ -633,137 +558,141 @@ interface AdaptiveResearchDriver {
|
|
|
633
558
|
* context's `acceptedThisRound` and the KB index. Use one driver per loop run.
|
|
634
559
|
*/
|
|
635
560
|
declare function createAdaptiveResearchDriver(options?: AdaptiveDriverOptions): AdaptiveResearchDriver;
|
|
636
|
-
|
|
561
|
+
//#endregion
|
|
562
|
+
//#region src/rag-optimization.d.ts
|
|
637
563
|
type RagOptimizationConfig = Record<string, AgentCandidateJsonValue>;
|
|
638
564
|
type RagOptimizationBaseOptions = Omit<RunSerializedKnowledgeOptimizationOptions<RagOptimizationConfig, RagAnswerEvalScenario, RagAnswerEvalArtifact>, 'baseline' | 'method' | 'trainScenarios' | 'selectionScenarios' | 'finalScenarios' | 'dispatchCandidate' | 'judges' | 'codec' | 'scenarioFingerprint'>;
|
|
639
565
|
interface RunRagOptimizationOptions extends RagOptimizationBaseOptions {
|
|
640
|
-
|
|
641
|
-
|
|
642
|
-
|
|
643
|
-
|
|
644
|
-
|
|
645
|
-
|
|
646
|
-
|
|
647
|
-
|
|
648
|
-
|
|
649
|
-
|
|
650
|
-
|
|
651
|
-
|
|
652
|
-
|
|
566
|
+
baseline: RagOptimizationConfig;
|
|
567
|
+
method: OptimizationMethod<RagAnswerEvalScenario, RagAnswerEvalArtifact>;
|
|
568
|
+
trainScenarios: readonly RagAnswerEvalScenario[];
|
|
569
|
+
selectionScenarios: readonly RagAnswerEvalScenario[];
|
|
570
|
+
finalScenarios: readonly RagAnswerEvalScenario[];
|
|
571
|
+
run(input: {
|
|
572
|
+
config: RagOptimizationConfig;
|
|
573
|
+
configSurface: string;
|
|
574
|
+
configSurfaceHash: string;
|
|
575
|
+
scenario: RagAnswerEvalScenario;
|
|
576
|
+
context: DispatchContext;
|
|
577
|
+
}): Promise<RagAnswerEvalArtifact>;
|
|
578
|
+
judges?: readonly JudgeConfig<RagAnswerEvalArtifact, RagAnswerEvalScenario>[];
|
|
653
579
|
}
|
|
654
580
|
interface RunRagOptimizationResult extends RunSerializedKnowledgeOptimizationResult<RagOptimizationConfig> {
|
|
655
|
-
|
|
656
|
-
|
|
657
|
-
|
|
658
|
-
|
|
659
|
-
|
|
581
|
+
baselineConfig: RagOptimizationConfig;
|
|
582
|
+
winnerConfig: RagOptimizationConfig;
|
|
583
|
+
trainScenarios: readonly RagAnswerEvalScenario[];
|
|
584
|
+
selectionScenarios: readonly RagAnswerEvalScenario[];
|
|
585
|
+
finalScenarios: readonly RagAnswerEvalScenario[];
|
|
660
586
|
}
|
|
661
587
|
/** Optimizes retrieval and answer behavior together as one serialized RAG configuration. */
|
|
662
588
|
declare function runRagOptimization(options: RunRagOptimizationOptions): Promise<RunRagOptimizationResult>;
|
|
663
|
-
|
|
589
|
+
//#endregion
|
|
590
|
+
//#region src/proposals.d.ts
|
|
664
591
|
interface ApplyWriteBlocksResult {
|
|
665
|
-
|
|
666
|
-
|
|
592
|
+
written: string[];
|
|
593
|
+
warnings: string[];
|
|
667
594
|
}
|
|
668
595
|
declare function applyKnowledgeWriteBlocks(root: string, proposalText: string): Promise<ApplyWriteBlocksResult>;
|
|
669
596
|
declare function applyKnowledgeWriteBlocksFile(root: string, proposalPath: string): Promise<ApplyWriteBlocksResult>;
|
|
670
|
-
|
|
597
|
+
//#endregion
|
|
598
|
+
//#region src/validate.d.ts
|
|
671
599
|
interface ValidateKnowledgeOptions {
|
|
672
|
-
|
|
600
|
+
strict?: boolean;
|
|
673
601
|
}
|
|
674
602
|
interface ValidateKnowledgeResult {
|
|
675
|
-
|
|
676
|
-
|
|
603
|
+
ok: boolean;
|
|
604
|
+
findings: KnowledgeLintFinding[];
|
|
677
605
|
}
|
|
678
606
|
declare function validateKnowledgeIndex(index: KnowledgeIndex, options?: ValidateKnowledgeOptions): ValidateKnowledgeResult;
|
|
679
|
-
|
|
607
|
+
//#endregion
|
|
608
|
+
//#region src/research-loop.d.ts
|
|
680
609
|
interface KnowledgeResearchLoopContext {
|
|
681
|
-
|
|
682
|
-
|
|
683
|
-
|
|
684
|
-
|
|
685
|
-
|
|
686
|
-
|
|
687
|
-
|
|
688
|
-
|
|
689
|
-
|
|
610
|
+
root: string;
|
|
611
|
+
goal: string;
|
|
612
|
+
iteration: number;
|
|
613
|
+
index: KnowledgeIndex;
|
|
614
|
+
lintFindings: KnowledgeLintFinding[];
|
|
615
|
+
validation: ValidateKnowledgeResult;
|
|
616
|
+
readiness?: EvalKnowledgeBundleBuildResult;
|
|
617
|
+
previousSteps: KnowledgeResearchLoopStep[];
|
|
618
|
+
signal?: AbortSignal;
|
|
690
619
|
}
|
|
691
620
|
interface KnowledgeResearchLoopDecision {
|
|
692
|
-
|
|
693
|
-
|
|
694
|
-
|
|
695
|
-
|
|
696
|
-
|
|
697
|
-
|
|
698
|
-
|
|
699
|
-
|
|
700
|
-
|
|
701
|
-
|
|
702
|
-
|
|
703
|
-
|
|
704
|
-
|
|
705
|
-
|
|
706
|
-
|
|
707
|
-
|
|
708
|
-
|
|
709
|
-
|
|
710
|
-
|
|
711
|
-
|
|
712
|
-
|
|
713
|
-
|
|
714
|
-
|
|
715
|
-
|
|
716
|
-
|
|
621
|
+
/**
|
|
622
|
+
* Free-form notes from the researcher. Keep this human-readable; products can
|
|
623
|
+
* store it as the research transcript.
|
|
624
|
+
*/
|
|
625
|
+
notes?: string;
|
|
626
|
+
/**
|
|
627
|
+
* Local files to register as immutable sources before applying proposals.
|
|
628
|
+
*/
|
|
629
|
+
sourcePaths?: string[];
|
|
630
|
+
/**
|
|
631
|
+
* Textual source artifacts discovered by an agent, browser worker, connector,
|
|
632
|
+
* or deep-research process.
|
|
633
|
+
*/
|
|
634
|
+
sourceTexts?: AddSourceTextInput[];
|
|
635
|
+
/**
|
|
636
|
+
* Safe write protocol text. The loop parses and applies only accepted
|
|
637
|
+
* `---FILE: knowledge/...---` blocks.
|
|
638
|
+
*/
|
|
639
|
+
proposalText?: string;
|
|
640
|
+
/**
|
|
641
|
+
* The researcher decides when the wiki is good enough. The loop deliberately
|
|
642
|
+
* does not encode a domain-specific definition of "done".
|
|
643
|
+
*/
|
|
644
|
+
done?: boolean;
|
|
645
|
+
metadata?: Record<string, unknown>;
|
|
717
646
|
}
|
|
718
647
|
interface KnowledgeResearchLoopStep {
|
|
719
|
-
|
|
720
|
-
|
|
721
|
-
|
|
722
|
-
|
|
723
|
-
|
|
724
|
-
|
|
725
|
-
|
|
726
|
-
|
|
727
|
-
|
|
728
|
-
|
|
648
|
+
iteration: number;
|
|
649
|
+
notes?: string;
|
|
650
|
+
addedSources: SourceRecord[];
|
|
651
|
+
applied?: ApplyWriteBlocksResult;
|
|
652
|
+
lintFindings: KnowledgeLintFinding[];
|
|
653
|
+
validation: ValidateKnowledgeResult;
|
|
654
|
+
readiness?: EvalKnowledgeBundleBuildResult;
|
|
655
|
+
event: KnowledgeEvent;
|
|
656
|
+
done: boolean;
|
|
657
|
+
metadata?: Record<string, unknown>;
|
|
729
658
|
}
|
|
730
659
|
interface RunKnowledgeResearchLoopOptions {
|
|
731
|
-
|
|
732
|
-
|
|
733
|
-
|
|
734
|
-
|
|
735
|
-
|
|
736
|
-
|
|
737
|
-
|
|
738
|
-
|
|
739
|
-
|
|
740
|
-
|
|
741
|
-
|
|
742
|
-
|
|
660
|
+
root: string;
|
|
661
|
+
goal: string;
|
|
662
|
+
maxIterations?: number;
|
|
663
|
+
actor?: string;
|
|
664
|
+
strict?: ValidateKnowledgeOptions['strict'];
|
|
665
|
+
readinessSpecs?: KnowledgeReadinessSpec[];
|
|
666
|
+
readinessTaskId?: string;
|
|
667
|
+
readiness?: Omit<BuildEvalKnowledgeBundleOptions, 'taskId' | 'index' | 'specs'>;
|
|
668
|
+
sourceOptions?: Pick<AddSourceOptions, 'adapters' | 'now'>;
|
|
669
|
+
signal?: AbortSignal;
|
|
670
|
+
step(context: KnowledgeResearchLoopContext): Promise<KnowledgeResearchLoopDecision> | KnowledgeResearchLoopDecision;
|
|
671
|
+
onStep?: (step: KnowledgeResearchLoopStep) => Promise<void> | void;
|
|
743
672
|
}
|
|
744
673
|
interface KnowledgeResearchLoopResult {
|
|
745
|
-
|
|
746
|
-
|
|
747
|
-
|
|
748
|
-
|
|
749
|
-
|
|
750
|
-
|
|
751
|
-
|
|
752
|
-
|
|
753
|
-
|
|
674
|
+
root: string;
|
|
675
|
+
goal: string;
|
|
676
|
+
iterations: number;
|
|
677
|
+
done: boolean;
|
|
678
|
+
index: KnowledgeIndex;
|
|
679
|
+
lintFindings: KnowledgeLintFinding[];
|
|
680
|
+
validation: ValidateKnowledgeResult;
|
|
681
|
+
readiness?: EvalKnowledgeBundleBuildResult;
|
|
682
|
+
steps: KnowledgeResearchLoopStep[];
|
|
754
683
|
}
|
|
755
684
|
type KnowledgeControlLoopState = KnowledgeResearchLoopContext;
|
|
756
685
|
type KnowledgeControlLoopAction = KnowledgeResearchLoopDecision;
|
|
757
686
|
type KnowledgeControlLoopActionResult = KnowledgeResearchLoopStep;
|
|
758
687
|
interface KnowledgeControlLoopAdapterOptions {
|
|
759
|
-
|
|
760
|
-
|
|
761
|
-
|
|
762
|
-
|
|
763
|
-
|
|
764
|
-
|
|
765
|
-
|
|
766
|
-
|
|
688
|
+
root: string;
|
|
689
|
+
goal: string;
|
|
690
|
+
actor?: string;
|
|
691
|
+
strict?: ValidateKnowledgeOptions['strict'];
|
|
692
|
+
readinessSpecs?: KnowledgeReadinessSpec[];
|
|
693
|
+
readinessTaskId?: string;
|
|
694
|
+
readiness?: Omit<BuildEvalKnowledgeBundleOptions, 'taskId' | 'index' | 'specs'>;
|
|
695
|
+
sourceOptions?: Pick<AddSourceOptions, 'adapters' | 'now'>;
|
|
767
696
|
}
|
|
768
697
|
type KnowledgeControlLoopAdapter = Pick<ControlRuntimeConfig<KnowledgeControlLoopState, KnowledgeControlLoopAction, KnowledgeControlLoopActionResult, ControlEvalResult>, 'intent' | 'observe' | 'validate' | 'act' | 'shouldStop'>;
|
|
769
698
|
/**
|
|
@@ -773,650 +702,660 @@ type KnowledgeControlLoopAdapter = Pick<ControlRuntimeConfig<KnowledgeControlLoo
|
|
|
773
702
|
*/
|
|
774
703
|
declare function createKnowledgeControlLoopAdapter(options: KnowledgeControlLoopAdapterOptions): KnowledgeControlLoopAdapter;
|
|
775
704
|
declare function runKnowledgeResearchLoop(options: RunKnowledgeResearchLoopOptions): Promise<KnowledgeResearchLoopResult>;
|
|
776
|
-
|
|
705
|
+
//#endregion
|
|
706
|
+
//#region src/retrieval-optimization.d.ts
|
|
777
707
|
type RetrievalOptimizationBaseOptions = Omit<RunSerializedKnowledgeOptimizationOptions<RetrievalConfig, RetrievalEvalScenario, RetrievalEvalArtifact>, 'baseline' | 'method' | 'trainScenarios' | 'selectionScenarios' | 'finalScenarios' | 'dispatchCandidate' | 'judges' | 'codec' | 'scenarioFingerprint'>;
|
|
778
708
|
interface RunRetrievalImprovementLoopOptions extends RetrievalOptimizationBaseOptions {
|
|
779
|
-
|
|
780
|
-
|
|
781
|
-
|
|
782
|
-
|
|
783
|
-
|
|
784
|
-
|
|
785
|
-
|
|
786
|
-
|
|
787
|
-
|
|
788
|
-
|
|
709
|
+
baseline: RetrievalConfig;
|
|
710
|
+
trainScenarios: readonly RetrievalEvalScenario[];
|
|
711
|
+
selectionScenarios: readonly RetrievalEvalScenario[];
|
|
712
|
+
finalScenarios: readonly RetrievalEvalScenario[];
|
|
713
|
+
method: OptimizationMethod<RetrievalEvalScenario, RetrievalEvalArtifact>;
|
|
714
|
+
index?: KnowledgeIndex;
|
|
715
|
+
defaultK?: number;
|
|
716
|
+
retrieve?: RetrievalEvalRetriever;
|
|
717
|
+
judges?: readonly JudgeConfig<RetrievalEvalArtifact, RetrievalEvalScenario>[];
|
|
718
|
+
metricWeights?: RetrievalMetricWeights;
|
|
789
719
|
}
|
|
790
720
|
interface RunRetrievalImprovementLoopResult extends RunSerializedKnowledgeOptimizationResult<RetrievalConfig> {
|
|
791
|
-
|
|
792
|
-
|
|
793
|
-
|
|
794
|
-
|
|
795
|
-
|
|
721
|
+
baselineConfig: RetrievalConfig;
|
|
722
|
+
winnerConfig: RetrievalConfig;
|
|
723
|
+
trainScenarios: readonly RetrievalEvalScenario[];
|
|
724
|
+
selectionScenarios: readonly RetrievalEvalScenario[];
|
|
725
|
+
finalScenarios: readonly RetrievalEvalScenario[];
|
|
796
726
|
}
|
|
797
727
|
declare function runRetrievalImprovementLoop(options: RunRetrievalImprovementLoopOptions): Promise<RunRetrievalImprovementLoopResult>;
|
|
798
|
-
|
|
728
|
+
//#endregion
|
|
729
|
+
//#region src/rag-improvement-loop.d.ts
|
|
799
730
|
type RagKnowledgeImprovementPhase = 'rag-optimization' | 'retrieval-tuning' | 'gap-diagnosis' | 'knowledge-acquisition' | 'knowledge-update' | 'answer-quality' | 'promotion';
|
|
800
731
|
type RagKnowledgeImprovementPhaseStatus = 'completed' | 'skipped' | 'failed';
|
|
801
732
|
type RagGapKind = 'missing-source' | 'stale-source' | 'retrieval-miss' | 'retrieval-noise' | 'chunking-mismatch' | 'missing-multihop-evidence' | 'generator-unsupported-claim' | 'citation-mismatch' | 'incorrect-abstention' | 'unknown';
|
|
802
733
|
type RagGapSeverity = 'info' | 'warning' | 'error' | 'critical';
|
|
803
734
|
interface RagGapFinding {
|
|
804
|
-
|
|
805
|
-
|
|
806
|
-
|
|
807
|
-
|
|
808
|
-
|
|
809
|
-
|
|
735
|
+
id: string;
|
|
736
|
+
kind: RagGapKind;
|
|
737
|
+
severity: RagGapSeverity;
|
|
738
|
+
message: string;
|
|
739
|
+
scenarioId?: string;
|
|
740
|
+
evidence?: Record<string, AgentCandidateJsonValue>;
|
|
810
741
|
}
|
|
811
742
|
interface RagKnowledgeImprovementPhaseResult {
|
|
812
|
-
|
|
813
|
-
|
|
814
|
-
|
|
815
|
-
|
|
816
|
-
|
|
817
|
-
|
|
743
|
+
phase: RagKnowledgeImprovementPhase;
|
|
744
|
+
status: RagKnowledgeImprovementPhaseStatus;
|
|
745
|
+
summary: string;
|
|
746
|
+
startedAt: string;
|
|
747
|
+
finishedAt: string;
|
|
748
|
+
metadata?: Record<string, AgentCandidateJsonValue>;
|
|
818
749
|
}
|
|
819
750
|
type RagOptimizationSelection = Pick<RunRagOptimizationResult, 'methodName' | 'baseline' | 'winner' | 'baselineConfig' | 'winnerConfig'>;
|
|
820
751
|
type RetrievalOptimizationSelection = Pick<RunRetrievalImprovementLoopResult, 'methodName' | 'baseline' | 'winner' | 'baselineConfig' | 'winnerConfig'>;
|
|
821
752
|
interface RagPhaseInputBase {
|
|
822
|
-
|
|
823
|
-
|
|
824
|
-
|
|
825
|
-
|
|
826
|
-
|
|
753
|
+
goal: string;
|
|
754
|
+
phases: readonly RagKnowledgeImprovementPhaseResult[];
|
|
755
|
+
/** Selected candidate only. Adaptive update callbacks run before final scoring starts. */
|
|
756
|
+
optimization?: RagOptimizationSelection;
|
|
757
|
+
signal?: AbortSignal;
|
|
827
758
|
}
|
|
828
759
|
interface RagDiagnosisInput extends RagPhaseInputBase {
|
|
829
|
-
|
|
760
|
+
retrieval?: RetrievalOptimizationSelection;
|
|
830
761
|
}
|
|
831
762
|
interface RagKnowledgeAcquisitionInput extends RagPhaseInputBase {
|
|
832
|
-
|
|
833
|
-
|
|
763
|
+
retrieval?: RetrievalOptimizationSelection;
|
|
764
|
+
findings: readonly RagGapFinding[];
|
|
834
765
|
}
|
|
835
766
|
interface RagKnowledgeUpdateInput extends RagPhaseInputBase {
|
|
836
|
-
|
|
837
|
-
|
|
838
|
-
|
|
767
|
+
retrieval?: RetrievalOptimizationSelection;
|
|
768
|
+
findings: readonly RagGapFinding[];
|
|
769
|
+
acquisition?: KnowledgeResearchLoopDecision;
|
|
839
770
|
}
|
|
840
771
|
interface RagKnowledgeUpdateResult {
|
|
841
|
-
|
|
842
|
-
|
|
843
|
-
|
|
844
|
-
|
|
772
|
+
applied: boolean;
|
|
773
|
+
summary: string;
|
|
774
|
+
research?: KnowledgeResearchLoopResult;
|
|
775
|
+
metadata?: Record<string, AgentCandidateJsonValue>;
|
|
845
776
|
}
|
|
846
777
|
interface RagAnswerQualityInput extends RagPhaseInputBase {
|
|
847
|
-
|
|
848
|
-
|
|
849
|
-
|
|
850
|
-
|
|
778
|
+
retrieval?: RetrievalOptimizationSelection;
|
|
779
|
+
findings: readonly RagGapFinding[];
|
|
780
|
+
acquisition?: KnowledgeResearchLoopDecision;
|
|
781
|
+
knowledgeUpdate?: RagKnowledgeUpdateResult;
|
|
851
782
|
}
|
|
852
783
|
interface RagAnswerQualityResult {
|
|
853
|
-
|
|
854
|
-
|
|
855
|
-
|
|
856
|
-
|
|
857
|
-
|
|
858
|
-
|
|
859
|
-
|
|
860
|
-
|
|
784
|
+
passed: boolean;
|
|
785
|
+
metrics: Record<string, number>;
|
|
786
|
+
finalScenarioIds: readonly string[];
|
|
787
|
+
datasetRef: string;
|
|
788
|
+
evaluatorRef: string;
|
|
789
|
+
cost: ComparisonCost;
|
|
790
|
+
findings?: readonly RagGapFinding[];
|
|
791
|
+
metadata?: Record<string, AgentCandidateJsonValue>;
|
|
861
792
|
}
|
|
862
793
|
interface RagPromotionInput extends RagPhaseInputBase {
|
|
863
|
-
|
|
864
|
-
|
|
865
|
-
|
|
866
|
-
|
|
867
|
-
|
|
868
|
-
|
|
869
|
-
|
|
870
|
-
|
|
871
|
-
|
|
794
|
+
retrieval?: RetrievalOptimizationSelection;
|
|
795
|
+
/** Full final-case result available only to the terminal promotion decision. */
|
|
796
|
+
optimizationComparison?: RunRagOptimizationResult['comparison'];
|
|
797
|
+
/** Full final-case result available only to the terminal promotion decision. */
|
|
798
|
+
retrievalComparison?: RunRetrievalImprovementLoopResult['comparison'];
|
|
799
|
+
findings: readonly RagGapFinding[];
|
|
800
|
+
acquisition?: KnowledgeResearchLoopDecision;
|
|
801
|
+
knowledgeUpdate?: RagKnowledgeUpdateResult;
|
|
802
|
+
answerQuality?: RagAnswerQualityResult;
|
|
872
803
|
}
|
|
873
804
|
interface RagPromotionResult {
|
|
874
|
-
|
|
875
|
-
|
|
876
|
-
|
|
805
|
+
promoted: boolean;
|
|
806
|
+
reason: string;
|
|
807
|
+
metadata?: Record<string, AgentCandidateJsonValue>;
|
|
877
808
|
}
|
|
878
809
|
interface RagKnowledgeResearchOptions extends Omit<RunKnowledgeResearchLoopOptions, 'goal' | 'signal' | 'step'> {
|
|
879
|
-
|
|
880
|
-
|
|
810
|
+
goal?: string;
|
|
811
|
+
step?: RunKnowledgeResearchLoopOptions['step'];
|
|
881
812
|
}
|
|
882
813
|
interface RunRagKnowledgeImprovementLoopOptions {
|
|
883
|
-
|
|
884
|
-
|
|
885
|
-
|
|
886
|
-
|
|
887
|
-
|
|
888
|
-
|
|
889
|
-
|
|
890
|
-
|
|
891
|
-
|
|
892
|
-
|
|
893
|
-
|
|
894
|
-
|
|
895
|
-
|
|
896
|
-
|
|
897
|
-
|
|
898
|
-
|
|
899
|
-
|
|
900
|
-
|
|
901
|
-
|
|
814
|
+
goal: string;
|
|
815
|
+
optimization?: RunRagOptimizationOptions;
|
|
816
|
+
retrieval?: RunRetrievalImprovementLoopOptions;
|
|
817
|
+
diagnose?: (input: RagDiagnosisInput) => MaybePromise$1<readonly RagGapFinding[]>;
|
|
818
|
+
acquireKnowledge?: (input: RagKnowledgeAcquisitionInput) => MaybePromise$1<KnowledgeResearchLoopDecision>;
|
|
819
|
+
knowledgeResearch?: RagKnowledgeResearchOptions;
|
|
820
|
+
updateKnowledge?: (input: RagKnowledgeUpdateInput) => MaybePromise$1<RagKnowledgeUpdateResult>;
|
|
821
|
+
evaluateAnswers?: (input: RagAnswerQualityInput) => MaybePromise$1<RagAnswerQualityResult>;
|
|
822
|
+
/** Maximum total answer-evaluation spend accepted for promotion. */
|
|
823
|
+
answerQualityCostCeiling?: number;
|
|
824
|
+
/**
|
|
825
|
+
* Makes a side-effect-free promotion decision after the library has rejected
|
|
826
|
+
* missing, regressing, unaccounted, or over-budget final evidence.
|
|
827
|
+
*/
|
|
828
|
+
decidePromotion?: (input: RagPromotionInput) => MaybePromise$1<RagPromotionResult>;
|
|
829
|
+
enabledPhases?: readonly RagKnowledgeImprovementPhase[];
|
|
830
|
+
requiredPhases?: readonly RagKnowledgeImprovementPhase[];
|
|
831
|
+
signal?: AbortSignal;
|
|
832
|
+
now?: () => Date;
|
|
902
833
|
}
|
|
903
834
|
interface RunRagKnowledgeImprovementLoopResult {
|
|
904
|
-
|
|
905
|
-
|
|
906
|
-
|
|
907
|
-
|
|
908
|
-
|
|
909
|
-
|
|
910
|
-
|
|
911
|
-
|
|
912
|
-
|
|
835
|
+
goal: string;
|
|
836
|
+
phases: readonly RagKnowledgeImprovementPhaseResult[];
|
|
837
|
+
optimization?: RunRagOptimizationResult;
|
|
838
|
+
retrieval?: RunRetrievalImprovementLoopResult;
|
|
839
|
+
findings: readonly RagGapFinding[];
|
|
840
|
+
acquisition?: KnowledgeResearchLoopDecision;
|
|
841
|
+
knowledgeUpdate?: RagKnowledgeUpdateResult;
|
|
842
|
+
answerQuality?: RagAnswerQualityResult;
|
|
843
|
+
promotion?: RagPromotionResult;
|
|
913
844
|
}
|
|
914
845
|
type MaybePromise$1<T> = T | Promise<T>;
|
|
915
846
|
declare function runRagKnowledgeImprovementLoop(options: RunRagKnowledgeImprovementLoopOptions): Promise<RunRagKnowledgeImprovementLoopResult>;
|
|
916
|
-
|
|
847
|
+
//#endregion
|
|
848
|
+
//#region src/rag-eval/contracts.d.ts
|
|
917
849
|
type RagEvalProvider = 'agent-knowledge' | 'ragas' | 'deepeval' | 'trulens' | 'ragchecker' | 'custom';
|
|
918
850
|
type RagEvalMetricKey = 'context_precision' | 'context_recall' | 'context_relevance' | 'context_sufficiency' | 'faithfulness' | 'groundedness' | 'answer_relevance' | 'answer_correctness' | 'citation_support' | 'abstention' | 'unsupported_answer_rate';
|
|
919
851
|
type RagEvalSlice = 'known-answer' | 'paraphrase' | 'distractor' | 'freshness' | 'multi-source' | 'unanswerable' | 'long-tail' | 'custom';
|
|
920
852
|
interface RagEvalContext {
|
|
921
|
-
|
|
922
|
-
|
|
923
|
-
|
|
924
|
-
|
|
925
|
-
|
|
926
|
-
|
|
927
|
-
|
|
928
|
-
|
|
853
|
+
id: string;
|
|
854
|
+
text: string;
|
|
855
|
+
rank?: number;
|
|
856
|
+
pageId?: string;
|
|
857
|
+
sourceId?: string;
|
|
858
|
+
anchorId?: string;
|
|
859
|
+
stale?: boolean;
|
|
860
|
+
metadata?: Record<string, AgentCandidateJsonValue>;
|
|
929
861
|
}
|
|
930
862
|
interface RagEvalCitation {
|
|
931
|
-
|
|
932
|
-
|
|
933
|
-
|
|
934
|
-
|
|
935
|
-
|
|
936
|
-
|
|
937
|
-
|
|
938
|
-
|
|
863
|
+
id: string;
|
|
864
|
+
claimId?: string;
|
|
865
|
+
contextId?: string;
|
|
866
|
+
pageId?: string;
|
|
867
|
+
sourceId?: string;
|
|
868
|
+
anchorId?: string;
|
|
869
|
+
quote?: string;
|
|
870
|
+
metadata?: Record<string, AgentCandidateJsonValue>;
|
|
939
871
|
}
|
|
940
872
|
interface RagEvalClaim {
|
|
941
|
-
|
|
942
|
-
|
|
943
|
-
|
|
944
|
-
|
|
873
|
+
id: string;
|
|
874
|
+
text: string;
|
|
875
|
+
citationIds?: readonly string[];
|
|
876
|
+
metadata?: Record<string, AgentCandidateJsonValue>;
|
|
945
877
|
}
|
|
946
878
|
interface RagRequiredContext {
|
|
947
|
-
|
|
948
|
-
|
|
949
|
-
|
|
950
|
-
|
|
951
|
-
|
|
879
|
+
id?: string;
|
|
880
|
+
text?: string;
|
|
881
|
+
pageId?: string;
|
|
882
|
+
sourceId?: string;
|
|
883
|
+
anchorId?: string;
|
|
952
884
|
}
|
|
953
885
|
interface RagAnswerEvalScenario extends Scenario {
|
|
954
|
-
|
|
955
|
-
|
|
956
|
-
|
|
957
|
-
|
|
958
|
-
|
|
959
|
-
|
|
960
|
-
|
|
961
|
-
|
|
962
|
-
|
|
963
|
-
|
|
886
|
+
kind: 'rag-answer-eval';
|
|
887
|
+
query: string;
|
|
888
|
+
referenceAnswer?: string;
|
|
889
|
+
expectedClaims?: readonly string[];
|
|
890
|
+
forbiddenClaims?: readonly string[];
|
|
891
|
+
requiredContext?: readonly RagRequiredContext[];
|
|
892
|
+
unanswerable?: boolean;
|
|
893
|
+
requireCitations?: boolean;
|
|
894
|
+
slices?: readonly RagEvalSlice[];
|
|
895
|
+
thresholds?: Partial<Record<RagEvalMetricKey, number>>;
|
|
964
896
|
}
|
|
965
897
|
interface ExternalRagEvalScore {
|
|
966
|
-
|
|
967
|
-
|
|
968
|
-
|
|
969
|
-
|
|
898
|
+
provider: RagEvalProvider | string;
|
|
899
|
+
scores: Record<string, number>;
|
|
900
|
+
reasons?: Record<string, string>;
|
|
901
|
+
metadata?: Record<string, AgentCandidateJsonValue>;
|
|
970
902
|
}
|
|
971
903
|
interface RagAnswerEvalArtifact {
|
|
972
|
-
|
|
973
|
-
|
|
974
|
-
|
|
975
|
-
|
|
976
|
-
|
|
977
|
-
|
|
978
|
-
|
|
979
|
-
|
|
980
|
-
|
|
981
|
-
|
|
904
|
+
query: string;
|
|
905
|
+
answer: string;
|
|
906
|
+
contexts: readonly RagEvalContext[];
|
|
907
|
+
claims?: readonly RagEvalClaim[];
|
|
908
|
+
citations?: readonly RagEvalCitation[];
|
|
909
|
+
abstained?: boolean;
|
|
910
|
+
durationMs?: number;
|
|
911
|
+
costUsd?: number;
|
|
912
|
+
externalScores?: readonly ExternalRagEvalScore[];
|
|
913
|
+
metadata?: Record<string, AgentCandidateJsonValue>;
|
|
982
914
|
}
|
|
983
915
|
interface RagAnswerMetricSummary {
|
|
984
|
-
|
|
985
|
-
|
|
986
|
-
|
|
987
|
-
|
|
988
|
-
|
|
989
|
-
|
|
990
|
-
|
|
991
|
-
|
|
992
|
-
|
|
993
|
-
|
|
994
|
-
|
|
916
|
+
metrics: Record<RagEvalMetricKey, number>;
|
|
917
|
+
composite: number;
|
|
918
|
+
passed: boolean;
|
|
919
|
+
findings: readonly RagGapFinding[];
|
|
920
|
+
claimCount: number;
|
|
921
|
+
supportedClaimCount: number;
|
|
922
|
+
citedClaimCount: number;
|
|
923
|
+
supportedCitationCount: number;
|
|
924
|
+
matchedRequiredContextCount: number;
|
|
925
|
+
requiredContextCount: number;
|
|
926
|
+
providerScores: Record<string, Record<RagEvalMetricKey, number>>;
|
|
995
927
|
}
|
|
996
928
|
interface RagAnswerQualityJudgeOptions {
|
|
997
|
-
|
|
998
|
-
|
|
999
|
-
|
|
1000
|
-
|
|
1001
|
-
|
|
929
|
+
name?: string;
|
|
930
|
+
thresholds?: Partial<Record<RagEvalMetricKey, number>>;
|
|
931
|
+
weights?: Partial<Record<RagEvalMetricKey, number>>;
|
|
932
|
+
externalScorePolicy?: 'prefer-external' | 'deterministic-first';
|
|
933
|
+
minClaimSupport?: number;
|
|
1002
934
|
}
|
|
1003
935
|
interface RagAnswerEvalCase {
|
|
1004
|
-
|
|
1005
|
-
|
|
936
|
+
scenario: RagAnswerEvalScenario;
|
|
937
|
+
artifact: RagAnswerEvalArtifact;
|
|
1006
938
|
}
|
|
1007
939
|
interface RagAnswerQualityHookOptions {
|
|
1008
|
-
|
|
1009
|
-
|
|
1010
|
-
|
|
1011
|
-
|
|
1012
|
-
|
|
1013
|
-
|
|
1014
|
-
|
|
1015
|
-
|
|
1016
|
-
|
|
940
|
+
scenarios: readonly RagAnswerEvalScenario[];
|
|
941
|
+
/** Immutable identity of generation, scoring, models, and external evaluator behavior. */
|
|
942
|
+
evaluatorRef: string;
|
|
943
|
+
/** Return observed spend after all generation and evaluation calls finish. */
|
|
944
|
+
cost: ComparisonCost | (() => MaybePromise<ComparisonCost>);
|
|
945
|
+
run: (scenario: RagAnswerEvalScenario) => MaybePromise<RagAnswerEvalArtifact>;
|
|
946
|
+
externalEvaluator?: (item: RagAnswerEvalCase) => MaybePromise<ExternalRagEvalScore | readonly ExternalRagEvalScore[] | undefined>;
|
|
947
|
+
thresholds?: Partial<Record<RagEvalMetricKey, number>>;
|
|
948
|
+
weights?: Partial<Record<RagEvalMetricKey, number>>;
|
|
1017
949
|
}
|
|
1018
950
|
interface RagCalibrationOptions {
|
|
1019
|
-
|
|
1020
|
-
|
|
1021
|
-
|
|
1022
|
-
|
|
1023
|
-
|
|
1024
|
-
|
|
1025
|
-
|
|
951
|
+
scenario: RagAnswerEvalScenario;
|
|
952
|
+
strong: RagAnswerEvalArtifact;
|
|
953
|
+
weak: RagAnswerEvalArtifact;
|
|
954
|
+
judge?: JudgeConfig<RagAnswerEvalArtifact, RagAnswerEvalScenario>;
|
|
955
|
+
minStrongScore?: number;
|
|
956
|
+
maxWeakScore?: number;
|
|
957
|
+
signal?: AbortSignal;
|
|
1026
958
|
}
|
|
1027
959
|
interface RagCalibrationResult {
|
|
1028
|
-
|
|
1029
|
-
|
|
1030
|
-
|
|
1031
|
-
|
|
960
|
+
passed: boolean;
|
|
961
|
+
strongScore: number;
|
|
962
|
+
weakScore: number;
|
|
963
|
+
gap: number;
|
|
1032
964
|
}
|
|
1033
965
|
interface KnowledgeBaseQualityOptions {
|
|
1034
|
-
|
|
1035
|
-
|
|
1036
|
-
|
|
1037
|
-
|
|
966
|
+
now?: Date;
|
|
967
|
+
strict?: boolean;
|
|
968
|
+
minCitationRate?: number;
|
|
969
|
+
maxStaleSourceRate?: number;
|
|
1038
970
|
}
|
|
1039
971
|
interface KnowledgeBaseQualityReport {
|
|
1040
|
-
|
|
1041
|
-
|
|
1042
|
-
|
|
1043
|
-
|
|
1044
|
-
|
|
1045
|
-
|
|
1046
|
-
|
|
1047
|
-
|
|
1048
|
-
|
|
1049
|
-
|
|
1050
|
-
|
|
1051
|
-
|
|
972
|
+
ok: boolean;
|
|
973
|
+
metrics: {
|
|
974
|
+
page_count: number;
|
|
975
|
+
source_count: number;
|
|
976
|
+
citation_rate: number;
|
|
977
|
+
source_backed_page_rate: number;
|
|
978
|
+
stale_source_rate: number;
|
|
979
|
+
duplicate_source_hash_rate: number;
|
|
980
|
+
lint_error_count: number;
|
|
981
|
+
lint_warning_count: number;
|
|
982
|
+
};
|
|
983
|
+
findings: readonly RagGapFinding[];
|
|
1052
984
|
}
|
|
1053
985
|
type MaybePromise<T> = T | Promise<T>;
|
|
1054
|
-
|
|
986
|
+
//#endregion
|
|
987
|
+
//#region src/rag-eval/calibration.d.ts
|
|
1055
988
|
declare function createRagAnswerQualityHook(options: RagAnswerQualityHookOptions): () => Promise<RagAnswerQualityResult>;
|
|
1056
989
|
declare function calibrateRagAnswerJudge(options: RagCalibrationOptions): Promise<RagCalibrationResult>;
|
|
1057
|
-
|
|
990
|
+
//#endregion
|
|
991
|
+
//#region src/rag-eval/knowledge-base.d.ts
|
|
1058
992
|
declare function scoreKnowledgeBaseIndex(index: KnowledgeIndex, options?: KnowledgeBaseQualityOptions): KnowledgeBaseQualityReport;
|
|
1059
|
-
|
|
993
|
+
//#endregion
|
|
994
|
+
//#region src/rag-eval/providers.d.ts
|
|
1060
995
|
declare function normalizeExternalRagScores(scores: readonly ExternalRagEvalScore[]): Record<string, Record<RagEvalMetricKey, number>>;
|
|
1061
996
|
declare function toRagasEvaluationRows(cases: readonly RagAnswerEvalCase[]): {
|
|
1062
|
-
|
|
1063
|
-
|
|
1064
|
-
|
|
1065
|
-
|
|
1066
|
-
|
|
997
|
+
user_input: string;
|
|
998
|
+
response: string;
|
|
999
|
+
retrieved_contexts: string[];
|
|
1000
|
+
reference: string | undefined;
|
|
1001
|
+
reference_contexts: string[];
|
|
1067
1002
|
}[];
|
|
1068
1003
|
declare function toDeepEvalTestCases(cases: readonly RagAnswerEvalCase[]): {
|
|
1069
|
-
|
|
1070
|
-
|
|
1071
|
-
|
|
1072
|
-
|
|
1073
|
-
|
|
1004
|
+
input: string;
|
|
1005
|
+
actual_output: string;
|
|
1006
|
+
expected_output: string | undefined;
|
|
1007
|
+
retrieval_context: string[];
|
|
1008
|
+
context: string[];
|
|
1074
1009
|
}[];
|
|
1075
1010
|
declare function toTruLensRecords(cases: readonly RagAnswerEvalCase[]): {
|
|
1076
|
-
|
|
1077
|
-
|
|
1078
|
-
|
|
1011
|
+
input: string;
|
|
1012
|
+
output: string;
|
|
1013
|
+
context: string;
|
|
1079
1014
|
}[];
|
|
1080
1015
|
declare function toRagCheckerRecords(cases: readonly RagAnswerEvalCase[]): {
|
|
1081
|
-
|
|
1082
|
-
|
|
1083
|
-
|
|
1084
|
-
|
|
1085
|
-
|
|
1086
|
-
|
|
1087
|
-
|
|
1088
|
-
|
|
1089
|
-
|
|
1016
|
+
query_id: string;
|
|
1017
|
+
query: string;
|
|
1018
|
+
gt_answer: string | undefined;
|
|
1019
|
+
response: string;
|
|
1020
|
+
retrieved_context: {
|
|
1021
|
+
doc_id: string;
|
|
1022
|
+
text: string;
|
|
1023
|
+
}[];
|
|
1024
|
+
claims: string[];
|
|
1090
1025
|
}[];
|
|
1091
|
-
|
|
1026
|
+
//#endregion
|
|
1027
|
+
//#region src/rag-eval/scoring.d.ts
|
|
1092
1028
|
declare function ragAnswerQualityJudge(options?: RagAnswerQualityJudgeOptions): JudgeConfig<RagAnswerEvalArtifact, RagAnswerEvalScenario>;
|
|
1093
1029
|
declare function scoreRagAnswerArtifact(artifact: RagAnswerEvalArtifact, scenario: RagAnswerEvalScenario, options?: RagAnswerQualityJudgeOptions): RagAnswerMetricSummary;
|
|
1094
1030
|
declare function diagnoseRagAnswerFailure(metrics: Record<RagEvalMetricKey, number>, scenario: RagAnswerEvalScenario, thresholds?: Partial<Record<RagEvalMetricKey, number>>): RagGapFinding[];
|
|
1095
|
-
|
|
1031
|
+
//#endregion
|
|
1032
|
+
//#region src/kb-improvement/contracts.d.ts
|
|
1096
1033
|
type KnowledgeImprovementStatus = 'running' | 'candidate-ready' | 'promoted' | 'rejected' | 'blocked';
|
|
1097
1034
|
interface KnowledgeImprovementMetricProvenanceBase {
|
|
1098
|
-
|
|
1099
|
-
|
|
1035
|
+
evaluator: string;
|
|
1036
|
+
version: string;
|
|
1100
1037
|
}
|
|
1101
1038
|
type KnowledgeImprovementMetricProvenance = (KnowledgeImprovementMetricProvenanceBase & {
|
|
1102
|
-
|
|
1039
|
+
method: 'deterministic';
|
|
1103
1040
|
}) | (KnowledgeImprovementMetricProvenanceBase & {
|
|
1104
|
-
|
|
1105
|
-
|
|
1106
|
-
|
|
1041
|
+
method: 'sampled' | 'composite';
|
|
1042
|
+
corpusHash: string;
|
|
1043
|
+
runRecords: RunRecord[];
|
|
1107
1044
|
}) | (KnowledgeImprovementMetricProvenanceBase & {
|
|
1108
|
-
|
|
1109
|
-
|
|
1110
|
-
|
|
1111
|
-
|
|
1045
|
+
method: 'model';
|
|
1046
|
+
model: string;
|
|
1047
|
+
corpusHash: string;
|
|
1048
|
+
runRecords: RunRecord[];
|
|
1112
1049
|
});
|
|
1113
1050
|
interface KnowledgeImprovementMetric {
|
|
1114
|
-
|
|
1115
|
-
|
|
1116
|
-
|
|
1117
|
-
|
|
1118
|
-
|
|
1051
|
+
score: number;
|
|
1052
|
+
passed: boolean;
|
|
1053
|
+
dimensions?: Record<string, number>;
|
|
1054
|
+
notes?: string;
|
|
1055
|
+
provenance: KnowledgeImprovementMetricProvenance;
|
|
1119
1056
|
}
|
|
1120
1057
|
interface KnowledgeImprovementEvaluationInput {
|
|
1121
|
-
|
|
1122
|
-
|
|
1123
|
-
|
|
1124
|
-
|
|
1125
|
-
|
|
1126
|
-
|
|
1127
|
-
|
|
1128
|
-
|
|
1129
|
-
|
|
1130
|
-
|
|
1131
|
-
|
|
1132
|
-
|
|
1133
|
-
|
|
1134
|
-
|
|
1058
|
+
runId: string;
|
|
1059
|
+
iteration: number;
|
|
1060
|
+
root: string;
|
|
1061
|
+
baselineRoot: string;
|
|
1062
|
+
candidateRoot: string;
|
|
1063
|
+
baselineIndex: KnowledgeIndex;
|
|
1064
|
+
candidateIndex: KnowledgeIndex;
|
|
1065
|
+
baseHash: string;
|
|
1066
|
+
candidateHash: string;
|
|
1067
|
+
validation: ValidateKnowledgeResult;
|
|
1068
|
+
readiness?: EvalKnowledgeBundleBuildResult;
|
|
1069
|
+
kbQuality: KnowledgeBaseQualityReport;
|
|
1070
|
+
lifecycle?: RunRagKnowledgeImprovementLoopResult;
|
|
1071
|
+
signal?: AbortSignal;
|
|
1135
1072
|
}
|
|
1136
1073
|
type KnowledgeImprovementEvaluator = (input: KnowledgeImprovementEvaluationInput) => Promise<KnowledgeImprovementMetric> | KnowledgeImprovementMetric;
|
|
1137
1074
|
interface KnowledgeImprovementCandidateRecord {
|
|
1138
|
-
|
|
1139
|
-
|
|
1140
|
-
|
|
1141
|
-
|
|
1142
|
-
|
|
1143
|
-
|
|
1144
|
-
|
|
1145
|
-
|
|
1146
|
-
|
|
1147
|
-
|
|
1148
|
-
|
|
1075
|
+
iteration: number;
|
|
1076
|
+
candidateId: string;
|
|
1077
|
+
baseHash: string;
|
|
1078
|
+
candidateHash?: string;
|
|
1079
|
+
evidenceHash?: string;
|
|
1080
|
+
promotionPlanHash?: string;
|
|
1081
|
+
/** Durable one-way boundary preventing final-case reuse after interruption. */
|
|
1082
|
+
finalEvaluationStartedAt?: string;
|
|
1083
|
+
status: KnowledgeImprovementStatus;
|
|
1084
|
+
createdAt: string;
|
|
1085
|
+
updatedAt: string;
|
|
1149
1086
|
}
|
|
1150
1087
|
interface KnowledgeImprovementRunState {
|
|
1151
|
-
|
|
1152
|
-
|
|
1153
|
-
|
|
1154
|
-
|
|
1155
|
-
|
|
1156
|
-
|
|
1157
|
-
|
|
1158
|
-
|
|
1159
|
-
|
|
1160
|
-
|
|
1161
|
-
|
|
1162
|
-
|
|
1088
|
+
runId: string;
|
|
1089
|
+
root: string;
|
|
1090
|
+
goal: string;
|
|
1091
|
+
implementationRef: string;
|
|
1092
|
+
status: KnowledgeImprovementStatus;
|
|
1093
|
+
baseHash: string;
|
|
1094
|
+
createdAt: string;
|
|
1095
|
+
updatedAt: string;
|
|
1096
|
+
ownerId?: string;
|
|
1097
|
+
candidates: KnowledgeImprovementCandidateRecord[];
|
|
1098
|
+
promotedCandidateId?: string;
|
|
1099
|
+
blockedReason?: string;
|
|
1163
1100
|
}
|
|
1164
1101
|
interface KnowledgeImprovementResult {
|
|
1165
|
-
|
|
1166
|
-
|
|
1167
|
-
|
|
1168
|
-
|
|
1169
|
-
|
|
1170
|
-
|
|
1171
|
-
|
|
1102
|
+
runId: string;
|
|
1103
|
+
state: KnowledgeImprovementRunState;
|
|
1104
|
+
candidate?: KnowledgeImprovementCandidateRecord;
|
|
1105
|
+
evaluation?: KnowledgeImprovementMetric;
|
|
1106
|
+
lifecycle?: RunRagKnowledgeImprovementLoopResult;
|
|
1107
|
+
promoted: boolean;
|
|
1108
|
+
blocked: boolean;
|
|
1172
1109
|
}
|
|
1173
1110
|
type KnowledgeImprovementTarget = 'candidate' | 'baseline';
|
|
1174
1111
|
interface KnowledgeImprovementMutationReceipt {
|
|
1175
|
-
|
|
1176
|
-
|
|
1177
|
-
|
|
1178
|
-
|
|
1179
|
-
|
|
1180
|
-
|
|
1112
|
+
target: KnowledgeImprovementTarget;
|
|
1113
|
+
beforeHash: string;
|
|
1114
|
+
afterHash: string;
|
|
1115
|
+
changed: boolean;
|
|
1116
|
+
transactionId: string | null;
|
|
1117
|
+
recovered: boolean;
|
|
1181
1118
|
}
|
|
1182
1119
|
interface KnowledgeImprovementMutationResult extends KnowledgeImprovementResult {
|
|
1183
|
-
|
|
1184
|
-
|
|
1185
|
-
|
|
1120
|
+
candidate: KnowledgeImprovementCandidateRecord;
|
|
1121
|
+
mutation: KnowledgeImprovementMutationReceipt;
|
|
1122
|
+
activationResult?: AgentImprovementActivationResult;
|
|
1186
1123
|
}
|
|
1187
1124
|
interface KnowledgeImprovementActivationPersistence {
|
|
1188
|
-
|
|
1189
|
-
|
|
1190
|
-
|
|
1191
|
-
|
|
1192
|
-
|
|
1125
|
+
activation: AgentImprovementActivation;
|
|
1126
|
+
attemptedAt: string;
|
|
1127
|
+
identity: string;
|
|
1128
|
+
/** May run again after interruption; keep this deterministic and free of external side effects. */
|
|
1129
|
+
createResult(mutation: KnowledgeImprovementMutationReceipt): Promise<AgentImprovementActivationResult> | AgentImprovementActivationResult;
|
|
1193
1130
|
}
|
|
1194
1131
|
declare const KnowledgeImprovementRunStateSchema: z.ZodObject<{
|
|
1195
|
-
|
|
1196
|
-
|
|
1197
|
-
|
|
1198
|
-
|
|
1132
|
+
runId: z.ZodString;
|
|
1133
|
+
root: z.ZodString;
|
|
1134
|
+
goal: z.ZodString;
|
|
1135
|
+
implementationRef: z.ZodString;
|
|
1136
|
+
status: z.ZodEnum<{
|
|
1137
|
+
blocked: "blocked";
|
|
1138
|
+
"candidate-ready": "candidate-ready";
|
|
1139
|
+
promoted: "promoted";
|
|
1140
|
+
rejected: "rejected";
|
|
1141
|
+
running: "running";
|
|
1142
|
+
}>;
|
|
1143
|
+
baseHash: z.ZodString;
|
|
1144
|
+
createdAt: z.ZodISODateTime;
|
|
1145
|
+
updatedAt: z.ZodISODateTime;
|
|
1146
|
+
ownerId: z.ZodOptional<z.ZodString>;
|
|
1147
|
+
candidates: z.ZodArray<z.ZodObject<{
|
|
1148
|
+
iteration: z.ZodNumber;
|
|
1149
|
+
candidateId: z.ZodString;
|
|
1150
|
+
baseHash: z.ZodString;
|
|
1151
|
+
candidateHash: z.ZodOptional<z.ZodString>;
|
|
1152
|
+
evidenceHash: z.ZodOptional<z.ZodString>;
|
|
1153
|
+
promotionPlanHash: z.ZodOptional<z.ZodString>;
|
|
1154
|
+
finalEvaluationStartedAt: z.ZodOptional<z.ZodISODateTime>;
|
|
1199
1155
|
status: z.ZodEnum<{
|
|
1200
|
-
|
|
1201
|
-
|
|
1202
|
-
|
|
1203
|
-
|
|
1204
|
-
|
|
1156
|
+
blocked: "blocked";
|
|
1157
|
+
"candidate-ready": "candidate-ready";
|
|
1158
|
+
promoted: "promoted";
|
|
1159
|
+
rejected: "rejected";
|
|
1160
|
+
running: "running";
|
|
1205
1161
|
}>;
|
|
1206
|
-
baseHash: z.ZodString;
|
|
1207
1162
|
createdAt: z.ZodISODateTime;
|
|
1208
1163
|
updatedAt: z.ZodISODateTime;
|
|
1209
|
-
|
|
1210
|
-
|
|
1211
|
-
|
|
1212
|
-
candidateId: z.ZodString;
|
|
1213
|
-
baseHash: z.ZodString;
|
|
1214
|
-
candidateHash: z.ZodOptional<z.ZodString>;
|
|
1215
|
-
evidenceHash: z.ZodOptional<z.ZodString>;
|
|
1216
|
-
promotionPlanHash: z.ZodOptional<z.ZodString>;
|
|
1217
|
-
finalEvaluationStartedAt: z.ZodOptional<z.ZodISODateTime>;
|
|
1218
|
-
status: z.ZodEnum<{
|
|
1219
|
-
rejected: "rejected";
|
|
1220
|
-
promoted: "promoted";
|
|
1221
|
-
running: "running";
|
|
1222
|
-
"candidate-ready": "candidate-ready";
|
|
1223
|
-
blocked: "blocked";
|
|
1224
|
-
}>;
|
|
1225
|
-
createdAt: z.ZodISODateTime;
|
|
1226
|
-
updatedAt: z.ZodISODateTime;
|
|
1227
|
-
}, z.core.$strict>>;
|
|
1228
|
-
promotedCandidateId: z.ZodOptional<z.ZodString>;
|
|
1229
|
-
blockedReason: z.ZodOptional<z.ZodString>;
|
|
1164
|
+
}, z.core.$strict>>;
|
|
1165
|
+
promotedCandidateId: z.ZodOptional<z.ZodString>;
|
|
1166
|
+
blockedReason: z.ZodOptional<z.ZodString>;
|
|
1230
1167
|
}, z.core.$strict>;
|
|
1231
1168
|
declare const KnowledgeImprovementEvidenceSchema: z.ZodObject<{
|
|
1232
|
-
|
|
1233
|
-
|
|
1234
|
-
|
|
1235
|
-
|
|
1236
|
-
|
|
1237
|
-
|
|
1238
|
-
|
|
1239
|
-
|
|
1240
|
-
|
|
1241
|
-
|
|
1242
|
-
|
|
1243
|
-
|
|
1244
|
-
|
|
1245
|
-
|
|
1246
|
-
|
|
1247
|
-
|
|
1248
|
-
|
|
1249
|
-
|
|
1250
|
-
|
|
1251
|
-
|
|
1252
|
-
|
|
1253
|
-
|
|
1254
|
-
|
|
1255
|
-
|
|
1256
|
-
|
|
1257
|
-
|
|
1258
|
-
|
|
1259
|
-
|
|
1260
|
-
|
|
1261
|
-
|
|
1262
|
-
|
|
1263
|
-
|
|
1264
|
-
|
|
1265
|
-
|
|
1266
|
-
|
|
1267
|
-
|
|
1268
|
-
|
|
1269
|
-
|
|
1270
|
-
|
|
1271
|
-
|
|
1169
|
+
kind: z.ZodLiteral<"knowledge-improvement-evidence">;
|
|
1170
|
+
runId: z.ZodString;
|
|
1171
|
+
candidateId: z.ZodString;
|
|
1172
|
+
iteration: z.ZodNumber;
|
|
1173
|
+
goalHash: z.ZodString;
|
|
1174
|
+
implementationRef: z.ZodString;
|
|
1175
|
+
baseHash: z.ZodString;
|
|
1176
|
+
candidateHash: z.ZodString;
|
|
1177
|
+
promotionPlanHash: z.ZodString;
|
|
1178
|
+
validation: z.ZodUnknown;
|
|
1179
|
+
readiness: z.ZodNullable<z.ZodUnknown>;
|
|
1180
|
+
kbQuality: z.ZodUnknown;
|
|
1181
|
+
evaluation: z.ZodObject<{
|
|
1182
|
+
score: z.ZodNumber;
|
|
1183
|
+
passed: z.ZodBoolean;
|
|
1184
|
+
dimensions: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodNumber>>;
|
|
1185
|
+
notes: z.ZodOptional<z.ZodString>;
|
|
1186
|
+
provenance: z.ZodDiscriminatedUnion<[z.ZodObject<{
|
|
1187
|
+
evaluator: z.ZodString;
|
|
1188
|
+
version: z.ZodString;
|
|
1189
|
+
method: z.ZodLiteral<"deterministic">;
|
|
1190
|
+
}, z.core.$strict>, z.ZodObject<{
|
|
1191
|
+
evaluator: z.ZodString;
|
|
1192
|
+
version: z.ZodString;
|
|
1193
|
+
method: z.ZodEnum<{
|
|
1194
|
+
composite: "composite";
|
|
1195
|
+
sampled: "sampled";
|
|
1196
|
+
}>;
|
|
1197
|
+
corpusHash: z.ZodString;
|
|
1198
|
+
runRecords: z.ZodArray<z.ZodCustom<RunRecord, RunRecord>>;
|
|
1199
|
+
}, z.core.$strict>, z.ZodObject<{
|
|
1200
|
+
evaluator: z.ZodString;
|
|
1201
|
+
version: z.ZodString;
|
|
1202
|
+
method: z.ZodLiteral<"model">;
|
|
1203
|
+
model: z.ZodString;
|
|
1204
|
+
corpusHash: z.ZodString;
|
|
1205
|
+
runRecords: z.ZodArray<z.ZodCustom<RunRecord, RunRecord>>;
|
|
1206
|
+
}, z.core.$strict>], "method">;
|
|
1207
|
+
}, z.core.$strict>;
|
|
1208
|
+
lifecycle: z.ZodNullable<z.ZodUnknown>;
|
|
1272
1209
|
}, z.core.$strict>;
|
|
1273
1210
|
type KnowledgeImprovementEvidence = z.infer<typeof KnowledgeImprovementEvidenceSchema>;
|
|
1274
1211
|
/** Portable identity of one measured candidate. Paths and mutable run state are deliberately excluded. */
|
|
1275
1212
|
declare const KnowledgeImprovementCandidateRefSchema: z.ZodObject<{
|
|
1276
|
-
|
|
1277
|
-
|
|
1278
|
-
|
|
1279
|
-
|
|
1280
|
-
|
|
1281
|
-
|
|
1282
|
-
|
|
1283
|
-
|
|
1213
|
+
kind: z.ZodLiteral<"knowledge-improvement-candidate">;
|
|
1214
|
+
runId: z.ZodString;
|
|
1215
|
+
candidateId: z.ZodString;
|
|
1216
|
+
goalHash: z.ZodString;
|
|
1217
|
+
baseHash: z.ZodString;
|
|
1218
|
+
candidateHash: z.ZodString;
|
|
1219
|
+
evidenceHash: z.ZodString;
|
|
1220
|
+
promotionPlanHash: z.ZodString;
|
|
1284
1221
|
}, z.core.$strict>;
|
|
1285
1222
|
type KnowledgeImprovementCandidateRef = z.infer<typeof KnowledgeImprovementCandidateRefSchema>;
|
|
1286
1223
|
interface PromoteKnowledgeCandidateOptions {
|
|
1287
|
-
|
|
1288
|
-
|
|
1289
|
-
|
|
1290
|
-
|
|
1291
|
-
|
|
1292
|
-
|
|
1293
|
-
|
|
1224
|
+
root: string;
|
|
1225
|
+
candidate: KnowledgeImprovementCandidateRef;
|
|
1226
|
+
activation?: KnowledgeImprovementActivationPersistence;
|
|
1227
|
+
ownerId?: string;
|
|
1228
|
+
leaseTtlMs?: number;
|
|
1229
|
+
now?: () => Date;
|
|
1230
|
+
onState?: (state: KnowledgeImprovementRunState) => Promise<void> | void;
|
|
1294
1231
|
}
|
|
1295
1232
|
type RestoreKnowledgeCandidateBaselineOptions = PromoteKnowledgeCandidateOptions;
|
|
1296
1233
|
interface LoadKnowledgeImprovementActivationResultOptions {
|
|
1297
|
-
|
|
1298
|
-
|
|
1299
|
-
|
|
1300
|
-
|
|
1234
|
+
root: string;
|
|
1235
|
+
candidate: KnowledgeImprovementCandidateRef;
|
|
1236
|
+
activation: AgentImprovementActivation;
|
|
1237
|
+
identity: string;
|
|
1301
1238
|
}
|
|
1302
1239
|
interface UseKnowledgeImprovementCandidateOptions {
|
|
1303
|
-
|
|
1304
|
-
|
|
1240
|
+
root: string;
|
|
1241
|
+
candidate: KnowledgeImprovementCandidateRef;
|
|
1305
1242
|
}
|
|
1306
1243
|
interface ResolvedKnowledgeImprovementComparisonSnapshot {
|
|
1307
|
-
|
|
1308
|
-
|
|
1244
|
+
root: string;
|
|
1245
|
+
hash: string;
|
|
1309
1246
|
}
|
|
1310
1247
|
interface ResolvedKnowledgeImprovementComparison {
|
|
1311
|
-
|
|
1312
|
-
|
|
1313
|
-
|
|
1314
|
-
|
|
1248
|
+
reference: KnowledgeImprovementCandidateRef;
|
|
1249
|
+
evaluation: KnowledgeImprovementMetric;
|
|
1250
|
+
baseline: ResolvedKnowledgeImprovementComparisonSnapshot;
|
|
1251
|
+
candidate: ResolvedKnowledgeImprovementComparisonSnapshot;
|
|
1315
1252
|
}
|
|
1316
1253
|
interface ResolvedKnowledgeImprovementCandidate {
|
|
1317
|
-
|
|
1318
|
-
|
|
1319
|
-
|
|
1254
|
+
root: string;
|
|
1255
|
+
candidate: KnowledgeImprovementCandidateRef;
|
|
1256
|
+
evaluation: KnowledgeImprovementMetric;
|
|
1320
1257
|
}
|
|
1321
1258
|
interface KnowledgeImprovementRetrievalOptions extends Omit<RunRetrievalImprovementLoopOptions, 'index' | 'runDir'> {
|
|
1322
|
-
|
|
1259
|
+
runDir?: RunRetrievalImprovementLoopOptions['runDir'];
|
|
1323
1260
|
}
|
|
1324
1261
|
type KnowledgeImprovementRagOptimizationRunInput = Parameters<RunRagOptimizationOptions['run']>[0] & {
|
|
1325
|
-
|
|
1326
|
-
|
|
1327
|
-
|
|
1328
|
-
|
|
1329
|
-
|
|
1330
|
-
|
|
1331
|
-
|
|
1332
|
-
|
|
1262
|
+
runId: string;
|
|
1263
|
+
iteration: number;
|
|
1264
|
+
candidateId: string;
|
|
1265
|
+
root: string;
|
|
1266
|
+
baselineRoot: string;
|
|
1267
|
+
candidateRoot: string;
|
|
1268
|
+
candidateIndex: KnowledgeIndex;
|
|
1269
|
+
baseHash: string;
|
|
1333
1270
|
};
|
|
1334
1271
|
interface KnowledgeImprovementRagOptimizationOptions extends Omit<RunRagOptimizationOptions, 'run' | 'runDir'> {
|
|
1335
|
-
|
|
1336
|
-
|
|
1272
|
+
runDir?: RunRagOptimizationOptions['runDir'];
|
|
1273
|
+
run(input: KnowledgeImprovementRagOptimizationRunInput): Promise<RagAnswerEvalArtifact>;
|
|
1337
1274
|
}
|
|
1338
1275
|
interface KnowledgeImprovementUpdateInput extends RagKnowledgeUpdateInput {
|
|
1339
|
-
|
|
1340
|
-
|
|
1341
|
-
|
|
1342
|
-
|
|
1343
|
-
|
|
1344
|
-
|
|
1345
|
-
|
|
1276
|
+
runId: string;
|
|
1277
|
+
iteration: number;
|
|
1278
|
+
candidateId: string;
|
|
1279
|
+
root: string;
|
|
1280
|
+
baselineRoot: string;
|
|
1281
|
+
candidateRoot: string;
|
|
1282
|
+
baseHash: string;
|
|
1346
1283
|
}
|
|
1347
1284
|
type KnowledgeImprovementUpdate = (input: KnowledgeImprovementUpdateInput) => Promise<RagKnowledgeUpdateResult> | RagKnowledgeUpdateResult;
|
|
1348
1285
|
interface KnowledgeImprovementOptions {
|
|
1349
|
-
|
|
1350
|
-
|
|
1351
|
-
|
|
1352
|
-
|
|
1353
|
-
|
|
1354
|
-
|
|
1355
|
-
|
|
1356
|
-
|
|
1357
|
-
|
|
1358
|
-
|
|
1359
|
-
|
|
1360
|
-
|
|
1361
|
-
|
|
1362
|
-
|
|
1363
|
-
|
|
1364
|
-
|
|
1365
|
-
|
|
1366
|
-
|
|
1367
|
-
|
|
1368
|
-
|
|
1369
|
-
|
|
1370
|
-
|
|
1371
|
-
|
|
1372
|
-
|
|
1373
|
-
|
|
1374
|
-
|
|
1375
|
-
|
|
1376
|
-
|
|
1377
|
-
|
|
1378
|
-
|
|
1379
|
-
|
|
1380
|
-
|
|
1381
|
-
|
|
1382
|
-
|
|
1383
|
-
|
|
1384
|
-
|
|
1385
|
-
|
|
1386
|
-
}
|
|
1387
|
-
|
|
1286
|
+
root: string;
|
|
1287
|
+
goal: string;
|
|
1288
|
+
/**
|
|
1289
|
+
* Immutable identity covering callbacks, evaluation policy, models, indexes,
|
|
1290
|
+
* external services, and all other behavior that can affect this run.
|
|
1291
|
+
*/
|
|
1292
|
+
implementationRef: string;
|
|
1293
|
+
runId?: string;
|
|
1294
|
+
ownerId?: string;
|
|
1295
|
+
leaseTtlMs?: number;
|
|
1296
|
+
resume?: boolean;
|
|
1297
|
+
maxCandidates?: number;
|
|
1298
|
+
candidateResearchIterations?: number;
|
|
1299
|
+
strict?: ValidateKnowledgeOptions['strict'];
|
|
1300
|
+
readinessSpecs?: KnowledgeReadinessSpec[];
|
|
1301
|
+
readinessTaskId?: string;
|
|
1302
|
+
readiness?: Omit<BuildEvalKnowledgeBundleOptions, 'taskId' | 'index' | 'specs'>;
|
|
1303
|
+
kbQuality?: KnowledgeBaseQualityOptions;
|
|
1304
|
+
step?: RunKnowledgeResearchLoopOptions['step'];
|
|
1305
|
+
knowledgeResearch?: Omit<RagKnowledgeResearchOptions, 'root'>;
|
|
1306
|
+
ragOptimization?: KnowledgeImprovementRagOptimizationOptions;
|
|
1307
|
+
retrieval?: KnowledgeImprovementRetrievalOptions;
|
|
1308
|
+
diagnose?: NonNullable<RunRagKnowledgeImprovementLoopOptions['diagnose']>;
|
|
1309
|
+
acquireKnowledge?: NonNullable<RunRagKnowledgeImprovementLoopOptions['acquireKnowledge']>;
|
|
1310
|
+
updateKnowledge?: KnowledgeImprovementUpdate;
|
|
1311
|
+
evaluateAnswers?: NonNullable<RunRagKnowledgeImprovementLoopOptions['evaluateAnswers']>;
|
|
1312
|
+
answerQualityCostCeiling?: RunRagKnowledgeImprovementLoopOptions['answerQualityCostCeiling'];
|
|
1313
|
+
decidePromotion?: NonNullable<RunRagKnowledgeImprovementLoopOptions['decidePromotion']>;
|
|
1314
|
+
enabledPhases?: readonly RagKnowledgeImprovementPhase[];
|
|
1315
|
+
requiredPhases?: readonly RagKnowledgeImprovementPhase[];
|
|
1316
|
+
/** Repeatable candidate screening that must not use final cases. */
|
|
1317
|
+
evaluateDevelopment?: KnowledgeImprovementEvaluator;
|
|
1318
|
+
/** Single-use final evaluator. A failure ends the run. */
|
|
1319
|
+
evaluate?: KnowledgeImprovementEvaluator;
|
|
1320
|
+
signal?: AbortSignal;
|
|
1321
|
+
now?: () => Date;
|
|
1322
|
+
onState?: (state: KnowledgeImprovementRunState) => Promise<void> | void;
|
|
1323
|
+
}
|
|
1324
|
+
//#endregion
|
|
1325
|
+
//#region src/kb-improvement/activation.d.ts
|
|
1388
1326
|
/** Load the durable result for one exact activation without changing knowledge or run state. */
|
|
1389
1327
|
declare function loadKnowledgeImprovementActivationResult(options: LoadKnowledgeImprovementActivationResultOptions): Promise<AgentImprovementActivationResult | null>;
|
|
1390
|
-
|
|
1328
|
+
//#endregion
|
|
1329
|
+
//#region src/kb-improvement/optimization.d.ts
|
|
1391
1330
|
type PolicyCandidateOptions = Omit<KnowledgeImprovementOptions, 'root' | 'goal' | 'implementationRef' | 'runId' | 'maxCandidates' | 'step' | 'knowledgeResearch' | 'updateKnowledge'>;
|
|
1392
1331
|
type PolicyOptimizationBaseOptions<TPolicy extends AgentCandidateJsonValue, TScenario extends Scenario, TArtifact> = Omit<RunSerializedKnowledgeOptimizationOptions<TPolicy, TScenario, TArtifact>, 'baseline' | 'method' | 'trainScenarios' | 'selectionScenarios' | 'finalScenarios' | 'executionRef'>;
|
|
1393
1332
|
interface OptimizeKnowledgeBasePolicyOptions<TPolicy extends AgentCandidateJsonValue, TScenario extends Scenario, TArtifact> extends PolicyOptimizationBaseOptions<TPolicy, TScenario, TArtifact> {
|
|
1394
|
-
|
|
1395
|
-
|
|
1396
|
-
|
|
1397
|
-
|
|
1398
|
-
|
|
1399
|
-
|
|
1400
|
-
|
|
1401
|
-
|
|
1402
|
-
|
|
1403
|
-
|
|
1404
|
-
|
|
1405
|
-
|
|
1406
|
-
|
|
1407
|
-
|
|
1408
|
-
|
|
1409
|
-
|
|
1410
|
-
|
|
1411
|
-
|
|
1412
|
-
|
|
1413
|
-
|
|
1414
|
-
|
|
1415
|
-
|
|
1333
|
+
root: string;
|
|
1334
|
+
goal: string;
|
|
1335
|
+
baselinePolicy: TPolicy;
|
|
1336
|
+
method: OptimizationMethod<TScenario, TArtifact>;
|
|
1337
|
+
trainScenarios: readonly TScenario[];
|
|
1338
|
+
selectionScenarios: readonly TScenario[];
|
|
1339
|
+
finalScenarios: readonly TScenario[];
|
|
1340
|
+
/** Commit or content identity for evaluation, applyPolicy, and external dependencies. */
|
|
1341
|
+
policyApplicationRef: string;
|
|
1342
|
+
/** Optional namespace for parallel materialization of the same measured policy. */
|
|
1343
|
+
candidateRunLabel?: string;
|
|
1344
|
+
candidate?: PolicyCandidateOptions;
|
|
1345
|
+
applyPolicy(input: KnowledgeImprovementUpdateInput & {
|
|
1346
|
+
policy: TPolicy;
|
|
1347
|
+
policySurface: string;
|
|
1348
|
+
policySurfaceHash: string;
|
|
1349
|
+
optimizationMethod: string;
|
|
1350
|
+
}): Promise<{
|
|
1351
|
+
applied: boolean;
|
|
1352
|
+
summary: string;
|
|
1353
|
+
metadata?: Record<string, AgentCandidateJsonValue>;
|
|
1354
|
+
}>;
|
|
1416
1355
|
}
|
|
1417
1356
|
interface OptimizeKnowledgeBasePolicyResult<TPolicy extends AgentCandidateJsonValue> {
|
|
1418
|
-
|
|
1419
|
-
|
|
1357
|
+
optimization: RunSerializedKnowledgeOptimizationResult<TPolicy>;
|
|
1358
|
+
improvement: KnowledgeImprovementResult;
|
|
1420
1359
|
}
|
|
1421
1360
|
/**
|
|
1422
1361
|
* Optimizes a serialized KB-maintenance policy, then materializes the selected
|
|
@@ -1424,29 +1363,33 @@ interface OptimizeKnowledgeBasePolicyResult<TPolicy extends AgentCandidateJsonVa
|
|
|
1424
1363
|
*/
|
|
1425
1364
|
declare function optimizeKnowledgeBasePolicy<TPolicy extends AgentCandidateJsonValue, TScenario extends Scenario, TArtifact>(options: OptimizeKnowledgeBasePolicyOptions<TPolicy, TScenario, TArtifact>): Promise<OptimizeKnowledgeBasePolicyResult<TPolicy>>;
|
|
1426
1365
|
type KnowledgePolicyDispatch<TPolicy extends AgentCandidateJsonValue, TScenario extends Scenario, TArtifact> = (input: {
|
|
1427
|
-
|
|
1428
|
-
|
|
1429
|
-
|
|
1430
|
-
|
|
1431
|
-
|
|
1366
|
+
candidate: TPolicy;
|
|
1367
|
+
candidateSurface: string;
|
|
1368
|
+
candidateSurfaceHash: string;
|
|
1369
|
+
scenario: TScenario;
|
|
1370
|
+
context: DispatchContext;
|
|
1432
1371
|
}) => Promise<TArtifact>;
|
|
1433
|
-
|
|
1372
|
+
//#endregion
|
|
1373
|
+
//#region src/kb-improvement/run.d.ts
|
|
1434
1374
|
declare function improveKnowledgeBase(options: KnowledgeImprovementOptions): Promise<KnowledgeImprovementResult>;
|
|
1435
|
-
|
|
1375
|
+
//#endregion
|
|
1376
|
+
//#region src/kb-improvement/state.d.ts
|
|
1436
1377
|
declare function knowledgeImprovementRunId(root: string, goal: string): string;
|
|
1437
1378
|
declare function knowledgeImprovementRunDir(root: string, runId: string): string;
|
|
1438
1379
|
declare function loadKnowledgeImprovementState(root: string, runId: string): Promise<KnowledgeImprovementRunState | null>;
|
|
1439
1380
|
interface KnowledgeImprovementEvent extends Record<string, unknown> {
|
|
1440
|
-
|
|
1441
|
-
|
|
1381
|
+
at: string;
|
|
1382
|
+
type: string;
|
|
1442
1383
|
}
|
|
1443
1384
|
declare function loadKnowledgeImprovementEvents(root: string, runId: string): Promise<KnowledgeImprovementEvent[]>;
|
|
1444
|
-
|
|
1385
|
+
//#endregion
|
|
1386
|
+
//#region src/kb-improvement/transition.d.ts
|
|
1445
1387
|
/** Promote one previously measured candidate without rerunning research or evaluation. */
|
|
1446
1388
|
declare function promoteKnowledgeCandidate(options: PromoteKnowledgeCandidateOptions): Promise<KnowledgeImprovementMutationResult>;
|
|
1447
1389
|
/** Restore the frozen baseline paired with one previously measured candidate. */
|
|
1448
1390
|
declare function restoreKnowledgeCandidateBaseline(options: RestoreKnowledgeCandidateBaselineOptions): Promise<KnowledgeImprovementMutationResult>;
|
|
1449
|
-
|
|
1391
|
+
//#endregion
|
|
1392
|
+
//#region src/kb-improvement/workspace.d.ts
|
|
1450
1393
|
/** Freeze the exact knowledge bytes and measured evidence a later approval may promote. */
|
|
1451
1394
|
declare function knowledgeImprovementCandidateRef(result: Pick<KnowledgeImprovementResult, 'runId' | 'state' | 'candidate'>): KnowledgeImprovementCandidateRef;
|
|
1452
1395
|
/** Use both frozen sides of one measured comparison in isolated, integrity-checked copies. */
|
|
@@ -1454,12 +1397,14 @@ declare function withKnowledgeImprovementComparison<T>(options: UseKnowledgeImpr
|
|
|
1454
1397
|
/** Use the frozen candidate side of one measured comparison. */
|
|
1455
1398
|
declare function withKnowledgeImprovementCandidate<T>(options: UseKnowledgeImprovementCandidateOptions, use: (candidate: ResolvedKnowledgeImprovementCandidate) => Promise<T> | T): Promise<T>;
|
|
1456
1399
|
declare function hashKnowledgeBase(root: string): Promise<string>;
|
|
1457
|
-
|
|
1400
|
+
//#endregion
|
|
1401
|
+
//#region src/agent-candidate.d.ts
|
|
1458
1402
|
/** Convert a measured knowledge candidate into the shared review and execution identity. */
|
|
1459
1403
|
declare function toAgentCandidateKnowledgeRef(candidate: KnowledgeImprovementCandidateRef): AgentCandidateKnowledgeRef;
|
|
1460
1404
|
/** Recover agent-knowledge's candidate identity from the shared contract. */
|
|
1461
1405
|
declare function fromAgentCandidateKnowledgeRef(candidate: AgentCandidateKnowledgeRef): KnowledgeImprovementCandidateRef;
|
|
1462
|
-
|
|
1406
|
+
//#endregion
|
|
1407
|
+
//#region src/changes.d.ts
|
|
1463
1408
|
/**
|
|
1464
1409
|
* Change detection across snapshots of one source's fragments.
|
|
1465
1410
|
*
|
|
@@ -1489,117 +1434,83 @@ declare function fromAgentCandidateKnowledgeRef(candidate: AgentCandidateKnowled
|
|
|
1489
1434
|
*/
|
|
1490
1435
|
type KnowledgeChangeKind = 'added' | 'removed' | 'modified';
|
|
1491
1436
|
interface KnowledgeChange {
|
|
1492
|
-
|
|
1493
|
-
|
|
1494
|
-
|
|
1495
|
-
|
|
1496
|
-
|
|
1497
|
-
|
|
1498
|
-
|
|
1499
|
-
|
|
1500
|
-
|
|
1501
|
-
|
|
1502
|
-
|
|
1503
|
-
|
|
1504
|
-
|
|
1505
|
-
|
|
1506
|
-
|
|
1507
|
-
|
|
1508
|
-
|
|
1509
|
-
|
|
1510
|
-
|
|
1511
|
-
|
|
1512
|
-
|
|
1513
|
-
|
|
1514
|
-
|
|
1515
|
-
|
|
1516
|
-
|
|
1517
|
-
|
|
1437
|
+
/** Source-scoped id (matches `KnowledgeFragment.id`). */
|
|
1438
|
+
fragmentId: string;
|
|
1439
|
+
kind: KnowledgeChangeKind;
|
|
1440
|
+
/**
|
|
1441
|
+
* For `added`: full body of the new fragment.
|
|
1442
|
+
* For `removed`: full body of the prior fragment.
|
|
1443
|
+
* For `modified`: unified-diff-style payload `{ before, after }` body strings.
|
|
1444
|
+
*/
|
|
1445
|
+
diff?: {
|
|
1446
|
+
before?: string;
|
|
1447
|
+
after?: string;
|
|
1448
|
+
};
|
|
1449
|
+
/**
|
|
1450
|
+
* Eval dimensions to re-score. Computed as the union of both fragments'
|
|
1451
|
+
* `dimensionHints`. The eval cron treats this as a set of campaign tags.
|
|
1452
|
+
*/
|
|
1453
|
+
affectedDimensions: string[];
|
|
1454
|
+
/** URL of the affected authority page (from whichever side has it). */
|
|
1455
|
+
url?: string;
|
|
1456
|
+
/**
|
|
1457
|
+
* Source-attested change time. For `modified`, takes the NEXT fragment's
|
|
1458
|
+
* `sourceUpdatedAt`. For `removed`, takes the PRIOR fragment's
|
|
1459
|
+
* `sourceUpdatedAt`. For `added`, takes the NEXT fragment's
|
|
1460
|
+
* `sourceUpdatedAt`. Consumers index changes by this date.
|
|
1461
|
+
*/
|
|
1462
|
+
detectedAt: string;
|
|
1518
1463
|
}
|
|
1519
1464
|
interface DetectChangesResult {
|
|
1520
|
-
|
|
1521
|
-
|
|
1522
|
-
|
|
1523
|
-
|
|
1524
|
-
|
|
1525
|
-
|
|
1526
|
-
|
|
1527
|
-
|
|
1528
|
-
|
|
1465
|
+
changes: KnowledgeChange[];
|
|
1466
|
+
/** Counts by kind — handy for dashboards. */
|
|
1467
|
+
summary: {
|
|
1468
|
+
added: number;
|
|
1469
|
+
removed: number;
|
|
1470
|
+
modified: number;
|
|
1471
|
+
};
|
|
1472
|
+
/** Non-fatal diagnostics (duplicate ids, dropped unverifiable fragments). */
|
|
1473
|
+
warnings: string[];
|
|
1529
1474
|
}
|
|
1530
1475
|
interface DetectChangesOptions {
|
|
1531
|
-
|
|
1532
|
-
|
|
1533
|
-
|
|
1534
|
-
|
|
1535
|
-
|
|
1536
|
-
|
|
1537
|
-
|
|
1538
|
-
|
|
1539
|
-
|
|
1540
|
-
|
|
1541
|
-
|
|
1542
|
-
|
|
1543
|
-
|
|
1476
|
+
/**
|
|
1477
|
+
* When true (default), unverifiable fragments are dropped from both
|
|
1478
|
+
* sides before comparison. Set false ONLY when debugging block-page
|
|
1479
|
+
* issues — comparing against unverifiable content emits false
|
|
1480
|
+
* `removed`/`modified` changes.
|
|
1481
|
+
*/
|
|
1482
|
+
skipUnverifiable?: boolean;
|
|
1483
|
+
/**
|
|
1484
|
+
* When provided, only changes whose `affectedDimensions` intersect this
|
|
1485
|
+
* set are returned. Useful for cron loops that schedule per-dimension
|
|
1486
|
+
* eval campaigns and only care about a subset.
|
|
1487
|
+
*/
|
|
1488
|
+
filterDimensions?: string[];
|
|
1544
1489
|
}
|
|
1545
1490
|
declare function detectChanges(prev: KnowledgeFragment[], next: KnowledgeFragment[], options?: DetectChangesOptions): DetectChangesResult;
|
|
1546
|
-
|
|
1491
|
+
//#endregion
|
|
1492
|
+
//#region src/chunking.d.ts
|
|
1547
1493
|
interface ChunkingOptions {
|
|
1548
|
-
|
|
1549
|
-
|
|
1550
|
-
|
|
1551
|
-
|
|
1494
|
+
targetChars: number;
|
|
1495
|
+
maxChars: number;
|
|
1496
|
+
minChars: number;
|
|
1497
|
+
overlapChars: number;
|
|
1552
1498
|
}
|
|
1553
1499
|
interface KnowledgeChunk {
|
|
1554
|
-
|
|
1555
|
-
|
|
1556
|
-
|
|
1557
|
-
|
|
1558
|
-
|
|
1559
|
-
|
|
1500
|
+
index: number;
|
|
1501
|
+
text: string;
|
|
1502
|
+
headingPath: string;
|
|
1503
|
+
charStart: number;
|
|
1504
|
+
charEnd: number;
|
|
1505
|
+
oversized: boolean;
|
|
1560
1506
|
}
|
|
1561
1507
|
declare function chunkMarkdown(content: string, options?: Partial<ChunkingOptions>): KnowledgeChunk[];
|
|
1562
1508
|
declare function stripFrontmatter(content: string): {
|
|
1563
|
-
|
|
1564
|
-
|
|
1509
|
+
body: string;
|
|
1510
|
+
bodyOffset: number;
|
|
1565
1511
|
};
|
|
1566
|
-
|
|
1567
|
-
|
|
1568
|
-
* Claim-grounding mode for `runVerifiedResearchLoop`.
|
|
1569
|
-
*
|
|
1570
|
-
* The two-agent loop's existing verifier judges a source's on-topic RELEVANCE
|
|
1571
|
-
* (is this page about the goal?). On the topic sets we have measured, its
|
|
1572
|
-
* cleanliness win is dominated by DE-DUPLICATION — which a deterministic
|
|
1573
|
-
* content-hash / canonical-URL check captures at ~none of the LLM premium (see
|
|
1574
|
-
* `docs/results/cost-quality.md`). That makes the LLM verifier look expensive
|
|
1575
|
-
* for what a cheap rule already does.
|
|
1576
|
-
*
|
|
1577
|
-
* Claim-grounding targets a DIFFERENT, harder error band: a citation that is
|
|
1578
|
-
* relevant and unique but **misattributed** — the page is on-topic, the URL is
|
|
1579
|
-
* real, yet the specific CLAIM the source is cited for does NOT actually appear
|
|
1580
|
-
* in the page. This is the citation-fabrication failure mode of LLM research:
|
|
1581
|
-
* the model writes a plausible sentence and hangs a real URL off it that never
|
|
1582
|
-
* says any such thing. Neither de-dup nor a relevance judge catches it (both can
|
|
1583
|
-
* pass a misattributed-but-on-topic page); only checking the claim against the
|
|
1584
|
-
* fetched text does.
|
|
1585
|
-
*
|
|
1586
|
-
* The check is EXECUTABLE GROUND TRUTH, not another LLM opinion: the worker
|
|
1587
|
-
* attaches the specific claim it is citing the source for, and the verifier
|
|
1588
|
-
* tests whether that claim is PRESENT (verbatim, normalized, or as a sufficient
|
|
1589
|
-
* content-word overlap / close paraphrase) in the `htmlToText` output of the
|
|
1590
|
-
* page the worker actually fetched. A claim that is not grounded is rejected as
|
|
1591
|
-
* misattributed. Because the oracle is deterministic text presence — not a model
|
|
1592
|
-
* call — it is a deployable, non-oracle verifier: it can run in production with
|
|
1593
|
-
* zero inference cost, OR be composed with the LLM relevance verifier so the
|
|
1594
|
-
* loop rejects BOTH off-topic AND misattributed sources.
|
|
1595
|
-
*
|
|
1596
|
-
* This module is content-free and any-topic: it adds (1) a way for a proposal to
|
|
1597
|
-
* carry the claim it is cited for, (2) the `groundClaimInText` oracle, and (3) a
|
|
1598
|
-
* `ResearchDriver` that gates on grounding. It composes the existing
|
|
1599
|
-
* `ResearchDriver` / `ResearchSourceProposal` contracts and the shipped
|
|
1600
|
-
* `htmlToText`; it reinvents none of them.
|
|
1601
|
-
*/
|
|
1602
|
-
|
|
1512
|
+
//#endregion
|
|
1513
|
+
//#region src/claim-grounding.d.ts
|
|
1603
1514
|
/**
|
|
1604
1515
|
* Metadata key under which a proposal carries the specific claim it is cited
|
|
1605
1516
|
* for. The worker sets `metadata[citedClaimKey] = '<the claim>'`; the
|
|
@@ -1611,31 +1522,31 @@ declare function citedClaimOf(source: ResearchSourceProposal): string | undefine
|
|
|
1611
1522
|
/** Attach a cited claim to a proposal (immutably returns a new proposal). */
|
|
1612
1523
|
declare function withCitedClaim(source: ResearchSourceProposal, claim: string): ResearchSourceProposal;
|
|
1613
1524
|
interface GroundingResult {
|
|
1614
|
-
|
|
1615
|
-
|
|
1616
|
-
|
|
1617
|
-
|
|
1618
|
-
|
|
1619
|
-
|
|
1620
|
-
|
|
1621
|
-
|
|
1622
|
-
|
|
1623
|
-
|
|
1624
|
-
|
|
1525
|
+
/** True when the claim is sufficiently present in the page text. */
|
|
1526
|
+
grounded: boolean;
|
|
1527
|
+
/** How the claim matched (or why it didn't). For audit/notes. */
|
|
1528
|
+
mode: 'verbatim' | 'normalized' | 'overlap' | 'absent' | 'empty-claim' | 'empty-text';
|
|
1529
|
+
/**
|
|
1530
|
+
* Fraction of the claim's content words found in the page text. 1 for a
|
|
1531
|
+
* verbatim/normalized hit; the measured overlap otherwise.
|
|
1532
|
+
*/
|
|
1533
|
+
overlap: number;
|
|
1534
|
+
/** Content words present in the claim but NOT in the page text. */
|
|
1535
|
+
missingWords: string[];
|
|
1625
1536
|
}
|
|
1626
1537
|
interface GroundClaimOptions {
|
|
1627
|
-
|
|
1628
|
-
|
|
1629
|
-
|
|
1630
|
-
|
|
1631
|
-
|
|
1632
|
-
|
|
1633
|
-
|
|
1634
|
-
|
|
1635
|
-
|
|
1636
|
-
|
|
1637
|
-
|
|
1638
|
-
|
|
1538
|
+
/**
|
|
1539
|
+
* Minimum fraction of the claim's content words that must appear in the page
|
|
1540
|
+
* text to count as a close paraphrase when there is no verbatim/normalized
|
|
1541
|
+
* hit. Default 0.7 — a high bar, because a misattribution is exactly a claim
|
|
1542
|
+
* whose specific words the page does not contain.
|
|
1543
|
+
*/
|
|
1544
|
+
minOverlap?: number;
|
|
1545
|
+
/**
|
|
1546
|
+
* Content words shorter than this are ignored (drops "the", "of", "is", …)
|
|
1547
|
+
* and never count toward overlap. Default 3.
|
|
1548
|
+
*/
|
|
1549
|
+
minWordLength?: number;
|
|
1639
1550
|
}
|
|
1640
1551
|
/**
|
|
1641
1552
|
* THE ORACLE. Is `claim` grounded in `pageText` (the `htmlToText` output of the
|
|
@@ -1654,21 +1565,21 @@ interface GroundClaimOptions {
|
|
|
1654
1565
|
*/
|
|
1655
1566
|
declare function groundClaimInText(claim: string, pageText: string, options?: GroundClaimOptions): GroundingResult;
|
|
1656
1567
|
interface ClaimGroundingDriverOptions extends GroundClaimOptions {
|
|
1657
|
-
|
|
1658
|
-
|
|
1659
|
-
|
|
1660
|
-
|
|
1661
|
-
|
|
1662
|
-
|
|
1663
|
-
|
|
1664
|
-
|
|
1665
|
-
|
|
1666
|
-
|
|
1667
|
-
|
|
1668
|
-
|
|
1669
|
-
|
|
1670
|
-
|
|
1671
|
-
|
|
1568
|
+
/**
|
|
1569
|
+
* Optional second verifier to compose AFTER grounding passes. When set, a
|
|
1570
|
+
* source must BOTH ground its claim AND pass this verifier (e.g. the LLM
|
|
1571
|
+
* relevance driver's `verifySource`). Lets the loop reject off-topic AND
|
|
1572
|
+
* misattributed sources in one driver. Omit for the pure, zero-inference
|
|
1573
|
+
* grounding gate.
|
|
1574
|
+
*/
|
|
1575
|
+
relevanceVerifier?: (source: ResearchSourceProposal, ctx: SourceVerificationContext) => Promise<SourceVerdict> | SourceVerdict;
|
|
1576
|
+
/**
|
|
1577
|
+
* What to do when a proposal carries NO cited claim. `'reject'` (default) is
|
|
1578
|
+
* fail-closed: in claim-grounding mode every source must declare what it is
|
|
1579
|
+
* cited for, so an un-annotated source is treated as ungrounded. `'accept'`
|
|
1580
|
+
* lets unannotated sources through to the relevance verifier, if present.
|
|
1581
|
+
*/
|
|
1582
|
+
onMissingClaim?: 'reject' | 'accept';
|
|
1672
1583
|
}
|
|
1673
1584
|
/**
|
|
1674
1585
|
* A `ResearchDriver`-shaped verifier (just the `verifySource` arm) that gates on
|
|
@@ -1681,10 +1592,10 @@ interface ClaimGroundingDriverOptions extends GroundClaimOptions {
|
|
|
1681
1592
|
*/
|
|
1682
1593
|
declare function createClaimGroundingVerifier(options?: ClaimGroundingDriverOptions): (source: ResearchSourceProposal, ctx: SourceVerificationContext) => Promise<SourceVerdict>;
|
|
1683
1594
|
interface WorkerClaimDecorationOptions {
|
|
1684
|
-
|
|
1685
|
-
|
|
1686
|
-
|
|
1687
|
-
|
|
1595
|
+
router?: RouterClient;
|
|
1596
|
+
router_options?: TangleRouterOptions;
|
|
1597
|
+
/** Max output tokens for the claim-extraction call. Default 1200 (glm floor). */
|
|
1598
|
+
maxTokens?: number;
|
|
1688
1599
|
}
|
|
1689
1600
|
/**
|
|
1690
1601
|
* Ask an LLM to state, for one source, the single specific factual claim a
|
|
@@ -1700,98 +1611,72 @@ interface WorkerClaimDecorationOptions {
|
|
|
1700
1611
|
* `onMissingClaim` policy then decides).
|
|
1701
1612
|
*/
|
|
1702
1613
|
declare function createClaimDecorator(options?: WorkerClaimDecorationOptions): (source: ResearchSourceProposal, goal: string) => Promise<ResearchSourceProposal>;
|
|
1703
|
-
|
|
1704
|
-
|
|
1705
|
-
* The SINGLE-AGENT COLLECTION driver — the blind-collection baseline (Arm A).
|
|
1706
|
-
*
|
|
1707
|
-
* This is the honest null the depth A/B is measured against. The other drivers
|
|
1708
|
-
* spend extra inference to do something differentiated:
|
|
1709
|
-
* - `createVerifyingResearchDriver` runs an LLM gate per source (Arm B),
|
|
1710
|
-
* - `createResearchDrivingDriver` extracts claims, tracks corroboration, and
|
|
1711
|
-
* synthesizes deep follow-up questions to drive depth (Arm C).
|
|
1712
|
-
*
|
|
1713
|
-
* This driver does NONE of that. It is a pass-through: it accepts every source
|
|
1714
|
-
* the worker proposes and contributes no research, no gating, and no steering of
|
|
1715
|
-
* its own. The loop still dedups exact-uri duplicates before calling
|
|
1716
|
-
* `verifySource` (that is the loop's job, not the driver's), and the default
|
|
1717
|
-
* `foldGaps` (a plain bulleted list of the still-open readiness gaps) still folds
|
|
1718
|
-
* the gaps into the worker's next prompt — so the worker keeps researching, but
|
|
1719
|
-
* NOTHING intelligent sits between the worker and the knowledge base.
|
|
1720
|
-
*
|
|
1721
|
-
* In other words: ONE agent (the worker) collects sources round after round, and
|
|
1722
|
-
* the "driver" is an inert rubber stamp. That is exactly what "single-agent
|
|
1723
|
-
* collection" means — the topology with zero coordinator intelligence — so its
|
|
1724
|
-
* material-facts score is the floor every other arm must beat to justify its
|
|
1725
|
-
* extra inference cost.
|
|
1726
|
-
*
|
|
1727
|
-
* It adds NO router calls of its own: `verifySource` is a synchronous accept and
|
|
1728
|
-
* `foldGaps` is omitted so the loop uses its built-in gap list. So Arm A's cost
|
|
1729
|
-
* is the worker's cost alone — the cleanest possible blind-collection baseline.
|
|
1730
|
-
*/
|
|
1731
|
-
|
|
1614
|
+
//#endregion
|
|
1615
|
+
//#region src/collection-research-driver.d.ts
|
|
1732
1616
|
/**
|
|
1733
1617
|
* Build the single-agent collection driver. Accepts every source; never gates,
|
|
1734
1618
|
* never researches, never steers beyond the loop's default open-gap list. The
|
|
1735
1619
|
* worker is the only agent that thinks.
|
|
1736
1620
|
*/
|
|
1737
1621
|
declare function createCollectionResearchDriver(): ResearchDriver;
|
|
1738
|
-
|
|
1622
|
+
//#endregion
|
|
1623
|
+
//#region src/discovery.d.ts
|
|
1739
1624
|
interface DiscoveryTask {
|
|
1740
|
-
|
|
1741
|
-
|
|
1742
|
-
|
|
1743
|
-
|
|
1744
|
-
|
|
1625
|
+
id: string;
|
|
1626
|
+
goal: string;
|
|
1627
|
+
query?: string;
|
|
1628
|
+
sourceHints?: string[];
|
|
1629
|
+
metadata?: Record<string, unknown>;
|
|
1745
1630
|
}
|
|
1746
1631
|
interface DiscoveryResult {
|
|
1747
|
-
|
|
1748
|
-
|
|
1749
|
-
|
|
1750
|
-
|
|
1751
|
-
|
|
1752
|
-
|
|
1753
|
-
|
|
1754
|
-
|
|
1755
|
-
|
|
1756
|
-
|
|
1632
|
+
taskId: string;
|
|
1633
|
+
summary: string;
|
|
1634
|
+
sourceUris?: string[];
|
|
1635
|
+
claims?: Array<{
|
|
1636
|
+
text: string;
|
|
1637
|
+
sourceUri?: string;
|
|
1638
|
+
confidence?: number;
|
|
1639
|
+
}>;
|
|
1640
|
+
followUpTasks?: DiscoveryTask[];
|
|
1641
|
+
metadata?: Record<string, unknown>;
|
|
1757
1642
|
}
|
|
1758
1643
|
interface KnowledgeDiscoveryWorker {
|
|
1759
|
-
|
|
1644
|
+
run(task: DiscoveryTask, signal?: AbortSignal): Promise<DiscoveryResult> | DiscoveryResult;
|
|
1760
1645
|
}
|
|
1761
1646
|
interface KnowledgeDiscoveryDispatcher {
|
|
1762
|
-
|
|
1763
|
-
|
|
1764
|
-
|
|
1765
|
-
|
|
1647
|
+
dispatch(tasks: DiscoveryTask[], options?: {
|
|
1648
|
+
concurrency?: number;
|
|
1649
|
+
signal?: AbortSignal;
|
|
1650
|
+
}): Promise<DiscoveryResult[]>;
|
|
1766
1651
|
}
|
|
1767
1652
|
type DiscoveryLoopStopReason = 'complete' | 'max-rounds' | 'max-tasks' | 'aborted';
|
|
1768
1653
|
interface DiscoveryLoopRound {
|
|
1769
|
-
|
|
1770
|
-
|
|
1771
|
-
|
|
1772
|
-
|
|
1654
|
+
round: number;
|
|
1655
|
+
tasks: DiscoveryTask[];
|
|
1656
|
+
results: DiscoveryResult[];
|
|
1657
|
+
queuedFollowUps: DiscoveryTask[];
|
|
1773
1658
|
}
|
|
1774
1659
|
interface DiscoveryLoopResult {
|
|
1775
|
-
|
|
1776
|
-
|
|
1777
|
-
|
|
1778
|
-
|
|
1779
|
-
|
|
1780
|
-
|
|
1781
|
-
|
|
1782
|
-
|
|
1660
|
+
stopReason: DiscoveryLoopStopReason;
|
|
1661
|
+
tasksDispatched: number;
|
|
1662
|
+
results: DiscoveryResult[];
|
|
1663
|
+
rounds: DiscoveryLoopRound[];
|
|
1664
|
+
/** Tasks retained, not dropped, when a configured limit stops the loop. */
|
|
1665
|
+
pendingTasks: DiscoveryTask[];
|
|
1666
|
+
/** Structurally identical task identities ignored to prevent cycles. */
|
|
1667
|
+
duplicateTaskIds: string[];
|
|
1783
1668
|
}
|
|
1784
1669
|
interface RunDiscoveryLoopOptions {
|
|
1785
|
-
|
|
1786
|
-
|
|
1787
|
-
|
|
1788
|
-
|
|
1789
|
-
|
|
1790
|
-
|
|
1791
|
-
|
|
1792
|
-
|
|
1793
|
-
|
|
1794
|
-
|
|
1670
|
+
dispatcher: KnowledgeDiscoveryDispatcher;
|
|
1671
|
+
initialTasks: readonly DiscoveryTask[];
|
|
1672
|
+
/** Maximum follow-up depth including the initial dispatch. Default 3. */
|
|
1673
|
+
maxRounds?: number;
|
|
1674
|
+
/** Maximum tasks dispatched across all rounds. Default 24. */
|
|
1675
|
+
maxTasks?: number;
|
|
1676
|
+
/** Forwarded to the dispatcher. Default 4. */
|
|
1677
|
+
concurrency?: number;
|
|
1678
|
+
signal?: AbortSignal;
|
|
1679
|
+
onRound?: (round: DiscoveryLoopRound) => Promise<void> | void;
|
|
1795
1680
|
}
|
|
1796
1681
|
/**
|
|
1797
1682
|
* Dispatch discovery tasks and recursively pursue worker-proposed follow-ups.
|
|
@@ -1801,36 +1686,38 @@ interface RunDiscoveryLoopOptions {
|
|
|
1801
1686
|
*/
|
|
1802
1687
|
declare function runDiscoveryLoop(options: RunDiscoveryLoopOptions): Promise<DiscoveryLoopResult>;
|
|
1803
1688
|
declare function createLocalDiscoveryDispatcher(worker: KnowledgeDiscoveryWorker): KnowledgeDiscoveryDispatcher;
|
|
1804
|
-
|
|
1689
|
+
//#endregion
|
|
1690
|
+
//#region src/events.d.ts
|
|
1805
1691
|
interface KnowledgeEventQuery {
|
|
1806
|
-
|
|
1807
|
-
|
|
1808
|
-
|
|
1692
|
+
type?: KnowledgeEventType;
|
|
1693
|
+
target?: string;
|
|
1694
|
+
limit?: number;
|
|
1809
1695
|
}
|
|
1810
1696
|
declare function createKnowledgeEvent(input: {
|
|
1811
|
-
|
|
1812
|
-
|
|
1813
|
-
|
|
1814
|
-
|
|
1815
|
-
|
|
1697
|
+
type: KnowledgeEventType;
|
|
1698
|
+
actor?: string;
|
|
1699
|
+
target?: string;
|
|
1700
|
+
metadata?: Record<string, unknown>;
|
|
1701
|
+
now?: () => Date;
|
|
1816
1702
|
}): KnowledgeEvent;
|
|
1817
|
-
|
|
1703
|
+
//#endregion
|
|
1704
|
+
//#region src/filesystem-search-provider.d.ts
|
|
1818
1705
|
interface FileSystemSearchProviderOptions {
|
|
1819
|
-
|
|
1820
|
-
|
|
1821
|
-
|
|
1822
|
-
|
|
1823
|
-
|
|
1824
|
-
|
|
1825
|
-
|
|
1826
|
-
|
|
1827
|
-
|
|
1828
|
-
|
|
1829
|
-
|
|
1706
|
+
/** Knowledge-base root containing `knowledge/` and `.agent-knowledge/`. */
|
|
1707
|
+
root: string;
|
|
1708
|
+
/** Optional warm index, useful when the caller already built one. */
|
|
1709
|
+
index?: KnowledgeIndex;
|
|
1710
|
+
/** Default result count for `search()`. Defaults to 10. */
|
|
1711
|
+
defaultLimit?: number;
|
|
1712
|
+
/**
|
|
1713
|
+
* `manual` caches the index until `refresh: true` or `invalidate()`.
|
|
1714
|
+
* `always` rebuilds from disk on every search.
|
|
1715
|
+
*/
|
|
1716
|
+
refresh?: 'manual' | 'always';
|
|
1830
1717
|
}
|
|
1831
1718
|
interface FileSystemSearchOptions {
|
|
1832
|
-
|
|
1833
|
-
|
|
1719
|
+
limit?: number;
|
|
1720
|
+
refresh?: boolean;
|
|
1834
1721
|
}
|
|
1835
1722
|
/**
|
|
1836
1723
|
* File-first retrieval over an `agent-knowledge` KB.
|
|
@@ -1839,19 +1726,20 @@ interface FileSystemSearchOptions {
|
|
|
1839
1726
|
* markdown knowledge files before adding embeddings, rerankers, or a vector DB.
|
|
1840
1727
|
*/
|
|
1841
1728
|
declare class FileSystemSearchProvider {
|
|
1842
|
-
|
|
1843
|
-
|
|
1844
|
-
|
|
1845
|
-
|
|
1846
|
-
|
|
1847
|
-
|
|
1848
|
-
|
|
1849
|
-
|
|
1850
|
-
|
|
1851
|
-
|
|
1729
|
+
readonly root: string;
|
|
1730
|
+
private index;
|
|
1731
|
+
private readonly defaultLimit;
|
|
1732
|
+
private readonly refreshMode;
|
|
1733
|
+
constructor(options: FileSystemSearchProviderOptions);
|
|
1734
|
+
getIndex(options?: FileSystemSearchOptions): Promise<KnowledgeIndex>;
|
|
1735
|
+
search(query: string, options?: FileSystemSearchOptions): Promise<KnowledgeSearchResult[]>;
|
|
1736
|
+
retrieve(query: string, options?: FileSystemSearchOptions): Promise<RetrievedKnowledgeHit[]>;
|
|
1737
|
+
asRetrievalEvalRetriever(): RetrievalEvalRetriever;
|
|
1738
|
+
invalidate(): void;
|
|
1852
1739
|
}
|
|
1853
1740
|
declare function createFileSystemSearchProvider(options: FileSystemSearchProviderOptions): FileSystemSearchProvider;
|
|
1854
|
-
|
|
1741
|
+
//#endregion
|
|
1742
|
+
//#region src/freshness.d.ts
|
|
1855
1743
|
/**
|
|
1856
1744
|
* Knowledge freshness store: tracks when each `(workspaceId, sourceId)` pair
|
|
1857
1745
|
* was last successfully refreshed, and reports staleness against a TTL.
|
|
@@ -1882,44 +1770,44 @@ declare function createFileSystemSearchProvider(options: FileSystemSearchProvide
|
|
|
1882
1770
|
*/
|
|
1883
1771
|
/** Identity for one freshness record. */
|
|
1884
1772
|
interface FreshnessKey {
|
|
1885
|
-
|
|
1886
|
-
|
|
1773
|
+
workspaceId: string;
|
|
1774
|
+
sourceId: string;
|
|
1887
1775
|
}
|
|
1888
1776
|
/** TTL bound for staleness checks. */
|
|
1889
1777
|
interface FreshnessTtl extends FreshnessKey {
|
|
1890
|
-
|
|
1891
|
-
|
|
1892
|
-
|
|
1893
|
-
|
|
1778
|
+
/** Milliseconds; the record is stale when `Date.now() - last() > ttlMs`. */
|
|
1779
|
+
ttlMs: number;
|
|
1780
|
+
/** Injected clock for deterministic tests; defaults to system time. */
|
|
1781
|
+
now?: Date;
|
|
1894
1782
|
}
|
|
1895
1783
|
/** Mark argument. */
|
|
1896
1784
|
interface FreshnessMark extends FreshnessKey {
|
|
1897
|
-
|
|
1898
|
-
|
|
1899
|
-
|
|
1785
|
+
when: Date;
|
|
1786
|
+
/** Optional content hash captured at refresh time; aids debugging. */
|
|
1787
|
+
contentHash?: string;
|
|
1900
1788
|
}
|
|
1901
1789
|
interface KnowledgeFreshnessStore {
|
|
1902
|
-
|
|
1903
|
-
|
|
1904
|
-
|
|
1905
|
-
|
|
1906
|
-
|
|
1907
|
-
|
|
1908
|
-
|
|
1909
|
-
|
|
1790
|
+
/** Last refresh time, or null if never refreshed. */
|
|
1791
|
+
last(key: FreshnessKey): Promise<Date | null>;
|
|
1792
|
+
/** Record a successful refresh. */
|
|
1793
|
+
mark(input: FreshnessMark): Promise<void>;
|
|
1794
|
+
/** True iff `last(key)` is null or older than `ttlMs`. */
|
|
1795
|
+
stale(input: FreshnessTtl): Promise<boolean>;
|
|
1796
|
+
/** All records for a workspace. */
|
|
1797
|
+
list(workspaceId: string): Promise<FreshnessRecord[]>;
|
|
1910
1798
|
}
|
|
1911
1799
|
interface FreshnessRecord {
|
|
1912
|
-
|
|
1913
|
-
|
|
1914
|
-
|
|
1915
|
-
|
|
1800
|
+
workspaceId: string;
|
|
1801
|
+
sourceId: string;
|
|
1802
|
+
lastRefreshedAt: string;
|
|
1803
|
+
contentHash?: string;
|
|
1916
1804
|
}
|
|
1917
1805
|
interface FileSystemFreshnessStoreOptions {
|
|
1918
|
-
|
|
1919
|
-
|
|
1920
|
-
|
|
1921
|
-
|
|
1922
|
-
|
|
1806
|
+
/**
|
|
1807
|
+
* Knowledge root. The store writes to `<root>/.agent-knowledge/freshness.json`,
|
|
1808
|
+
* mirroring the convention used by `sources.json`.
|
|
1809
|
+
*/
|
|
1810
|
+
root: string;
|
|
1923
1811
|
}
|
|
1924
1812
|
/**
|
|
1925
1813
|
* Filesystem-backed implementation. Single JSON file per knowledge root,
|
|
@@ -1947,74 +1835,80 @@ declare function createFileSystemFreshnessStore(options: FileSystemFreshnessStor
|
|
|
1947
1835
|
* ```
|
|
1948
1836
|
*/
|
|
1949
1837
|
interface D1Adapter {
|
|
1950
|
-
|
|
1951
|
-
|
|
1952
|
-
|
|
1838
|
+
get(workspaceId: string, sourceId: string): Promise<FreshnessRecord | null>;
|
|
1839
|
+
upsert(record: FreshnessRecord): Promise<void>;
|
|
1840
|
+
listByWorkspace(workspaceId: string): Promise<FreshnessRecord[]>;
|
|
1953
1841
|
}
|
|
1954
1842
|
declare function createD1FreshnessStoreStub(adapter: D1Adapter): KnowledgeFreshnessStore;
|
|
1955
|
-
|
|
1843
|
+
//#endregion
|
|
1844
|
+
//#region src/frontmatter.d.ts
|
|
1956
1845
|
interface ParsedFrontmatter {
|
|
1957
|
-
|
|
1958
|
-
|
|
1846
|
+
frontmatter: Record<string, unknown>;
|
|
1847
|
+
body: string;
|
|
1959
1848
|
}
|
|
1960
1849
|
declare function parseFrontmatter(content: string): ParsedFrontmatter;
|
|
1961
1850
|
declare function formatFrontmatter(frontmatter: Record<string, unknown>, body: string): string;
|
|
1962
|
-
|
|
1851
|
+
//#endregion
|
|
1852
|
+
//#region src/graph.d.ts
|
|
1963
1853
|
declare function buildKnowledgeGraph(pages: KnowledgePage[]): KnowledgeGraph;
|
|
1964
|
-
|
|
1854
|
+
//#endregion
|
|
1855
|
+
//#region src/ids.d.ts
|
|
1965
1856
|
declare function sha256(text: string): string;
|
|
1966
1857
|
declare function slugify(input: string): string;
|
|
1967
1858
|
declare function stableId(prefix: string, content: string): string;
|
|
1968
|
-
|
|
1859
|
+
//#endregion
|
|
1860
|
+
//#region src/indexer.d.ts
|
|
1969
1861
|
declare function buildKnowledgeIndex(root: string): Promise<KnowledgeIndex>;
|
|
1970
1862
|
declare function writeKnowledgeIndex(root: string): Promise<KnowledgeIndex>;
|
|
1971
|
-
|
|
1863
|
+
//#endregion
|
|
1864
|
+
//#region src/inspect.d.ts
|
|
1972
1865
|
interface KnowledgeInspection {
|
|
1973
|
-
|
|
1974
|
-
|
|
1975
|
-
|
|
1976
|
-
|
|
1977
|
-
|
|
1978
|
-
|
|
1979
|
-
|
|
1980
|
-
|
|
1981
|
-
|
|
1982
|
-
|
|
1983
|
-
|
|
1984
|
-
|
|
1985
|
-
|
|
1986
|
-
|
|
1987
|
-
|
|
1866
|
+
pageCount: number;
|
|
1867
|
+
sourceCount: number;
|
|
1868
|
+
expiredSourceCount: number;
|
|
1869
|
+
staleSourceCount: number;
|
|
1870
|
+
edgeCount: number;
|
|
1871
|
+
findingCount: number;
|
|
1872
|
+
blockingFindingCount: number;
|
|
1873
|
+
topPages: Array<{
|
|
1874
|
+
path: string;
|
|
1875
|
+
title: string;
|
|
1876
|
+
degree: number;
|
|
1877
|
+
sources: number;
|
|
1878
|
+
}>;
|
|
1879
|
+
sourceFreshness: SourceFreshnessInspection[];
|
|
1880
|
+
findings: KnowledgeLintFinding[];
|
|
1988
1881
|
}
|
|
1989
1882
|
interface SourceFreshnessInspection {
|
|
1990
|
-
|
|
1991
|
-
|
|
1992
|
-
|
|
1993
|
-
|
|
1994
|
-
|
|
1995
|
-
|
|
1883
|
+
id: string;
|
|
1884
|
+
title?: string;
|
|
1885
|
+
uri: string;
|
|
1886
|
+
status: 'fresh' | 'expired' | 'unknown';
|
|
1887
|
+
validUntil?: string;
|
|
1888
|
+
lastVerifiedAt?: string;
|
|
1996
1889
|
}
|
|
1997
1890
|
declare function inspectKnowledgeIndex(index: KnowledgeIndex, options?: {
|
|
1998
|
-
|
|
1891
|
+
now?: Date;
|
|
1999
1892
|
}): KnowledgeInspection;
|
|
2000
1893
|
interface KnowledgeExplanation {
|
|
2001
|
-
|
|
2002
|
-
|
|
2003
|
-
|
|
2004
|
-
|
|
2005
|
-
|
|
2006
|
-
|
|
2007
|
-
|
|
2008
|
-
|
|
2009
|
-
|
|
2010
|
-
|
|
2011
|
-
|
|
2012
|
-
|
|
2013
|
-
|
|
2014
|
-
|
|
1894
|
+
target: string;
|
|
1895
|
+
page?: KnowledgePage;
|
|
1896
|
+
sources: Array<{
|
|
1897
|
+
id: string;
|
|
1898
|
+
title?: string;
|
|
1899
|
+
uri: string;
|
|
1900
|
+
}>;
|
|
1901
|
+
links: string[];
|
|
1902
|
+
inbound: string[];
|
|
1903
|
+
related: Array<{
|
|
1904
|
+
path: string;
|
|
1905
|
+
title: string;
|
|
1906
|
+
score: number;
|
|
1907
|
+
}>;
|
|
2015
1908
|
}
|
|
2016
1909
|
declare function explainKnowledgeTarget(index: KnowledgeIndex, target: string): KnowledgeExplanation;
|
|
2017
|
-
|
|
1910
|
+
//#endregion
|
|
1911
|
+
//#region src/investment-thesis-set.d.ts
|
|
2018
1912
|
/**
|
|
2019
1913
|
* HELD-OUT INVESTMENT-RESEARCH EVAL SET.
|
|
2020
1914
|
*
|
|
@@ -2064,71 +1958,71 @@ declare function explainKnowledgeTarget(index: KnowledgeIndex, target: string):
|
|
|
2064
1958
|
*/
|
|
2065
1959
|
/** A required answer component: satisfied when any synonym fragment is present. */
|
|
2066
1960
|
interface ExpectedGroup {
|
|
2067
|
-
|
|
2068
|
-
|
|
2069
|
-
|
|
2070
|
-
|
|
1961
|
+
/** Human label for the component (for the doc / audit). */
|
|
1962
|
+
label: string;
|
|
1963
|
+
/** Case-insensitive substring fragments; any one present satisfies the group. */
|
|
1964
|
+
anyOf: string[];
|
|
2071
1965
|
}
|
|
2072
1966
|
/** Lens the fact belongs to — so a set can be checked for category coverage. */
|
|
2073
1967
|
type MaterialFactLens = 'concentration' | 'leverage' | 'margin-trend' | 'liquidity' | 'capital-return' | 'governance' | 'off-balance-sheet' | 'regulatory';
|
|
2074
1968
|
/** One held-out material fact with a checkable expected answer + its provenance. */
|
|
2075
1969
|
interface MaterialFact {
|
|
2076
|
-
|
|
2077
|
-
|
|
2078
|
-
|
|
2079
|
-
|
|
2080
|
-
|
|
2081
|
-
|
|
2082
|
-
|
|
2083
|
-
|
|
2084
|
-
|
|
2085
|
-
|
|
2086
|
-
|
|
2087
|
-
|
|
2088
|
-
|
|
2089
|
-
|
|
2090
|
-
|
|
2091
|
-
|
|
2092
|
-
|
|
2093
|
-
|
|
2094
|
-
|
|
2095
|
-
|
|
2096
|
-
|
|
2097
|
-
|
|
2098
|
-
|
|
2099
|
-
|
|
2100
|
-
|
|
2101
|
-
|
|
2102
|
-
|
|
2103
|
-
|
|
2104
|
-
|
|
2105
|
-
|
|
2106
|
-
|
|
2107
|
-
|
|
1970
|
+
/** Stable id, `ticker/fN`. */
|
|
1971
|
+
id: string;
|
|
1972
|
+
/** Which analyst lens this fact exercises. For coverage + the doc. */
|
|
1973
|
+
lens: MaterialFactLens;
|
|
1974
|
+
/**
|
|
1975
|
+
* The material fact, in plain words — for the doc/audit. NEVER shown to a loop.
|
|
1976
|
+
* This is the thing a thorough analyst would flag and a ticker search misses.
|
|
1977
|
+
*/
|
|
1978
|
+
fact: string;
|
|
1979
|
+
/**
|
|
1980
|
+
* The checkable answer as required keyword GROUPS. The thesis text must contain
|
|
1981
|
+
* at least `minGroups` of these groups (default: all). A group is satisfied
|
|
1982
|
+
* when ANY of its `anyOf` fragments appears (case-insensitive substring).
|
|
1983
|
+
*/
|
|
1984
|
+
expected: ExpectedGroup[];
|
|
1985
|
+
/**
|
|
1986
|
+
* Minimum number of `expected` groups the thesis must contain to count the
|
|
1987
|
+
* fact SURFACED. Default = all groups (the strict bar). Lowered (and documented
|
|
1988
|
+
* inline) only when the fact is genuinely satisfiable by a subset.
|
|
1989
|
+
*/
|
|
1990
|
+
minGroups?: number;
|
|
1991
|
+
/**
|
|
1992
|
+
* PROVENANCE. The primary source URL this fact was read from — an SEC EDGAR
|
|
1993
|
+
* 10-K primary document, fetched live during curation.
|
|
1994
|
+
*/
|
|
1995
|
+
sourceUrl: string;
|
|
1996
|
+
/**
|
|
1997
|
+
* The literal value / phrase read out of `sourceUrl` that grounds the fact.
|
|
1998
|
+
* This is the "cite the actual filing + the value" requirement — verbatim or
|
|
1999
|
+
* near-verbatim from the filing, with the figure.
|
|
2000
|
+
*/
|
|
2001
|
+
evidence: string;
|
|
2108
2002
|
}
|
|
2109
2003
|
/** A company + the cutoff a loop researches as-of + its held-out material facts. */
|
|
2110
2004
|
interface CompanyEvalCase {
|
|
2111
|
-
|
|
2112
|
-
|
|
2113
|
-
|
|
2114
|
-
|
|
2115
|
-
|
|
2116
|
-
|
|
2117
|
-
|
|
2118
|
-
|
|
2119
|
-
|
|
2120
|
-
|
|
2121
|
-
|
|
2122
|
-
|
|
2123
|
-
|
|
2124
|
-
|
|
2125
|
-
|
|
2126
|
-
|
|
2127
|
-
|
|
2128
|
-
|
|
2129
|
-
|
|
2130
|
-
|
|
2131
|
-
|
|
2005
|
+
/** Ticker as of the cutoff. */
|
|
2006
|
+
ticker: string;
|
|
2007
|
+
/** Legal name as of the cutoff (what the loop is told to research). */
|
|
2008
|
+
company: string;
|
|
2009
|
+
/** SEC Central Index Key (CIK), zero-stripped — the EDGAR filer id. */
|
|
2010
|
+
cik: string;
|
|
2011
|
+
/**
|
|
2012
|
+
* Research-as-of date (ISO). The loop must reason as if it is this date; every
|
|
2013
|
+
* `evidence` value was knowable on or before it. >= 18 months before this set
|
|
2014
|
+
* was curated, so the outcome is known but is NOT a checklist item.
|
|
2015
|
+
*/
|
|
2016
|
+
cutoff: string;
|
|
2017
|
+
/** Sector, for coverage / the curation-bias disclosure. */
|
|
2018
|
+
sector: string;
|
|
2019
|
+
/**
|
|
2020
|
+
* The known POST-cutoff outcome — recorded for the reader ONLY, never graded.
|
|
2021
|
+
* Keeping it out of `facts` is what makes the set hindsight-free.
|
|
2022
|
+
*/
|
|
2023
|
+
knownOutcome: string;
|
|
2024
|
+
/** The held-out material facts for this company. */
|
|
2025
|
+
facts: MaterialFact[];
|
|
2132
2026
|
}
|
|
2133
2027
|
/**
|
|
2134
2028
|
* The eval set. 5 public companies, 5-8 held-out material facts each, every fact
|
|
@@ -2151,54 +2045,35 @@ declare const investmentThesisSet: CompanyEvalCase[];
|
|
|
2151
2045
|
* reproducible — so the eval never leaks into a model the loop could observe.
|
|
2152
2046
|
*/
|
|
2153
2047
|
declare function gradeFactAgainstText(fact: MaterialFact, thesisText: string): {
|
|
2154
|
-
|
|
2155
|
-
|
|
2156
|
-
|
|
2157
|
-
|
|
2048
|
+
surfaced: boolean;
|
|
2049
|
+
groupsFound: number;
|
|
2050
|
+
groupsTotal: number;
|
|
2051
|
+
foundLabels: string[];
|
|
2158
2052
|
};
|
|
2159
2053
|
/** Grade a whole company's thesis text: how many of its held-out facts it surfaces. */
|
|
2160
2054
|
declare function gradeCompanyAgainstText(company: CompanyEvalCase, thesisText: string): {
|
|
2161
|
-
|
|
2162
|
-
|
|
2163
|
-
|
|
2055
|
+
surfaced: number;
|
|
2056
|
+
total: number;
|
|
2057
|
+
perFact: ReturnType<typeof gradeFactAgainstText>[];
|
|
2164
2058
|
};
|
|
2165
2059
|
/** Total held-out facts across the set (the denominator the doc reports). */
|
|
2166
2060
|
declare function totalMaterialFacts(set?: CompanyEvalCase[]): number;
|
|
2167
2061
|
/** Count facts per lens across the set — used to report (and bound) curation bias. */
|
|
2168
2062
|
declare function lensDistribution(set?: CompanyEvalCase[]): Record<MaterialFactLens, number>;
|
|
2169
|
-
|
|
2170
|
-
|
|
2171
|
-
* The INVESTMENT-THESIS research task.
|
|
2172
|
-
*
|
|
2173
|
-
* Given `{ company, ticker, cik, cutoff }`, drive the SAME two-agent research
|
|
2174
|
-
* loop the ML deep-question A/B uses (`runVerifiedResearchLoop` + the real web
|
|
2175
|
-
* worker) to research the company AS OF the cutoff — web + SEC EDGAR, both public
|
|
2176
|
-
* — and produce an investment-thesis PAGE in the knowledge base: a judgment, the
|
|
2177
|
-
* drivers, and the risks, grounded in what it fetched.
|
|
2178
|
-
*
|
|
2179
|
-
* This file builds NOTHING new for the loop: it composes the existing worker +
|
|
2180
|
-
* driver + loop, supplies the readiness specs that steer the worker toward the
|
|
2181
|
-
* filing-level evidence (the analyst lenses), then writes a synthesis thesis page
|
|
2182
|
-
* the metric (`materialFactsSurfaced`) grades against the HELD-OUT checklist.
|
|
2183
|
-
*
|
|
2184
|
-
* THE FIREWALL: the task is told ONLY company + ticker + cutoff (+ the generic
|
|
2185
|
-
* analyst-lens readiness specs every company gets). It is NEVER shown the
|
|
2186
|
-
* checklist. The checklist is read only afterward, by the metric. So a high score
|
|
2187
|
-
* is research depth, not teaching-to-the-test.
|
|
2188
|
-
*/
|
|
2189
|
-
|
|
2063
|
+
//#endregion
|
|
2064
|
+
//#region src/investment-thesis-task.d.ts
|
|
2190
2065
|
/** The minimal brief a thesis run is given — the firewall boundary. */
|
|
2191
2066
|
interface ThesisTaskInput {
|
|
2192
|
-
|
|
2193
|
-
|
|
2194
|
-
|
|
2195
|
-
|
|
2196
|
-
|
|
2197
|
-
|
|
2198
|
-
|
|
2199
|
-
|
|
2200
|
-
|
|
2201
|
-
|
|
2067
|
+
/** Legal name as of the cutoff — what the loop researches. */
|
|
2068
|
+
company: string;
|
|
2069
|
+
/** Ticker as of the cutoff. */
|
|
2070
|
+
ticker: string;
|
|
2071
|
+
/** SEC Central Index Key (CIK), zero-stripped — the EDGAR filer id. */
|
|
2072
|
+
cik: string;
|
|
2073
|
+
/** Research-as-of date (ISO). The loop must reason as if it is this date. */
|
|
2074
|
+
cutoff: string;
|
|
2075
|
+
/** Sector, for the readiness query context (NOT a checklist hint). */
|
|
2076
|
+
sector?: string;
|
|
2202
2077
|
}
|
|
2203
2078
|
/**
|
|
2204
2079
|
* The generic analyst-lens readiness specs every company gets. They are the ONLY
|
|
@@ -2214,26 +2089,26 @@ interface ThesisTaskInput {
|
|
|
2214
2089
|
*/
|
|
2215
2090
|
declare function thesisReadinessSpecs(input: ThesisTaskInput): KnowledgeReadinessSpec[];
|
|
2216
2091
|
interface ThesisRunOptions {
|
|
2217
|
-
|
|
2218
|
-
|
|
2219
|
-
|
|
2220
|
-
|
|
2221
|
-
|
|
2222
|
-
|
|
2223
|
-
|
|
2224
|
-
|
|
2225
|
-
|
|
2226
|
-
|
|
2227
|
-
|
|
2228
|
-
|
|
2229
|
-
|
|
2092
|
+
/** The KB root the loop writes into. */
|
|
2093
|
+
root: string;
|
|
2094
|
+
/** Shared router client (web search + chat). Defaults to env creds. */
|
|
2095
|
+
router: RouterClient;
|
|
2096
|
+
/** The driver — verify/dedup or research-driving. The loop's coordinator. */
|
|
2097
|
+
driver: ResearchDriver;
|
|
2098
|
+
/** Round budget. Default 3 (the depth-driving driver needs >1). */
|
|
2099
|
+
maxRounds?: number;
|
|
2100
|
+
/** Worker tuning forwarded to `createWebResearchWorker`. */
|
|
2101
|
+
workerOptions?: Omit<WebResearchWorkerOptions, 'router'>;
|
|
2102
|
+
/** Max tokens for the synthesis pass. Default 1600 (above glm-5.2's reasoning floor). */
|
|
2103
|
+
synthesisMaxTokens?: number;
|
|
2104
|
+
signal?: AbortSignal;
|
|
2230
2105
|
}
|
|
2231
2106
|
interface ThesisRunResult {
|
|
2232
|
-
|
|
2233
|
-
|
|
2234
|
-
|
|
2235
|
-
|
|
2236
|
-
|
|
2107
|
+
loop: VerifiedResearchLoopResult;
|
|
2108
|
+
/** The synthesized thesis text. */
|
|
2109
|
+
thesis: string;
|
|
2110
|
+
/** Path of the thesis page written into the KB. */
|
|
2111
|
+
thesisPath: string;
|
|
2237
2112
|
}
|
|
2238
2113
|
/**
|
|
2239
2114
|
* Run the full thesis task: drive the two-agent loop to research the company AS
|
|
@@ -2242,101 +2117,79 @@ interface ThesisRunResult {
|
|
|
2242
2117
|
* `materialFactsSurfaced(root, checklist)` — the checklist is never passed here.
|
|
2243
2118
|
*/
|
|
2244
2119
|
declare function runInvestmentThesisTask(input: ThesisTaskInput, options: ThesisRunOptions): Promise<ThesisRunResult>;
|
|
2245
|
-
|
|
2120
|
+
//#endregion
|
|
2121
|
+
//#region src/kb-store.d.ts
|
|
2246
2122
|
interface KbStore {
|
|
2247
|
-
|
|
2248
|
-
|
|
2249
|
-
|
|
2250
|
-
|
|
2251
|
-
|
|
2252
|
-
|
|
2253
|
-
|
|
2254
|
-
|
|
2255
|
-
|
|
2256
|
-
|
|
2123
|
+
putSource(source: SourceRecord): Promise<void>;
|
|
2124
|
+
getSource(id: string): Promise<SourceRecord | null>;
|
|
2125
|
+
listSources(): Promise<SourceRecord[]>;
|
|
2126
|
+
putPage(page: KnowledgePage): Promise<void>;
|
|
2127
|
+
getPage(idOrPath: string): Promise<KnowledgePage | null>;
|
|
2128
|
+
listPages(): Promise<KnowledgePage[]>;
|
|
2129
|
+
putIndex(index: KnowledgeIndex): Promise<void>;
|
|
2130
|
+
getIndex(): Promise<KnowledgeIndex | null>;
|
|
2131
|
+
putEvent(event: KnowledgeEvent): Promise<void>;
|
|
2132
|
+
listEvents(query?: KnowledgeEventQuery): Promise<KnowledgeEvent[]>;
|
|
2257
2133
|
}
|
|
2258
2134
|
declare class MemoryKbStore implements KbStore {
|
|
2259
|
-
|
|
2260
|
-
|
|
2261
|
-
|
|
2262
|
-
|
|
2263
|
-
|
|
2264
|
-
|
|
2265
|
-
|
|
2266
|
-
|
|
2267
|
-
|
|
2268
|
-
|
|
2269
|
-
|
|
2270
|
-
|
|
2271
|
-
|
|
2272
|
-
|
|
2135
|
+
private readonly sources;
|
|
2136
|
+
private readonly pages;
|
|
2137
|
+
private readonly events;
|
|
2138
|
+
private index;
|
|
2139
|
+
putSource(source: SourceRecord): Promise<void>;
|
|
2140
|
+
getSource(id: string): Promise<SourceRecord | null>;
|
|
2141
|
+
listSources(): Promise<SourceRecord[]>;
|
|
2142
|
+
putPage(page: KnowledgePage): Promise<void>;
|
|
2143
|
+
getPage(idOrPath: string): Promise<KnowledgePage | null>;
|
|
2144
|
+
listPages(): Promise<KnowledgePage[]>;
|
|
2145
|
+
putIndex(index: KnowledgeIndex): Promise<void>;
|
|
2146
|
+
getIndex(): Promise<KnowledgeIndex | null>;
|
|
2147
|
+
putEvent(event: KnowledgeEvent): Promise<void>;
|
|
2148
|
+
listEvents(query?: KnowledgeEventQuery): Promise<KnowledgeEvent[]>;
|
|
2273
2149
|
}
|
|
2274
2150
|
declare class FileSystemKbStore implements KbStore {
|
|
2275
|
-
|
|
2276
|
-
|
|
2277
|
-
|
|
2278
|
-
|
|
2279
|
-
|
|
2280
|
-
|
|
2281
|
-
|
|
2282
|
-
|
|
2283
|
-
|
|
2284
|
-
|
|
2285
|
-
|
|
2286
|
-
|
|
2287
|
-
|
|
2288
|
-
|
|
2289
|
-
|
|
2290
|
-
}
|
|
2291
|
-
|
|
2151
|
+
private readonly dir;
|
|
2152
|
+
constructor(dir: string);
|
|
2153
|
+
putSource(source: SourceRecord): Promise<void>;
|
|
2154
|
+
getSource(id: string): Promise<SourceRecord | null>;
|
|
2155
|
+
listSources(): Promise<SourceRecord[]>;
|
|
2156
|
+
putPage(page: KnowledgePage): Promise<void>;
|
|
2157
|
+
getPage(idOrPath: string): Promise<KnowledgePage | null>;
|
|
2158
|
+
listPages(): Promise<KnowledgePage[]>;
|
|
2159
|
+
putIndex(index: KnowledgeIndex): Promise<void>;
|
|
2160
|
+
getIndex(): Promise<KnowledgeIndex | null>;
|
|
2161
|
+
putEvent(event: KnowledgeEvent): Promise<void>;
|
|
2162
|
+
listEvents(query?: KnowledgeEventQuery): Promise<KnowledgeEvent[]>;
|
|
2163
|
+
private updateIndex;
|
|
2164
|
+
private readIndex;
|
|
2165
|
+
private readEvents;
|
|
2166
|
+
}
|
|
2167
|
+
//#endregion
|
|
2168
|
+
//#region src/lint.d.ts
|
|
2292
2169
|
declare function lintKnowledgeIndex(index: KnowledgeIndex): KnowledgeLintFinding[];
|
|
2293
|
-
|
|
2294
|
-
|
|
2295
|
-
* `materialFactsSurfaced` — the held-out investment-research METRIC.
|
|
2296
|
-
*
|
|
2297
|
-
* Given a knowledge base a research loop built for a company and the company's
|
|
2298
|
-
* HELD-OUT material-fact checklist (`tests/eval/investment-thesis-set.ts`, never
|
|
2299
|
-
* shown to the loop), this returns the FRACTION of checklist items the KB's
|
|
2300
|
-
* pages surface + ground. The check is the same `$0`, model-free, deterministic
|
|
2301
|
-
* substring grader the loop's checklist already ships (`gradeFactAgainstText` /
|
|
2302
|
-
* `gradeCompanyAgainstText`) — so the answer key never reaches a model the loop
|
|
2303
|
-
* could observe, exactly the firewall the ML deep-question exam uses.
|
|
2304
|
-
*
|
|
2305
|
-
* The ONLY thing this file adds over the raw grader is the KB→text join: it reads
|
|
2306
|
-
* the curated pages (and the raw source text) the loop wrote and hands their
|
|
2307
|
-
* concatenation to the grader. That join mirrors `kbText` in the research-quality
|
|
2308
|
-
* A/B (research-driving-ab.test.ts) so the thesis metric and the ML-exam metric
|
|
2309
|
-
* read a KB the same way.
|
|
2310
|
-
*
|
|
2311
|
-
* WHY pages AND source text: an honest thesis surfaces a buried fact in its
|
|
2312
|
-
* curated thesis PAGE (the judgment), but a loop whose page is thin while its
|
|
2313
|
-
* fetched filings are rich should still get credit for what it actually pulled.
|
|
2314
|
-
* Grading the union is the faithful, not the lenient, choice — it rewards the
|
|
2315
|
-
* loop that REACHED the filing even if its synthesis was terse, and it cannot
|
|
2316
|
-
* manufacture a hit the underlying evidence does not contain.
|
|
2317
|
-
*/
|
|
2318
|
-
|
|
2170
|
+
//#endregion
|
|
2171
|
+
//#region src/material-facts-metric.d.ts
|
|
2319
2172
|
/** Per-fact grade plus the fact's id/lens, for the audit trail. */
|
|
2320
2173
|
interface FactResult {
|
|
2321
|
-
|
|
2322
|
-
|
|
2323
|
-
|
|
2324
|
-
|
|
2325
|
-
|
|
2326
|
-
|
|
2174
|
+
id: string;
|
|
2175
|
+
lens: CompanyEvalCase['facts'][number]['lens'];
|
|
2176
|
+
surfaced: boolean;
|
|
2177
|
+
groupsFound: number;
|
|
2178
|
+
groupsTotal: number;
|
|
2179
|
+
foundLabels: string[];
|
|
2327
2180
|
}
|
|
2328
2181
|
/** The metric's result for one company: the surfaced fraction + the per-fact trail. */
|
|
2329
2182
|
interface MaterialFactsResult {
|
|
2330
|
-
|
|
2331
|
-
|
|
2332
|
-
|
|
2333
|
-
|
|
2334
|
-
|
|
2335
|
-
|
|
2336
|
-
|
|
2337
|
-
|
|
2338
|
-
|
|
2339
|
-
|
|
2183
|
+
ticker: string;
|
|
2184
|
+
company: string;
|
|
2185
|
+
/** Held-out facts the KB surfaced + grounded. */
|
|
2186
|
+
surfaced: number;
|
|
2187
|
+
/** Total held-out facts for this company (the denominator). */
|
|
2188
|
+
total: number;
|
|
2189
|
+
/** `surfaced / total` in [0, 1]. */
|
|
2190
|
+
fraction: number;
|
|
2191
|
+
/** Per-fact grade, in checklist order, for the doc / audit. */
|
|
2192
|
+
perFact: FactResult[];
|
|
2340
2193
|
}
|
|
2341
2194
|
/**
|
|
2342
2195
|
* Join a KB index into the single text blob the grader scans: every curated PAGE
|
|
@@ -2364,82 +2217,59 @@ declare function materialFactsSurfacedInText(company: CompanyEvalCase, kbText: s
|
|
|
2364
2217
|
* never passed to the loop, and is read only here, after the loop finished.
|
|
2365
2218
|
*/
|
|
2366
2219
|
declare function materialFactsSurfaced(kb: string | KnowledgeIndex, checklist: CompanyEvalCase): Promise<MaterialFactsResult>;
|
|
2367
|
-
|
|
2220
|
+
//#endregion
|
|
2221
|
+
//#region src/mutation-lock.d.ts
|
|
2368
2222
|
interface PendingKnowledgeMutation {
|
|
2369
|
-
|
|
2370
|
-
|
|
2371
|
-
|
|
2372
|
-
|
|
2373
|
-
|
|
2374
|
-
|
|
2223
|
+
transactionId: string;
|
|
2224
|
+
purpose: string;
|
|
2225
|
+
recoveryOwner?: string;
|
|
2226
|
+
createdAt: string;
|
|
2227
|
+
direction: 'apply' | 'rollback';
|
|
2228
|
+
paths: string[];
|
|
2375
2229
|
}
|
|
2376
2230
|
interface RecoverPendingKnowledgeMutationOptions {
|
|
2377
|
-
|
|
2378
|
-
|
|
2231
|
+
transactionId: string;
|
|
2232
|
+
action: 'apply' | 'rollback';
|
|
2379
2233
|
}
|
|
2380
2234
|
declare function inspectPendingKnowledgeMutation(root: string): Promise<PendingKnowledgeMutation | null>;
|
|
2381
2235
|
declare function recoverPendingKnowledgeMutation(root: string, options: RecoverPendingKnowledgeMutationOptions): Promise<void>;
|
|
2382
|
-
|
|
2383
|
-
|
|
2384
|
-
* Bridge from `AnalystFinding` (agent-eval) to knowledge proposals.
|
|
2385
|
-
*
|
|
2386
|
-
* Closes the failure → wiki side of the recursive-self-improvement
|
|
2387
|
-
* loop: a knowledge-gap or knowledge-poisoning finding produced by an
|
|
2388
|
-
* analyst becomes a concrete proposal an operator (or auto-merge bot)
|
|
2389
|
-
* can review and apply. The bridge is intentionally lossless on the
|
|
2390
|
-
* fail-loud side — a finding the parser can't classify returns a
|
|
2391
|
-
* `KnowledgeProposalParseError` rather than a silent skip, so the
|
|
2392
|
-
* loop never accepts an underspecified edit.
|
|
2393
|
-
*
|
|
2394
|
-
* Subject grammar this bridge understands (analyst-side convention,
|
|
2395
|
-
* stamped in the kind prompts):
|
|
2396
|
-
*
|
|
2397
|
-
* agent-knowledge:wiki:<page-slug> create / update page
|
|
2398
|
-
* agent-knowledge:wiki:<page-slug>#<heading> insert section under page
|
|
2399
|
-
* agent-knowledge:claim:<topic> draft claim row
|
|
2400
|
-
* agent-knowledge:raw:<source-id> lift raw → curated
|
|
2401
|
-
* agent-knowledge:stale:<page-slug> mark page superseded
|
|
2402
|
-
*
|
|
2403
|
-
* Anything else (websearch:outdated:*, tool-doc:*, system-prompt:*,
|
|
2404
|
-
* memory:*) is NOT a knowledge-base concern and returns `null` so the
|
|
2405
|
-
* loop's improvement-applier handles it.
|
|
2406
|
-
*/
|
|
2407
|
-
|
|
2236
|
+
//#endregion
|
|
2237
|
+
//#region src/propose-from-finding.d.ts
|
|
2408
2238
|
interface KnowledgeProposal {
|
|
2409
|
-
|
|
2410
|
-
|
|
2411
|
-
|
|
2412
|
-
|
|
2413
|
-
|
|
2414
|
-
|
|
2415
|
-
|
|
2416
|
-
|
|
2417
|
-
|
|
2418
|
-
|
|
2419
|
-
|
|
2420
|
-
|
|
2421
|
-
|
|
2422
|
-
|
|
2423
|
-
|
|
2424
|
-
|
|
2425
|
-
|
|
2426
|
-
|
|
2427
|
-
|
|
2428
|
-
|
|
2429
|
-
|
|
2430
|
-
|
|
2431
|
-
|
|
2432
|
-
|
|
2433
|
-
|
|
2434
|
-
|
|
2435
|
-
|
|
2436
|
-
|
|
2437
|
-
|
|
2239
|
+
/**
|
|
2240
|
+
* Stable id derived from the finding so cross-run diffs share an
|
|
2241
|
+
* identity. Re-proposing the same finding produces the same id.
|
|
2242
|
+
*/
|
|
2243
|
+
id: string;
|
|
2244
|
+
/** The finding that generated this proposal — useful for audit + revert. */
|
|
2245
|
+
sourceFindingId: string;
|
|
2246
|
+
/** What the proposal does. */
|
|
2247
|
+
kind: 'create-page' | 'update-page' | 'append-section' | 'create-claim' | 'lift-raw' | 'mark-stale';
|
|
2248
|
+
/** Locus on disk (page slug or claim topic). */
|
|
2249
|
+
locus: string;
|
|
2250
|
+
/**
|
|
2251
|
+
* Page write blocks the standard `applyKnowledgeWriteBlocks` consumer
|
|
2252
|
+
* accepts. Empty for proposals that don't change page text (e.g.
|
|
2253
|
+
* `create-claim` produces a `claim` field instead).
|
|
2254
|
+
*/
|
|
2255
|
+
writeBlocks: KnowledgeWriteBlock[];
|
|
2256
|
+
/**
|
|
2257
|
+
* Granular claim draft for proposals whose unit-of-change is a claim
|
|
2258
|
+
* row rather than a whole page. `status: 'draft'` until reviewed.
|
|
2259
|
+
*/
|
|
2260
|
+
claim?: KnowledgeClaim;
|
|
2261
|
+
/** Per-proposal metadata: severity, confidence, source span. */
|
|
2262
|
+
metadata: {
|
|
2263
|
+
severity: AnalystSeverity;
|
|
2264
|
+
confidence: number;
|
|
2265
|
+
evidence_uri?: string;
|
|
2266
|
+
analyst_id: string;
|
|
2267
|
+
};
|
|
2438
2268
|
}
|
|
2439
2269
|
declare class KnowledgeProposalParseError extends Error {
|
|
2440
|
-
|
|
2441
|
-
|
|
2442
|
-
|
|
2270
|
+
readonly findingId: string;
|
|
2271
|
+
readonly subject: string;
|
|
2272
|
+
constructor(findingId: string, subject: string, message: string);
|
|
2443
2273
|
}
|
|
2444
2274
|
/**
|
|
2445
2275
|
* Convert one `AnalystFinding` into a knowledge proposal. Returns
|
|
@@ -2460,194 +2290,153 @@ declare function proposeFromFinding(finding: AnalystFinding): KnowledgeProposal
|
|
|
2460
2290
|
* decides per-error whether to abort or continue.
|
|
2461
2291
|
*/
|
|
2462
2292
|
interface ProposeFromFindingsResult {
|
|
2463
|
-
|
|
2464
|
-
|
|
2465
|
-
|
|
2293
|
+
proposals: KnowledgeProposal[];
|
|
2294
|
+
skipped: number;
|
|
2295
|
+
errors: KnowledgeProposalParseError[];
|
|
2466
2296
|
}
|
|
2467
2297
|
declare function proposeFromFindings(findings: ReadonlyArray<AnalystFinding>): ProposeFromFindingsResult;
|
|
2468
|
-
|
|
2298
|
+
//#endregion
|
|
2299
|
+
//#region src/readiness-check.d.ts
|
|
2469
2300
|
interface EvaluateKnowledgeBaseReadinessOptions {
|
|
2470
|
-
|
|
2471
|
-
|
|
2472
|
-
|
|
2473
|
-
|
|
2474
|
-
|
|
2475
|
-
|
|
2476
|
-
|
|
2301
|
+
root: string;
|
|
2302
|
+
goal: string;
|
|
2303
|
+
readinessSpecs?: readonly KnowledgeReadinessSpec[];
|
|
2304
|
+
readinessTaskId?: string;
|
|
2305
|
+
readiness?: Omit<BuildEvalKnowledgeBundleOptions, 'taskId' | 'index' | 'specs'>;
|
|
2306
|
+
strict?: ValidateKnowledgeOptions['strict'];
|
|
2307
|
+
kbQuality?: KnowledgeBaseQualityOptions;
|
|
2477
2308
|
}
|
|
2478
2309
|
interface KnowledgeBaseReadinessEvaluation {
|
|
2479
|
-
|
|
2480
|
-
|
|
2481
|
-
|
|
2482
|
-
|
|
2483
|
-
|
|
2484
|
-
|
|
2485
|
-
|
|
2486
|
-
|
|
2487
|
-
|
|
2488
|
-
|
|
2489
|
-
|
|
2310
|
+
ready: boolean;
|
|
2311
|
+
summary: string;
|
|
2312
|
+
index: KnowledgeIndex;
|
|
2313
|
+
validation: ValidateKnowledgeResult;
|
|
2314
|
+
readiness?: EvalKnowledgeBundleBuildResult;
|
|
2315
|
+
kbQuality: KnowledgeBaseQualityReport;
|
|
2316
|
+
dimensions: {
|
|
2317
|
+
validation: number;
|
|
2318
|
+
kb_quality: number;
|
|
2319
|
+
blocking_readiness: number;
|
|
2320
|
+
};
|
|
2490
2321
|
}
|
|
2491
2322
|
declare function evaluateKnowledgeBaseReadiness(options: EvaluateKnowledgeBaseReadinessOptions): Promise<KnowledgeBaseReadinessEvaluation>;
|
|
2492
|
-
|
|
2323
|
+
//#endregion
|
|
2324
|
+
//#region src/release.d.ts
|
|
2493
2325
|
interface KnowledgeReleaseReport {
|
|
2494
|
-
|
|
2495
|
-
|
|
2496
|
-
|
|
2497
|
-
|
|
2326
|
+
release: KnowledgeRelease;
|
|
2327
|
+
scorecard: ReleaseConfidenceScorecard;
|
|
2328
|
+
candidateRuns: RunRecord[];
|
|
2329
|
+
baselineRuns: RunRecord[];
|
|
2498
2330
|
}
|
|
2499
2331
|
/**
|
|
2500
2332
|
* Build a knowledge release report from candidate and baseline run records,
|
|
2501
2333
|
* optional trace evidence, and an optional decision record.
|
|
2502
2334
|
*/
|
|
2503
2335
|
interface KnowledgeReleaseInput {
|
|
2504
|
-
|
|
2505
|
-
|
|
2506
|
-
|
|
2507
|
-
|
|
2508
|
-
|
|
2509
|
-
|
|
2510
|
-
|
|
2511
|
-
|
|
2512
|
-
|
|
2513
|
-
|
|
2514
|
-
|
|
2515
|
-
|
|
2516
|
-
|
|
2517
|
-
|
|
2518
|
-
|
|
2519
|
-
|
|
2520
|
-
|
|
2336
|
+
candidateId: string;
|
|
2337
|
+
baselineId?: string;
|
|
2338
|
+
candidateRuns: RunRecord[];
|
|
2339
|
+
baselineRuns?: RunRecord[];
|
|
2340
|
+
traces?: ReleaseTraceEvidence[];
|
|
2341
|
+
gateDecision?: GateDecision | null;
|
|
2342
|
+
/** Scenario corpus used to prove train and holdout split coverage. */
|
|
2343
|
+
scenarios?: readonly DatasetScenario[];
|
|
2344
|
+
/**
|
|
2345
|
+
* Require both a holdout scenario and a holdout run.
|
|
2346
|
+
* Provide `scenarios` with at least one `split: 'holdout'` item when true.
|
|
2347
|
+
*/
|
|
2348
|
+
hasHoldout?: boolean;
|
|
2349
|
+
/** Candidate is the search-best variant — a promotion precondition. Default true. */
|
|
2350
|
+
promotedIsBest?: boolean;
|
|
2351
|
+
createdAt?: string;
|
|
2352
|
+
minScore?: number;
|
|
2521
2353
|
}
|
|
2522
2354
|
declare function knowledgeReleaseReport(input: KnowledgeReleaseInput): KnowledgeReleaseReport;
|
|
2523
|
-
|
|
2524
|
-
|
|
2525
|
-
* Research-DRIVING driver for `runVerifiedResearchLoop`.
|
|
2526
|
-
*
|
|
2527
|
-
* The shipped drivers all FILTER the worker's sources:
|
|
2528
|
-
* - `createVerifyingResearchDriver` judges on-topic relevance,
|
|
2529
|
-
* - `createAdaptiveResearchDriver` dedups then triages then escalates,
|
|
2530
|
-
* - `createClaimGroundingVerifier` rejects misattributed citations.
|
|
2531
|
-
*
|
|
2532
|
-
* This driver does the OPPOSITE job: instead of narrowing the worker's output,
|
|
2533
|
-
* it DRIVES the research DEEPER each round. Its value is not "fewer sources" —
|
|
2534
|
-
* it is "more answered, better-corroborated sub-questions". Concretely, each
|
|
2535
|
-
* round it:
|
|
2536
|
-
*
|
|
2537
|
-
* 1. EXTRACTS the key claims from the worker's new sources (one LLM pass per
|
|
2538
|
-
* source, in `verifySource`; falls back to a deterministic sentence-pull
|
|
2539
|
-
* when the model is unavailable so a round never silently extracts nothing).
|
|
2540
|
-
* 2. TRACKS each claim's support — the set of INDEPENDENT sources (by canonical
|
|
2541
|
-
* host) that assert it — and detects CONTRADICTIONS between a new claim and
|
|
2542
|
-
* one already on the ledger.
|
|
2543
|
-
* 3. GENERATES the next round's DEEP sub-questions from the accumulated claims,
|
|
2544
|
-
* in four kinds — comparative ("how does X's tradeoff differ from Y's?"),
|
|
2545
|
-
* mechanism ("under what precise condition does X fail?"), gap ("what
|
|
2546
|
-
* specific result is missing?"), and contradiction ("does any source
|
|
2547
|
-
* challenge claim Z?").
|
|
2548
|
-
* 4. FLAGS weakly-supported claims (only ONE independent source) and
|
|
2549
|
-
* contradicted claims as INVALIDATION targets and demands the worker find
|
|
2550
|
-
* corroborating / refuting evidence for them.
|
|
2551
|
-
* 5. FOLDS the deep sub-questions + invalidation challenges into the worker's
|
|
2552
|
-
* next prompt via the loop's `foldGaps` → `steer` channel — that is the
|
|
2553
|
-
* mechanism that drives DEPTH and VALIDATION rather than breadth.
|
|
2554
|
-
*
|
|
2555
|
-
* COMPLETION (`isComplete` / the `done` judgment the caller gates on) does NOT
|
|
2556
|
-
* look at source COUNT. It is done only when every deep sub-question it raised
|
|
2557
|
-
* has been addressed AND every key claim is either supported by >= 2 independent
|
|
2558
|
-
* sources OR explicitly marked CONTESTED (a contradiction the loop surfaced and
|
|
2559
|
-
* could not resolve). A KB with twenty sources all asserting one unchallenged
|
|
2560
|
-
* claim is NOT done; a KB whose handful of claims are each corroborated or
|
|
2561
|
-
* contested IS.
|
|
2562
|
-
*
|
|
2563
|
-
* It reuses `runVerifiedResearchLoop` (it is a plain `ResearchDriver`), the web
|
|
2564
|
-
* worker, `sha256` (claim identity), `canonicalizeUrl` (independent-source
|
|
2565
|
-
* identity), and the `RouterClient` chat surface; it reinvents none of them.
|
|
2566
|
-
*/
|
|
2567
|
-
|
|
2355
|
+
//#endregion
|
|
2356
|
+
//#region src/research-driving-driver.d.ts
|
|
2568
2357
|
/** The four deep sub-question kinds the driver generates to drive depth. */
|
|
2569
2358
|
type DeepQuestionKind = 'comparative' | 'mechanism' | 'gap' | 'contradiction';
|
|
2570
2359
|
/** A deep sub-question the driver folds into the worker's next prompt. */
|
|
2571
2360
|
interface DeepQuestion {
|
|
2572
|
-
|
|
2573
|
-
|
|
2574
|
-
|
|
2575
|
-
|
|
2576
|
-
|
|
2577
|
-
|
|
2578
|
-
|
|
2579
|
-
|
|
2580
|
-
|
|
2581
|
-
|
|
2361
|
+
kind: DeepQuestionKind;
|
|
2362
|
+
text: string;
|
|
2363
|
+
/** sha256-derived stable id, so "addressed" can be tracked across rounds. */
|
|
2364
|
+
id: string;
|
|
2365
|
+
/** Claim id(s) this question interrogates (for contradiction/mechanism kinds). */
|
|
2366
|
+
claimIds: string[];
|
|
2367
|
+
/** True once a later round's evidence addressed it (see `markAddressed`). */
|
|
2368
|
+
addressed: boolean;
|
|
2369
|
+
/** The round this question was raised in. */
|
|
2370
|
+
raisedRound: number;
|
|
2582
2371
|
}
|
|
2583
2372
|
/** One tracked claim plus the independent sources that assert it. */
|
|
2584
2373
|
interface TrackedClaim {
|
|
2585
|
-
|
|
2586
|
-
|
|
2587
|
-
|
|
2588
|
-
|
|
2589
|
-
|
|
2590
|
-
|
|
2591
|
-
|
|
2592
|
-
|
|
2593
|
-
|
|
2594
|
-
|
|
2595
|
-
|
|
2596
|
-
|
|
2597
|
-
|
|
2598
|
-
|
|
2599
|
-
|
|
2600
|
-
|
|
2374
|
+
id: string;
|
|
2375
|
+
/** The claim text as first extracted (kept for prompts/audit). */
|
|
2376
|
+
text: string;
|
|
2377
|
+
/** Canonical hosts of the INDEPENDENT sources that assert this claim. */
|
|
2378
|
+
supportingHosts: Set<string>;
|
|
2379
|
+
/** Source URIs that assert this claim (provenance; may share a host). */
|
|
2380
|
+
supportingUris: string[];
|
|
2381
|
+
/** Claim ids this claim was found to CONTRADICT (and vice versa). */
|
|
2382
|
+
contradicts: Set<string>;
|
|
2383
|
+
/**
|
|
2384
|
+
* CONTESTED = a contradiction the loop surfaced but could not resolve to a
|
|
2385
|
+
* single supported claim. A contested claim counts as "settled enough to be
|
|
2386
|
+
* done" (we report the disagreement) even with < 2 independent sources.
|
|
2387
|
+
*/
|
|
2388
|
+
contested: boolean;
|
|
2389
|
+
firstSeenRound: number;
|
|
2601
2390
|
}
|
|
2602
2391
|
/** The driver's accumulated research state — the completion oracle reads this. */
|
|
2603
2392
|
interface ResearchDrivingState {
|
|
2604
|
-
|
|
2605
|
-
|
|
2606
|
-
|
|
2607
|
-
|
|
2608
|
-
|
|
2609
|
-
|
|
2610
|
-
|
|
2611
|
-
|
|
2612
|
-
|
|
2613
|
-
|
|
2614
|
-
|
|
2615
|
-
|
|
2616
|
-
|
|
2617
|
-
|
|
2393
|
+
/** Every claim extracted from the worker's sources, by id. */
|
|
2394
|
+
claims: TrackedClaim[];
|
|
2395
|
+
/** Every deep sub-question raised, by id. */
|
|
2396
|
+
questions: DeepQuestion[];
|
|
2397
|
+
/** Claims with exactly one independent source AND not contested. */
|
|
2398
|
+
weaklySupported: TrackedClaim[];
|
|
2399
|
+
/** Claims supported by >= 2 independent sources. */
|
|
2400
|
+
corroborated: TrackedClaim[];
|
|
2401
|
+
/** Claims marked contested (a surfaced, unresolved contradiction). */
|
|
2402
|
+
contested: TrackedClaim[];
|
|
2403
|
+
/** Deep questions still unaddressed. */
|
|
2404
|
+
openQuestions: DeepQuestion[];
|
|
2405
|
+
/** How many rounds the driver has folded steer for. */
|
|
2406
|
+
rounds: number;
|
|
2618
2407
|
}
|
|
2619
2408
|
interface ResearchDrivingDriverOptions {
|
|
2620
|
-
|
|
2621
|
-
|
|
2622
|
-
|
|
2623
|
-
|
|
2624
|
-
|
|
2625
|
-
|
|
2626
|
-
|
|
2627
|
-
|
|
2628
|
-
|
|
2629
|
-
|
|
2630
|
-
|
|
2631
|
-
|
|
2632
|
-
|
|
2633
|
-
|
|
2634
|
-
|
|
2635
|
-
|
|
2636
|
-
|
|
2637
|
-
|
|
2638
|
-
|
|
2639
|
-
|
|
2409
|
+
/** Router client for claim extraction + deep-question generation. */
|
|
2410
|
+
router?: RouterClient;
|
|
2411
|
+
router_options?: TangleRouterOptions;
|
|
2412
|
+
/**
|
|
2413
|
+
* A claim is CORROBORATED at this many INDEPENDENT supporting sources (distinct
|
|
2414
|
+
* canonical hosts). Default 2 — the task's ">= 2 independent sources" bar.
|
|
2415
|
+
*/
|
|
2416
|
+
minIndependentSources?: number;
|
|
2417
|
+
/** Max deep sub-questions to fold into one round's steer. Default 6. */
|
|
2418
|
+
maxQuestionsPerRound?: number;
|
|
2419
|
+
/** Max claims to extract from a single source. Default 3. */
|
|
2420
|
+
maxClaimsPerSource?: number;
|
|
2421
|
+
/**
|
|
2422
|
+
* When the extractor LLM is unavailable, fall back to a deterministic claim
|
|
2423
|
+
* pull (the source's leading sentences) so the driver still drives. Default
|
|
2424
|
+
* true. Set false to require the model (claims will be empty without it).
|
|
2425
|
+
*/
|
|
2426
|
+
deterministicFallback?: boolean;
|
|
2427
|
+
/** Observe each round's generated steer (for instrumentation / the script). */
|
|
2428
|
+
onSteer?: (steer: ResearchDrivingSteer) => void;
|
|
2640
2429
|
}
|
|
2641
2430
|
/** What the driver folded into one round's worker prompt, surfaced for audit. */
|
|
2642
2431
|
interface ResearchDrivingSteer {
|
|
2643
|
-
|
|
2644
|
-
|
|
2645
|
-
|
|
2646
|
-
|
|
2647
|
-
|
|
2648
|
-
|
|
2649
|
-
|
|
2650
|
-
|
|
2432
|
+
round: number;
|
|
2433
|
+
deepQuestions: DeepQuestion[];
|
|
2434
|
+
/** Claims it demanded corroborating/refuting evidence for this round. */
|
|
2435
|
+
invalidationTargets: TrackedClaim[];
|
|
2436
|
+
/** The readiness gaps it interleaved (passed through from the loop). */
|
|
2437
|
+
gaps: KnowledgeGap[];
|
|
2438
|
+
/** The full steer text handed to the worker. */
|
|
2439
|
+
text: string;
|
|
2651
2440
|
}
|
|
2652
2441
|
/**
|
|
2653
2442
|
* The research-driving driver. It is a `ResearchDriver` (drops straight into
|
|
@@ -2655,25 +2444,45 @@ interface ResearchDrivingSteer {
|
|
|
2655
2444
|
* how `createAdaptiveResearchDriver` exposes `stats()`.
|
|
2656
2445
|
*/
|
|
2657
2446
|
interface ResearchDrivingDriver extends ResearchDriver {
|
|
2658
|
-
|
|
2659
|
-
|
|
2660
|
-
|
|
2661
|
-
|
|
2662
|
-
|
|
2663
|
-
|
|
2664
|
-
|
|
2665
|
-
|
|
2666
|
-
|
|
2667
|
-
|
|
2668
|
-
|
|
2669
|
-
|
|
2670
|
-
|
|
2671
|
-
|
|
2672
|
-
|
|
2447
|
+
/** Live snapshot of the claim ledger + deep questions. */
|
|
2448
|
+
researchState(): ResearchDrivingState;
|
|
2449
|
+
/**
|
|
2450
|
+
* The completion oracle — gate `done` on THIS, not on source count. True when
|
|
2451
|
+
* every deep sub-question is addressed AND every claim is corroborated
|
|
2452
|
+
* (>= `minIndependentSources` independent sources) or explicitly contested.
|
|
2453
|
+
* False while any claim is weakly-supported or any deep question is open.
|
|
2454
|
+
* Returns false before any claim has been seen (nothing researched yet).
|
|
2455
|
+
*/
|
|
2456
|
+
isComplete(): boolean;
|
|
2457
|
+
/**
|
|
2458
|
+
* The last round's generated steer, or undefined before the first fold. Useful
|
|
2459
|
+
* to assert the driver produced deeper questions / invalidation challenges.
|
|
2460
|
+
*/
|
|
2461
|
+
lastSteer(): ResearchDrivingSteer | undefined;
|
|
2673
2462
|
}
|
|
2674
2463
|
declare function createResearchDrivingDriver(options?: ResearchDrivingDriverOptions): ResearchDrivingDriver;
|
|
2675
|
-
|
|
2464
|
+
//#endregion
|
|
2465
|
+
//#region src/schemas.d.ts
|
|
2676
2466
|
declare const SourceAnchorSchema: z.ZodObject<{
|
|
2467
|
+
id: z.ZodString;
|
|
2468
|
+
sourceId: z.ZodString;
|
|
2469
|
+
label: z.ZodOptional<z.ZodString>;
|
|
2470
|
+
page: z.ZodOptional<z.ZodNumber>;
|
|
2471
|
+
lineStart: z.ZodOptional<z.ZodNumber>;
|
|
2472
|
+
lineEnd: z.ZodOptional<z.ZodNumber>;
|
|
2473
|
+
charStart: z.ZodOptional<z.ZodNumber>;
|
|
2474
|
+
charEnd: z.ZodOptional<z.ZodNumber>;
|
|
2475
|
+
timestampMs: z.ZodOptional<z.ZodNumber>;
|
|
2476
|
+
metadata: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
|
|
2477
|
+
}, z.core.$strip>;
|
|
2478
|
+
declare const SourceRecordSchema: z.ZodObject<{
|
|
2479
|
+
id: z.ZodString;
|
|
2480
|
+
uri: z.ZodString;
|
|
2481
|
+
title: z.ZodOptional<z.ZodString>;
|
|
2482
|
+
mediaType: z.ZodOptional<z.ZodString>;
|
|
2483
|
+
contentHash: z.ZodString;
|
|
2484
|
+
text: z.ZodOptional<z.ZodString>;
|
|
2485
|
+
anchors: z.ZodOptional<z.ZodArray<z.ZodObject<{
|
|
2677
2486
|
id: z.ZodString;
|
|
2678
2487
|
sourceId: z.ZodString;
|
|
2679
2488
|
label: z.ZodOptional<z.ZodString>;
|
|
@@ -2684,8 +2493,41 @@ declare const SourceAnchorSchema: z.ZodObject<{
|
|
|
2684
2493
|
charEnd: z.ZodOptional<z.ZodNumber>;
|
|
2685
2494
|
timestampMs: z.ZodOptional<z.ZodNumber>;
|
|
2686
2495
|
metadata: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
|
|
2496
|
+
}, z.core.$strip>>>;
|
|
2497
|
+
validUntil: z.ZodOptional<z.ZodISODateTime>;
|
|
2498
|
+
lastVerifiedAt: z.ZodOptional<z.ZodISODateTime>;
|
|
2499
|
+
metadata: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
|
|
2500
|
+
createdAt: z.ZodString;
|
|
2687
2501
|
}, z.core.$strip>;
|
|
2688
|
-
declare const
|
|
2502
|
+
declare const KnowledgePageSchema: z.ZodObject<{
|
|
2503
|
+
id: z.ZodString;
|
|
2504
|
+
path: z.ZodString;
|
|
2505
|
+
title: z.ZodString;
|
|
2506
|
+
text: z.ZodString;
|
|
2507
|
+
frontmatter: z.ZodRecord<z.ZodString, z.ZodUnknown>;
|
|
2508
|
+
sourceIds: z.ZodArray<z.ZodString>;
|
|
2509
|
+
tags: z.ZodArray<z.ZodString>;
|
|
2510
|
+
outLinks: z.ZodArray<z.ZodString>;
|
|
2511
|
+
}, z.core.$strip>;
|
|
2512
|
+
declare const KnowledgeGraphNodeSchema: z.ZodObject<{
|
|
2513
|
+
id: z.ZodString;
|
|
2514
|
+
title: z.ZodString;
|
|
2515
|
+
path: z.ZodString;
|
|
2516
|
+
tags: z.ZodArray<z.ZodString>;
|
|
2517
|
+
sourceIds: z.ZodArray<z.ZodString>;
|
|
2518
|
+
outDegree: z.ZodNumber;
|
|
2519
|
+
inDegree: z.ZodNumber;
|
|
2520
|
+
}, z.core.$strip>;
|
|
2521
|
+
declare const KnowledgeGraphEdgeSchema: z.ZodObject<{
|
|
2522
|
+
source: z.ZodString;
|
|
2523
|
+
target: z.ZodString;
|
|
2524
|
+
weight: z.ZodNumber;
|
|
2525
|
+
reasons: z.ZodArray<z.ZodString>;
|
|
2526
|
+
}, z.core.$strip>;
|
|
2527
|
+
declare const KnowledgeIndexSchema: z.ZodObject<{
|
|
2528
|
+
root: z.ZodString;
|
|
2529
|
+
generatedAt: z.ZodString;
|
|
2530
|
+
sources: z.ZodArray<z.ZodObject<{
|
|
2689
2531
|
id: z.ZodString;
|
|
2690
2532
|
uri: z.ZodString;
|
|
2691
2533
|
title: z.ZodOptional<z.ZodString>;
|
|
@@ -2693,23 +2535,23 @@ declare const SourceRecordSchema: z.ZodObject<{
|
|
|
2693
2535
|
contentHash: z.ZodString;
|
|
2694
2536
|
text: z.ZodOptional<z.ZodString>;
|
|
2695
2537
|
anchors: z.ZodOptional<z.ZodArray<z.ZodObject<{
|
|
2696
|
-
|
|
2697
|
-
|
|
2698
|
-
|
|
2699
|
-
|
|
2700
|
-
|
|
2701
|
-
|
|
2702
|
-
|
|
2703
|
-
|
|
2704
|
-
|
|
2705
|
-
|
|
2538
|
+
id: z.ZodString;
|
|
2539
|
+
sourceId: z.ZodString;
|
|
2540
|
+
label: z.ZodOptional<z.ZodString>;
|
|
2541
|
+
page: z.ZodOptional<z.ZodNumber>;
|
|
2542
|
+
lineStart: z.ZodOptional<z.ZodNumber>;
|
|
2543
|
+
lineEnd: z.ZodOptional<z.ZodNumber>;
|
|
2544
|
+
charStart: z.ZodOptional<z.ZodNumber>;
|
|
2545
|
+
charEnd: z.ZodOptional<z.ZodNumber>;
|
|
2546
|
+
timestampMs: z.ZodOptional<z.ZodNumber>;
|
|
2547
|
+
metadata: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
|
|
2706
2548
|
}, z.core.$strip>>>;
|
|
2707
2549
|
validUntil: z.ZodOptional<z.ZodISODateTime>;
|
|
2708
2550
|
lastVerifiedAt: z.ZodOptional<z.ZodISODateTime>;
|
|
2709
2551
|
metadata: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
|
|
2710
2552
|
createdAt: z.ZodString;
|
|
2711
|
-
}, z.core.$strip
|
|
2712
|
-
|
|
2553
|
+
}, z.core.$strip>>;
|
|
2554
|
+
pages: z.ZodArray<z.ZodObject<{
|
|
2713
2555
|
id: z.ZodString;
|
|
2714
2556
|
path: z.ZodString;
|
|
2715
2557
|
title: z.ZodString;
|
|
@@ -2718,147 +2560,97 @@ declare const KnowledgePageSchema: z.ZodObject<{
|
|
|
2718
2560
|
sourceIds: z.ZodArray<z.ZodString>;
|
|
2719
2561
|
tags: z.ZodArray<z.ZodString>;
|
|
2720
2562
|
outLinks: z.ZodArray<z.ZodString>;
|
|
2721
|
-
}, z.core.$strip
|
|
2722
|
-
|
|
2723
|
-
|
|
2724
|
-
|
|
2725
|
-
|
|
2726
|
-
|
|
2727
|
-
|
|
2728
|
-
|
|
2729
|
-
|
|
2730
|
-
|
|
2731
|
-
declare const KnowledgeGraphEdgeSchema: z.ZodObject<{
|
|
2732
|
-
source: z.ZodString;
|
|
2733
|
-
target: z.ZodString;
|
|
2734
|
-
weight: z.ZodNumber;
|
|
2735
|
-
reasons: z.ZodArray<z.ZodString>;
|
|
2736
|
-
}, z.core.$strip>;
|
|
2737
|
-
declare const KnowledgeIndexSchema: z.ZodObject<{
|
|
2738
|
-
root: z.ZodString;
|
|
2739
|
-
generatedAt: z.ZodString;
|
|
2740
|
-
sources: z.ZodArray<z.ZodObject<{
|
|
2741
|
-
id: z.ZodString;
|
|
2742
|
-
uri: z.ZodString;
|
|
2743
|
-
title: z.ZodOptional<z.ZodString>;
|
|
2744
|
-
mediaType: z.ZodOptional<z.ZodString>;
|
|
2745
|
-
contentHash: z.ZodString;
|
|
2746
|
-
text: z.ZodOptional<z.ZodString>;
|
|
2747
|
-
anchors: z.ZodOptional<z.ZodArray<z.ZodObject<{
|
|
2748
|
-
id: z.ZodString;
|
|
2749
|
-
sourceId: z.ZodString;
|
|
2750
|
-
label: z.ZodOptional<z.ZodString>;
|
|
2751
|
-
page: z.ZodOptional<z.ZodNumber>;
|
|
2752
|
-
lineStart: z.ZodOptional<z.ZodNumber>;
|
|
2753
|
-
lineEnd: z.ZodOptional<z.ZodNumber>;
|
|
2754
|
-
charStart: z.ZodOptional<z.ZodNumber>;
|
|
2755
|
-
charEnd: z.ZodOptional<z.ZodNumber>;
|
|
2756
|
-
timestampMs: z.ZodOptional<z.ZodNumber>;
|
|
2757
|
-
metadata: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
|
|
2758
|
-
}, z.core.$strip>>>;
|
|
2759
|
-
validUntil: z.ZodOptional<z.ZodISODateTime>;
|
|
2760
|
-
lastVerifiedAt: z.ZodOptional<z.ZodISODateTime>;
|
|
2761
|
-
metadata: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
|
|
2762
|
-
createdAt: z.ZodString;
|
|
2563
|
+
}, z.core.$strip>>;
|
|
2564
|
+
graph: z.ZodObject<{
|
|
2565
|
+
nodes: z.ZodArray<z.ZodObject<{
|
|
2566
|
+
id: z.ZodString;
|
|
2567
|
+
title: z.ZodString;
|
|
2568
|
+
path: z.ZodString;
|
|
2569
|
+
tags: z.ZodArray<z.ZodString>;
|
|
2570
|
+
sourceIds: z.ZodArray<z.ZodString>;
|
|
2571
|
+
outDegree: z.ZodNumber;
|
|
2572
|
+
inDegree: z.ZodNumber;
|
|
2763
2573
|
}, z.core.$strip>>;
|
|
2764
|
-
|
|
2765
|
-
|
|
2766
|
-
|
|
2767
|
-
|
|
2768
|
-
|
|
2769
|
-
frontmatter: z.ZodRecord<z.ZodString, z.ZodUnknown>;
|
|
2770
|
-
sourceIds: z.ZodArray<z.ZodString>;
|
|
2771
|
-
tags: z.ZodArray<z.ZodString>;
|
|
2772
|
-
outLinks: z.ZodArray<z.ZodString>;
|
|
2574
|
+
edges: z.ZodArray<z.ZodObject<{
|
|
2575
|
+
source: z.ZodString;
|
|
2576
|
+
target: z.ZodString;
|
|
2577
|
+
weight: z.ZodNumber;
|
|
2578
|
+
reasons: z.ZodArray<z.ZodString>;
|
|
2773
2579
|
}, z.core.$strip>>;
|
|
2774
|
-
|
|
2775
|
-
nodes: z.ZodArray<z.ZodObject<{
|
|
2776
|
-
id: z.ZodString;
|
|
2777
|
-
title: z.ZodString;
|
|
2778
|
-
path: z.ZodString;
|
|
2779
|
-
tags: z.ZodArray<z.ZodString>;
|
|
2780
|
-
sourceIds: z.ZodArray<z.ZodString>;
|
|
2781
|
-
outDegree: z.ZodNumber;
|
|
2782
|
-
inDegree: z.ZodNumber;
|
|
2783
|
-
}, z.core.$strip>>;
|
|
2784
|
-
edges: z.ZodArray<z.ZodObject<{
|
|
2785
|
-
source: z.ZodString;
|
|
2786
|
-
target: z.ZodString;
|
|
2787
|
-
weight: z.ZodNumber;
|
|
2788
|
-
reasons: z.ZodArray<z.ZodString>;
|
|
2789
|
-
}, z.core.$strip>>;
|
|
2790
|
-
}, z.core.$strip>;
|
|
2580
|
+
}, z.core.$strip>;
|
|
2791
2581
|
}, z.core.$strip>;
|
|
2792
2582
|
declare const KnowledgeEventSchema: z.ZodObject<{
|
|
2793
|
-
|
|
2794
|
-
|
|
2795
|
-
|
|
2796
|
-
|
|
2797
|
-
|
|
2798
|
-
|
|
2799
|
-
|
|
2800
|
-
|
|
2801
|
-
|
|
2802
|
-
|
|
2803
|
-
|
|
2804
|
-
|
|
2805
|
-
|
|
2806
|
-
|
|
2583
|
+
id: z.ZodString;
|
|
2584
|
+
type: z.ZodEnum<{
|
|
2585
|
+
"index.built": "index.built";
|
|
2586
|
+
"lint.run": "lint.run";
|
|
2587
|
+
"optimization.run": "optimization.run";
|
|
2588
|
+
"proposal.applied": "proposal.applied";
|
|
2589
|
+
"release.promoted": "release.promoted";
|
|
2590
|
+
"release.rejected": "release.rejected";
|
|
2591
|
+
"source.added": "source.added";
|
|
2592
|
+
}>;
|
|
2593
|
+
createdAt: z.ZodString;
|
|
2594
|
+
actor: z.ZodOptional<z.ZodString>;
|
|
2595
|
+
target: z.ZodOptional<z.ZodString>;
|
|
2596
|
+
metadata: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
|
|
2807
2597
|
}, z.core.$strip>;
|
|
2808
2598
|
declare const KnowledgeBaseCandidateSchema: z.ZodObject<{
|
|
2599
|
+
id: z.ZodString;
|
|
2600
|
+
units: z.ZodArray<z.ZodObject<{
|
|
2809
2601
|
id: z.ZodString;
|
|
2810
|
-
|
|
2811
|
-
|
|
2812
|
-
|
|
2813
|
-
|
|
2814
|
-
|
|
2815
|
-
|
|
2816
|
-
|
|
2817
|
-
|
|
2818
|
-
|
|
2819
|
-
|
|
2820
|
-
|
|
2821
|
-
|
|
2822
|
-
|
|
2823
|
-
|
|
2824
|
-
|
|
2825
|
-
|
|
2826
|
-
|
|
2827
|
-
|
|
2828
|
-
|
|
2829
|
-
|
|
2830
|
-
|
|
2831
|
-
|
|
2832
|
-
|
|
2833
|
-
|
|
2834
|
-
|
|
2835
|
-
|
|
2836
|
-
|
|
2837
|
-
|
|
2838
|
-
sourceIds: z.ZodOptional<z.ZodArray<z.ZodString>>;
|
|
2839
|
-
tags: z.ZodOptional<z.ZodArray<z.ZodString>>;
|
|
2840
|
-
metadata: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
|
|
2841
|
-
updatedAt: z.ZodOptional<z.ZodString>;
|
|
2842
|
-
}, z.core.$strip>>;
|
|
2843
|
-
retrievalPolicy: z.ZodOptional<z.ZodString>;
|
|
2844
|
-
synthesisPolicy: z.ZodOptional<z.ZodString>;
|
|
2845
|
-
questionPolicy: z.ZodOptional<z.ZodString>;
|
|
2846
|
-
updatePolicy: z.ZodOptional<z.ZodString>;
|
|
2602
|
+
title: z.ZodString;
|
|
2603
|
+
text: z.ZodString;
|
|
2604
|
+
claims: z.ZodOptional<z.ZodArray<z.ZodObject<{
|
|
2605
|
+
id: z.ZodString;
|
|
2606
|
+
text: z.ZodString;
|
|
2607
|
+
refs: z.ZodArray<z.ZodObject<{
|
|
2608
|
+
sourceId: z.ZodString;
|
|
2609
|
+
anchorId: z.ZodOptional<z.ZodString>;
|
|
2610
|
+
quote: z.ZodOptional<z.ZodString>;
|
|
2611
|
+
}, z.core.$strip>>;
|
|
2612
|
+
confidence: z.ZodOptional<z.ZodNumber>;
|
|
2613
|
+
status: z.ZodOptional<z.ZodEnum<{
|
|
2614
|
+
active: "active";
|
|
2615
|
+
draft: "draft";
|
|
2616
|
+
rejected: "rejected";
|
|
2617
|
+
superseded: "superseded";
|
|
2618
|
+
}>>;
|
|
2619
|
+
metadata: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
|
|
2620
|
+
}, z.core.$strip>>>;
|
|
2621
|
+
relations: z.ZodOptional<z.ZodArray<z.ZodObject<{
|
|
2622
|
+
sourceId: z.ZodString;
|
|
2623
|
+
targetId: z.ZodString;
|
|
2624
|
+
predicate: z.ZodString;
|
|
2625
|
+
weight: z.ZodOptional<z.ZodNumber>;
|
|
2626
|
+
metadata: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
|
|
2627
|
+
}, z.core.$strip>>>;
|
|
2628
|
+
sourceIds: z.ZodOptional<z.ZodArray<z.ZodString>>;
|
|
2629
|
+
tags: z.ZodOptional<z.ZodArray<z.ZodString>>;
|
|
2847
2630
|
metadata: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
|
|
2631
|
+
updatedAt: z.ZodOptional<z.ZodString>;
|
|
2632
|
+
}, z.core.$strip>>;
|
|
2633
|
+
retrievalPolicy: z.ZodOptional<z.ZodString>;
|
|
2634
|
+
synthesisPolicy: z.ZodOptional<z.ZodString>;
|
|
2635
|
+
questionPolicy: z.ZodOptional<z.ZodString>;
|
|
2636
|
+
updatePolicy: z.ZodOptional<z.ZodString>;
|
|
2637
|
+
metadata: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
|
|
2848
2638
|
}, z.core.$strip>;
|
|
2849
|
-
|
|
2639
|
+
//#endregion
|
|
2640
|
+
//#region src/search.d.ts
|
|
2850
2641
|
declare function searchKnowledge(index: KnowledgeIndex, query: string, limit?: number): KnowledgeSearchResult[];
|
|
2851
2642
|
declare function tokenizeQuery(query: string): string[];
|
|
2852
2643
|
declare function reciprocalRankFusion(rankLists: string[][], k?: number): Map<string, number>;
|
|
2853
|
-
|
|
2644
|
+
//#endregion
|
|
2645
|
+
//#region src/store.d.ts
|
|
2854
2646
|
interface KnowledgeLayout {
|
|
2855
|
-
|
|
2856
|
-
|
|
2857
|
-
|
|
2858
|
-
|
|
2859
|
-
|
|
2860
|
-
|
|
2861
|
-
|
|
2647
|
+
root: string;
|
|
2648
|
+
knowledgeDir: string;
|
|
2649
|
+
rawSourcesDir: string;
|
|
2650
|
+
sourceRegistryPath: string;
|
|
2651
|
+
indexPath: string;
|
|
2652
|
+
logPath: string;
|
|
2653
|
+
cacheDir: string;
|
|
2862
2654
|
}
|
|
2863
2655
|
declare function layoutFor(root: string): KnowledgeLayout;
|
|
2864
2656
|
/**
|
|
@@ -2881,12 +2673,15 @@ declare function isScaffoldPath(path: string): boolean;
|
|
|
2881
2673
|
declare function initKnowledgeBase(root: string): Promise<KnowledgeLayout>;
|
|
2882
2674
|
declare function loadKnowledgePages(root: string): Promise<KnowledgePage[]>;
|
|
2883
2675
|
declare function writeJson(path: string, value: unknown): Promise<void>;
|
|
2884
|
-
|
|
2676
|
+
//#endregion
|
|
2677
|
+
//#region src/wikilinks.d.ts
|
|
2885
2678
|
declare const WIKILINK_REGEX: RegExp;
|
|
2886
2679
|
declare function extractWikilinks(content: string): string[];
|
|
2887
2680
|
declare function normalizeLinkTarget(target: string): string;
|
|
2888
|
-
|
|
2681
|
+
//#endregion
|
|
2682
|
+
//#region src/write-protocol.d.ts
|
|
2889
2683
|
declare function isSafeKnowledgePath(path: string, allowedPrefixes?: string[]): boolean;
|
|
2890
2684
|
declare function parseKnowledgeWriteBlocks(text: string, allowedPrefixes?: string[]): KnowledgeWriteParseResult;
|
|
2891
|
-
|
|
2892
|
-
export { type AdaptiveDecision, type AdaptiveDriverOptions, type AdaptiveResearchDriver, type AdaptiveStats, type AddSourceOptions, type AddSourceTextInput, type ApplyWriteBlocksResult, type BuildEvalKnowledgeBundleOptions, type ChunkingOptions, type ClaimGroundingDriverOptions, type CompanyEvalCase, type D1Adapter, type DedupReason, type DeepQuestion, type DeepQuestionKind, type DefineReadinessSpecInput, type DetectChangesOptions, type DetectChangesResult, type DiscoveryLoopResult, type DiscoveryLoopRound, type DiscoveryLoopStopReason, type DiscoveryResult, type DiscoveryTask, type DriverResearchContext, type EvalKnowledgeBundleBuildResult, type EvaluateKnowledgeBaseReadinessOptions, type ExpectedGroup, type ExternalRagEvalScore, type FactResult, type FileSystemFreshnessStoreOptions, FileSystemKbStore, type FileSystemSearchOptions, FileSystemSearchProvider, type FileSystemSearchProviderOptions, type FreshnessKey, type FreshnessMark, type FreshnessRecord, type FreshnessTtl, type GroundClaimOptions, type GroundingResult, type KbStore, KnowledgeBaseCandidateSchema, type KnowledgeBaseQualityOptions, type KnowledgeBaseQualityReport, type KnowledgeBaseReadinessEvaluation, type KnowledgeChange, type KnowledgeChangeKind, type KnowledgeChunk, KnowledgeClaim, type KnowledgeControlLoopAction, type KnowledgeControlLoopActionResult, type KnowledgeControlLoopAdapter, type KnowledgeControlLoopAdapterOptions, type KnowledgeControlLoopState, type KnowledgeDiscoveryDispatcher, type KnowledgeDiscoveryWorker, KnowledgeEvent, type KnowledgeEventQuery, KnowledgeEventSchema, KnowledgeEventType, type KnowledgeExplanation, KnowledgeFragment, type KnowledgeFreshnessStore, type KnowledgeGap, KnowledgeGraph, KnowledgeGraphEdgeSchema, KnowledgeGraphNodeSchema, type KnowledgeImprovementActivationPersistence, type KnowledgeImprovementCandidateRecord, type KnowledgeImprovementCandidateRef, KnowledgeImprovementCandidateRefSchema, type KnowledgeImprovementEvaluationInput, type KnowledgeImprovementEvaluator, type KnowledgeImprovementEvent, type KnowledgeImprovementEvidence, KnowledgeImprovementEvidenceSchema, type KnowledgeImprovementMetric, type KnowledgeImprovementMetricProvenance, type KnowledgeImprovementMutationReceipt, type KnowledgeImprovementMutationResult, type KnowledgeImprovementOptions, type KnowledgeImprovementRagOptimizationOptions, type KnowledgeImprovementRagOptimizationRunInput, type KnowledgeImprovementResult, type KnowledgeImprovementRetrievalOptions, type KnowledgeImprovementRunState, KnowledgeImprovementRunStateSchema, type KnowledgeImprovementStatus, type KnowledgeImprovementTarget, type KnowledgeImprovementUpdate, type KnowledgeImprovementUpdateInput, KnowledgeIndex, KnowledgeIndexSchema, type KnowledgeInspection, type KnowledgeLayout, KnowledgeLintFinding, KnowledgePage, KnowledgePageSchema, type KnowledgePolicyDispatch, type KnowledgeProposal, KnowledgeProposalParseError, type KnowledgeReadinessSpec, KnowledgeRelease, type KnowledgeReleaseInput, type KnowledgeReleaseReport, type KnowledgeResearchLoopContext, type KnowledgeResearchLoopDecision, type KnowledgeResearchLoopResult, type KnowledgeResearchLoopStep, KnowledgeSearchResult, KnowledgeWriteBlock, KnowledgeWriteParseResult, type LoadKnowledgeImprovementActivationResultOptions, type MaterialFact, type MaterialFactLens, type MaterialFactsResult, MemoryKbStore, type OptimizeKnowledgeBasePolicyOptions, type OptimizeKnowledgeBasePolicyResult, type ParsedFrontmatter, type PendingKnowledgeMutation, type PromoteKnowledgeCandidateOptions, type ProposeFromFindingsResult, READINESS_SPEC_DEFAULTS, type RagAnswerEvalArtifact, type RagAnswerEvalCase, type RagAnswerEvalScenario, type RagAnswerMetricSummary, type RagAnswerQualityHookOptions, type RagAnswerQualityInput, type RagAnswerQualityJudgeOptions, type RagAnswerQualityResult, type RagCalibrationOptions, type RagCalibrationResult, type RagDiagnosisInput, type RagEvalCitation, type RagEvalClaim, type RagEvalContext, type RagEvalMetricKey, type RagEvalProvider, type RagEvalSlice, type RagGapFinding, type RagGapKind, type RagGapSeverity, type RagKnowledgeAcquisitionInput, type RagKnowledgeImprovementPhase, type RagKnowledgeImprovementPhaseResult, type RagKnowledgeImprovementPhaseStatus, type RagKnowledgeResearchOptions, type RagKnowledgeUpdateInput, type RagKnowledgeUpdateResult, type RagOptimizationConfig, type RagOptimizationSelection, type RagPhaseInputBase, type RagPromotionInput, type RagPromotionResult, type RagRequiredContext, type RecoverPendingKnowledgeMutationOptions, type RejectedSource, type ResearchContribution, type ResearchDriver, type ResearchDrivingDriver, type ResearchDrivingDriverOptions, type ResearchDrivingState, type ResearchDrivingSteer, type ResearchSourceProposal, type ResearchWorker, type ResolvedKnowledgeImprovementCandidate, type ResolvedKnowledgeImprovementComparison, type ResolvedKnowledgeImprovementComparisonSnapshot, type RestoreKnowledgeCandidateBaselineOptions, RetrievalConfig, RetrievalEvalArtifact, RetrievalEvalRetriever, RetrievalEvalScenario, RetrievalMetricWeights, type RetrievalOptimizationSelection, RetrievedKnowledgeHit, type RouterClient, RouterError, type RouterUsage, type RunDiscoveryLoopOptions, type RunKnowledgeResearchLoopOptions, type RunRagKnowledgeImprovementLoopOptions, type RunRagKnowledgeImprovementLoopResult, type RunRagOptimizationOptions, type RunRagOptimizationResult, type RunRetrievalImprovementLoopOptions, type RunRetrievalImprovementLoopResult, RunSerializedKnowledgeOptimizationOptions, RunSerializedKnowledgeOptimizationResult, SCAFFOLD_PAGE_BASENAMES, type SourceAdapter, type SourceAdapterInput, type SourceAdapterOutput, SourceAnchorSchema, type SourceFreshnessInspection, SourceRecord, SourceRecordSchema, SourceRegistry, type SourceVerdict, type SourceVerificationContext, type TangleRouterOptions, type ThesisRunOptions, type ThesisRunResult, type ThesisTaskInput, type TrackedClaim, type TriageClass, type TwoAgentResearchLoopOptions, type TwoAgentResearchLoopResult, type TwoAgentResearchRound, type UseKnowledgeImprovementCandidateOptions, type ValidateKnowledgeOptions, type ValidateKnowledgeResult, type VerifiedResearchLoopOptions, type VerifiedResearchLoopResult, type VerifiedResearchRound, type VerifyingDriverOptions, WIKILINK_REGEX, type WebResearchWorkerOptions, type WebSearchHit, type WorkerClaimDecorationOptions, type WorkerResearchContext, addSourcePath, addSourceText, applyKnowledgeWriteBlocks, applyKnowledgeWriteBlocksFile, buildEvalKnowledgeBundle, buildKnowledgeGraph, buildKnowledgeIndex, calibrateRagAnswerJudge, canonicalizeUrl, chunkMarkdown, citedClaimKey, citedClaimOf, contentKey, createAdaptiveResearchDriver, createClaimDecorator, createClaimGroundingVerifier, createCollectionResearchDriver, createD1FreshnessStoreStub, createFileSystemFreshnessStore, createFileSystemSearchProvider, createKnowledgeControlLoopAdapter, createKnowledgeEvent, createLocalDiscoveryDispatcher, createRagAnswerQualityHook, createResearchDrivingDriver, createTangleRouterClient, createVerifyingResearchDriver, createWebResearchWorker, defineReadinessSpec, detectChanges, diagnoseRagAnswerFailure, evaluateKnowledgeBaseReadiness, explainKnowledgeTarget, extractWikilinks, formatFrontmatter, fromAgentCandidateKnowledgeRef, gradeCompanyAgainstText, gradeFactAgainstText, groundClaimInText, hashKnowledgeBase, improveKnowledgeBase, initKnowledgeBase, inspectKnowledgeIndex, inspectPendingKnowledgeMutation, investmentThesisSet, isSafeKnowledgePath, isScaffoldPath, kbIndexToText, knowledgeImprovementCandidateRef, knowledgeImprovementRunDir, knowledgeImprovementRunId, knowledgeReleaseReport, layoutFor, lensDistribution, lintKnowledgeIndex, loadKnowledgeImprovementActivationResult, loadKnowledgeImprovementEvents, loadKnowledgeImprovementState, loadKnowledgePages, loadSourceRegistry, materialFactsSurfaced, materialFactsSurfacedInText, mediaTypeFor, normalizeExternalRagScores, normalizeLinkTarget, optimizeKnowledgeBasePolicy, parseFrontmatter, parseKnowledgeWriteBlocks, promoteKnowledgeCandidate, proposeFromFinding, proposeFromFindings, ragAnswerQualityJudge, reciprocalRankFusion, recoverPendingKnowledgeMutation, restoreKnowledgeCandidateBaseline, runDiscoveryLoop, runInvestmentThesisTask, runKnowledgeResearchLoop, runRagKnowledgeImprovementLoop, runRagOptimization, runRetrievalImprovementLoop, runTwoAgentResearchLoop, runVerifiedResearchLoop, scoreKnowledgeBaseIndex, scoreRagAnswerArtifact, searchKnowledge, sha256, slugify, sourceMatchesGaps, sourceRegistryPath, stableId, stripFrontmatter, textSourceAdapter, thesisReadinessSpecs, toAgentCandidateKnowledgeRef, toDeepEvalTestCases, toRagCheckerRecords, toRagasEvaluationRows, toTruLensRecords, tokenizeQuery, totalMaterialFacts, triageSource, validateKnowledgeIndex, withCitedClaim, withKnowledgeImprovementCandidate, withKnowledgeImprovementComparison, writeJson, writeKnowledgeIndex, writeSourceRegistry };
|
|
2685
|
+
//#endregion
|
|
2686
|
+
export { AdaptiveDecision, AdaptiveDriverOptions, AdaptiveResearchDriver, AdaptiveStats, AddSourceOptions, AddSourceTextInput, AgentMemoryAcquireRunLease, type AgentMemoryActivation, type AgentMemoryActivationDriver, AgentMemoryAdapter, type AgentMemoryAttemptEvent, AgentMemoryBranch, AgentMemoryBranchIsolation, AgentMemoryBranchLifetime, AgentMemoryBranchSnapshot, AgentMemoryContext, AgentMemoryControllerMode, type AgentMemoryDimensionComparison, type AgentMemoryExecutionContext, type AgentMemoryExecutionCostMeter, type AgentMemoryExecutionCostReceipt, type AgentMemoryExecutionPaidCallInput, type AgentMemoryExecutionPaidCallResult, type AgentMemoryExecutionStep, type AgentMemoryExperimentCandidate, type AgentMemoryExperimentRankingRow, type AgentMemoryExperimentRunLease, type AgentMemoryFinalEvaluation, type AgentMemoryFinalPair, AgentMemoryHit, AgentMemoryHitSchema, type AgentMemoryImprovementRunLease, AgentMemoryJournalEntry, AgentMemoryKind, AgentMemoryKindSchema, AgentMemoryLifecycleTimeoutError, AgentMemoryLifecycleUnsafeError, type AgentMemoryPromotionDecision, AgentMemoryRunLease, AgentMemoryScope, AgentMemoryScopeSchema, AgentMemorySearchOptions, type AgentMemorySequence, type AgentMemorySequenceArtifact, type AgentMemorySequenceProbe, type AgentMemorySequenceProbeResult, type AgentMemorySequenceScenario, type AgentMemorySequenceStep, AgentMemorySharingPolicy, AgentMemoryVisibility, AgentMemoryWriteInput, AgentMemoryWriteInputSchema, AgentMemoryWriteResult, ApplyWriteBlocksResult, type BuildAgentMemorySequencesFromBenchmarkCasesOptions, BuildEvalKnowledgeBundleOptions, type BuildRetrievalBenchmarkCasesFromQrelsOptions, BuildRetrievalEvalDispatchOptions, ChunkingOptions, ClaimGroundingDriverOptions, ClaimRef, CompanyEvalCase, CornellLiiSelector, CornellLiiSourceOptions, CreateAgentMemoryBranchOptions, D1Adapter, DEFAULT_MEMORY_CLEANUP_TIMEOUT_MS, DedupReason, DeepQuestion, DeepQuestionKind, DefineReadinessSpecInput, DetectChangesOptions, DetectChangesResult, DiscoveryLoopResult, DiscoveryLoopRound, DiscoveryLoopStopReason, DiscoveryResult, DiscoveryTask, DriverResearchContext, EvalKnowledgeBundleBuildResult, EvaluateKnowledgeBaseReadinessOptions, ExpectedGroup, type ExternalRagEvalScore, FactResult, FetchOpts, FileSystemFreshnessStoreOptions, FileSystemKbStore, FileSystemSearchOptions, FileSystemSearchProvider, FileSystemSearchProviderOptions, ForkAgentMemoryBranchSnapshotOptions, FragmentProvenance, FreshnessKey, FreshnessMark, FreshnessRecord, FreshnessTtl, GraphitiMcpClientLike, GraphitiMemoryAdapterOptions, GraphitiToolNames, GroundClaimOptions, GroundingResult, INDUSTRY_MEMORY_BENCHMARKS, INDUSTRY_RAG_BENCHMARKS, IRS_DIMENSION_HINTS, IrsPublicationsSourceOptions, KbStore, type KnowledgeAnswerBenchmarkCase, type KnowledgeAnswerBenchmarkTaskKind, KnowledgeBaseCandidate, KnowledgeBaseCandidateSchema, type KnowledgeBaseQualityOptions, type KnowledgeBaseQualityReport, KnowledgeBaseReadinessEvaluation, type KnowledgeBenchmarkArtifact, type KnowledgeBenchmarkCase, type KnowledgeBenchmarkCaseBase, type KnowledgeBenchmarkDistribution, type KnowledgeBenchmarkEvaluation, type KnowledgeBenchmarkFamily, type KnowledgeBenchmarkReport, type KnowledgeBenchmarkResponder, type KnowledgeBenchmarkScenario, type KnowledgeBenchmarkSliceSummary, type KnowledgeBenchmarkSource, type KnowledgeBenchmarkSpec, type KnowledgeBenchmarkSplit, type KnowledgeBenchmarkTaskKind, KnowledgeChange, KnowledgeChangeKind, KnowledgeChunk, KnowledgeClaim, type KnowledgeClaimMatcher, KnowledgeControlLoopAction, KnowledgeControlLoopActionResult, KnowledgeControlLoopAdapter, KnowledgeControlLoopAdapterOptions, KnowledgeControlLoopState, KnowledgeDiscoveryDispatcher, KnowledgeDiscoveryWorker, KnowledgeEvent, KnowledgeEventQuery, KnowledgeEventSchema, KnowledgeEventType, KnowledgeExplanation, KnowledgeFragment, KnowledgeFreshnessStore, KnowledgeGap, KnowledgeGraph, KnowledgeGraphEdge, KnowledgeGraphEdgeSchema, KnowledgeGraphNode, KnowledgeGraphNodeSchema, KnowledgeId, type KnowledgeImprovementActivationPersistence, type KnowledgeImprovementCandidateRecord, type KnowledgeImprovementCandidateRef, KnowledgeImprovementCandidateRefSchema, type KnowledgeImprovementEvaluationInput, type KnowledgeImprovementEvaluator, type KnowledgeImprovementEvent, type KnowledgeImprovementEvidence, KnowledgeImprovementEvidenceSchema, type KnowledgeImprovementMetric, type KnowledgeImprovementMetricProvenance, type KnowledgeImprovementMutationReceipt, type KnowledgeImprovementMutationResult, type KnowledgeImprovementOptions, type KnowledgeImprovementRagOptimizationOptions, type KnowledgeImprovementRagOptimizationRunInput, type KnowledgeImprovementResult, type KnowledgeImprovementRetrievalOptions, type KnowledgeImprovementRunState, KnowledgeImprovementRunStateSchema, type KnowledgeImprovementStatus, type KnowledgeImprovementTarget, type KnowledgeImprovementUpdate, type KnowledgeImprovementUpdateInput, KnowledgeIndex, KnowledgeIndexSchema, KnowledgeInspection, KnowledgeLayout, KnowledgeLintFinding, type KnowledgeMemoryBenchmarkCase, type KnowledgeMemoryBenchmarkTaskKind, type KnowledgeMemoryEvent, type KnowledgeMemoryFactMatcher, KnowledgePage, KnowledgePageSchema, KnowledgePolicy, type KnowledgePolicyDispatch, KnowledgeProposal, KnowledgeProposalParseError, KnowledgeReadinessSpec, KnowledgeRelation, KnowledgeRelease, KnowledgeReleaseInput, KnowledgeReleaseReport, KnowledgeResearchLoopContext, KnowledgeResearchLoopDecision, KnowledgeResearchLoopResult, KnowledgeResearchLoopStep, type KnowledgeRetrievalBenchmarkCase, type KnowledgeRetrievalBenchmarkQrel, type KnowledgeRetrievalBenchmarkQuery, KnowledgeSearchResult, KnowledgeSource, KnowledgeUnit, KnowledgeWriteBlock, KnowledgeWriteParseResult, type LoadKnowledgeImprovementActivationResultOptions, MAX_RESPONSE_BYTES, MIN_REQUEST_GAP_MS, MaterialFact, MaterialFactLens, MaterialFactsResult, Mem0ClientMode, Mem0HostedClient, Mem0HostedMemoryAdapterOptions, Mem0MemoryAdapterOptions, Mem0OssClient, Mem0OssMemoryAdapterOptions, type MemoryAdapterBenchmarkCandidate, type MemoryAdapterBenchmarkRankingRow, type MemoryConfigScenario, MemoryKbStore, Neo4jAgentMemoryAdapterOptions, type OptimizeKnowledgeBasePolicyOptions, type OptimizeKnowledgeBasePolicyResult, OwnedAgentMemoryRunLease, POLITE_USER_AGENT, ParsedFrontmatter, PartitionRetrievalScenariosOptions, type PendingKnowledgeMutation, PoliteFetchOptions, PoliteFetchResult, type PromoteKnowledgeCandidateOptions, ProposeFromFindingsResult, READINESS_SPEC_DEFAULTS, type RagAnswerEvalArtifact, type RagAnswerEvalCase, type RagAnswerEvalScenario, type RagAnswerMetricSummary, type RagAnswerQualityHookOptions, RagAnswerQualityInput, type RagAnswerQualityJudgeOptions, RagAnswerQualityResult, type RagCalibrationOptions, type RagCalibrationResult, RagDiagnosisInput, type RagEvalCitation, type RagEvalClaim, type RagEvalContext, type RagEvalMetricKey, type RagEvalProvider, type RagEvalSlice, RagGapFinding, RagGapKind, RagGapSeverity, RagKnowledgeAcquisitionInput, RagKnowledgeImprovementPhase, RagKnowledgeImprovementPhaseResult, RagKnowledgeImprovementPhaseStatus, RagKnowledgeResearchOptions, RagKnowledgeUpdateInput, RagKnowledgeUpdateResult, RagOptimizationConfig, RagOptimizationSelection, RagPhaseInputBase, RagPromotionInput, RagPromotionResult, type RagRequiredContext, type RecoverPendingKnowledgeMutationOptions, RejectedSource, ResearchContribution, ResearchDriver, ResearchDrivingDriver, ResearchDrivingDriverOptions, ResearchDrivingState, ResearchDrivingSteer, ResearchSourceProposal, ResearchWorker, type ResolvedKnowledgeImprovementCandidate, type ResolvedKnowledgeImprovementComparison, type ResolvedKnowledgeImprovementComparisonSnapshot, type RestoreKnowledgeCandidateBaselineOptions, RetrievalConfig, RetrievalEvalArtifact, RetrievalEvalRetriever, RetrievalEvalRetrieverInput, RetrievalEvalRetrieverResult, RetrievalEvalScenario, RetrievalGoldTarget, RetrievalHoldoutBypassReason, RetrievalHoldoutCallContext, RetrievalHoldoutConfig, RetrievalHoldoutEligibleItem, RetrievalHoldoutEvent, RetrievalHoldoutOffPolicyOptions, RetrievalHoldoutOffPolicyResult, RetrievalHoldoutResult, RetrievalHoldoutSessionState, RetrievalHoldoutSessionSummary, RetrievalMetricSummary, RetrievalMetricWeights, RetrievalOptimizationSelection, RetrievalRecallJudgeOptions, RetrievalScenarioPartitions, RetrievedKnowledgeHit, RetrievedSourceSpan, RouterClient, RouterError, RouterUsage, type RunAgentMemoryExperimentOptions, type RunAgentMemoryExperimentResult, type RunAgentMemoryImprovementOptions, type RunAgentMemoryImprovementResult, RunDiscoveryLoopOptions, type RunKnowledgeBenchmarkSuiteOptions, type RunKnowledgeBenchmarkSuiteResult, RunKnowledgeResearchLoopOptions, type RunMemoryAdapterBenchmarkOptions, type RunMemoryAdapterBenchmarkResult, RunRagKnowledgeImprovementLoopOptions, RunRagKnowledgeImprovementLoopResult, RunRagOptimizationOptions, RunRagOptimizationResult, RunRetrievalImprovementLoopOptions, RunRetrievalImprovementLoopResult, RunSerializedKnowledgeOptimizationOptions, RunSerializedKnowledgeOptimizationResult, SCAFFOLD_PAGE_BASENAMES, SerializedCandidate, SerializedCandidateCodec, SourceAdapter, SourceAdapterInput, SourceAdapterOutput, SourceAnchor, SourceAnchorSchema, SourceFreshnessInspection, SourceRecord, SourceRecordSchema, SourceRegistry, SourceVerdict, SourceVerificationContext, StateSosEntity, StateSosSourceConfig, TangleRouterOptions, ThesisRunOptions, ThesisRunResult, ThesisTaskInput, TrackedClaim, TriageClass, type UseKnowledgeImprovementCandidateOptions, ValidateKnowledgeOptions, ValidateKnowledgeResult, VerifiedResearchLoopOptions, VerifiedResearchLoopResult, VerifiedResearchRound, VerifyingDriverOptions, WIKILINK_REGEX, WebResearchWorkerOptions, WebSearchHit, WorkerClaimDecorationOptions, WorkerResearchContext, __resetHttpThrottle, acquireAgentMemoryRunLease, addSourcePath, addSourceText, agentMemorySequenceJudge, applyKnowledgeWriteBlocks, applyKnowledgeWriteBlocksFile, applyRetrievalHoldout, applySessionStickyRetrievalHoldout, buildAgentMemorySequenceScenarios, buildAgentMemorySequencesFromBenchmarkCases, buildEvalKnowledgeBundle, buildFirstPartyMemoryLifecycleBenchmarkCases, buildIndustryMemoryBenchmarkSmokeCases, buildIndustryRagBenchmarkSmokeCases, buildKnowledgeBenchmarkScenarios, buildKnowledgeGraph, buildKnowledgeIndex, buildRetrievalBenchmarkCasesFromQrels, buildRetrievalEvalDispatch, calibrateRagAnswerJudge, canonicalizeUrl, chunkMarkdown, citedClaimKey, citedClaimOf, contentKey, createAdaptiveResearchDriver, createAgentMemoryBranch, createClaimDecorator, createClaimGroundingVerifier, createCollectionResearchDriver, createCornellLiiSource, createD1FreshnessStoreStub, createFileSystemFreshnessStore, createFileSystemSearchProvider, createGraphitiMemoryAdapter, createInMemoryBenchmarkAdapter, createIrsPublicationsSource, createKnowledgeControlLoopAdapter, createKnowledgeEvent, createLocalDiscoveryDispatcher, createMem0MemoryAdapter, createMemoryExecutionPool, createNeo4jAgentMemoryAdapter, createNoopMemoryBenchmarkAdapter, createRagAnswerQualityHook, createResearchDrivingDriver, createStateSosSource, createTangleRouterClient, createVerifyingResearchDriver, createWebResearchWorker, defaultGetMemoryContext, defineReadinessSpec, detectChanges, deterministicRng, diagnoseRagAnswerFailure, emitRetrievalHoldoutBypass, evaluateKnowledgeBaseReadiness, explainKnowledgeTarget, extractLinks, extractWikilinks, firstMatch, forkAgentMemoryBranchSnapshot, formatFrontmatter, fromAgentCandidateKnowledgeRef, gradeCompanyAgainstText, gradeFactAgainstText, graphitiMemoryAdapterIdentity, groundClaimInText, hashKnowledgeBase, htmlToText, improveKnowledgeBase, initKnowledgeBase, innerHtmlById, inspectKnowledgeIndex, inspectPendingKnowledgeMutation, investmentThesisSet, isKnowledgeMemoryBenchmarkCase, isSafeKnowledgePath, isScaffoldPath, jsonCandidateCodec, jsonObjectCandidateCodec, kbIndexToText, knowledgeBenchmarkJudge, knowledgeImprovementCandidateRef, knowledgeImprovementRunDir, knowledgeImprovementRunId, knowledgeReleaseReport, layoutFor, lensDistribution, lintKnowledgeIndex, loadKnowledgeImprovementActivationResult, loadKnowledgeImprovementEvents, loadKnowledgeImprovementState, loadKnowledgePages, loadSourceRegistry, looksLikeBlockPage, materialFactsSurfaced, materialFactsSurfacedInText, mediaTypeFor, mem0MemoryAdapterIdentity, memoryHitToSourceRecord, memoryRecoveryDelayMs, memoryWriteResultToSourceRecord, normalizeExternalRagScores, normalizeLinkTarget, optimizeKnowledgeBasePolicy, parseFrontmatter, parseKnowledgeBenchmarkJsonl, parseKnowledgeBenchmarkQrels, parseKnowledgeWriteBlocks, partitionRetrievalScenarios, politeFetch, promoteKnowledgeCandidate, proposeFromFinding, proposeFromFindings, ragAnswerQualityJudge, reciprocalRankFusion, recoverPendingKnowledgeMutation, renderKnowledgeBenchmarkReportMarkdown, renderMemoryContext, resetRetrievalHoldoutRegistry, resolveMemoryCleanupTimeoutMs, respondToIndustryMemoryBenchmarkSmokeCase, respondToIndustryRagBenchmarkSmokeCase, restoreKnowledgeCandidateBaseline, retrievalConfigFromSurface, retrievalConfigSurface, retrievalHoldoutConfigHash, retrievalRecallJudge, runAgentMemoryExperiment, runAgentMemoryImprovement, runBoundedMemoryLifecycle, runDiscoveryLoop, runInvestmentThesisTask, runKnowledgeBenchmarkSuite, runKnowledgeResearchLoop, runMemoryAdapterBenchmark, runRagKnowledgeImprovementLoop, runRagOptimization, runRetrievalImprovementLoop, runSerializedKnowledgeOptimization, runVerifiedResearchLoop, scenarioContentFingerprint, scoreKnowledgeBaseIndex, scoreKnowledgeBenchmarkArtifact, scoreMemoryBenchmarkArtifact, scoreRagAnswerArtifact, scoreRetrievalArtifact, searchKnowledge, sha256, sleepForMemoryRecovery, slugify, sourceMatchesGaps, sourceRegistryPath, stableId, stripFrontmatter, summarizeKnowledgeBenchmarkCampaign, textSourceAdapter, thesisReadinessSpecs, toAgentCandidateKnowledgeRef, toDeepEvalTestCases, toOffPolicyTrajectory, toRagCheckerRecords, toRagasEvaluationRows, toTruLensRecords, tokenizeQuery, totalMaterialFacts, triageSource, validateKnowledgeIndex, withCitedClaim, withKnowledgeImprovementCandidate, withKnowledgeImprovementComparison, writeJson, writeKnowledgeIndex, writeSourceRegistry };
|
|
2687
|
+
//# sourceMappingURL=index.d.ts.map
|