@tangle-network/agent-knowledge 6.0.0 → 6.1.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +17 -0
- package/README.md +1 -1
- package/dist/benchmarks/index.d.ts +2 -53
- package/dist/benchmarks/index.js +2 -49
- package/dist/benchmarks-CmW6iORW.js +2718 -0
- package/dist/benchmarks-CmW6iORW.js.map +1 -0
- package/dist/cli.d.ts +1 -1
- package/dist/cli.js +180 -274
- package/dist/cli.js.map +1 -1
- package/dist/ids-DRqPZ42_.js +15 -0
- package/dist/ids-DRqPZ42_.js.map +1 -0
- package/dist/index-CGBctbit.d.ts +857 -0
- package/dist/index-CGBctbit.d.ts.map +1 -0
- package/dist/index-CIW3G4s_.d.ts +680 -0
- package/dist/index-CIW3G4s_.d.ts.map +1 -0
- package/dist/index.d.ts +1671 -1868
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +5836 -6528
- package/dist/index.js.map +1 -1
- package/dist/inspect-D5iarJc2.js +1864 -0
- package/dist/inspect-D5iarJc2.js.map +1 -0
- package/dist/memory/index.d.ts +3 -8
- package/dist/memory/index.js +3 -81
- package/dist/memory-C6KPRhoU.js +4494 -0
- package/dist/memory-C6KPRhoU.js.map +1 -0
- package/dist/search-CP0QtBJZ.js +113 -0
- package/dist/search-CP0QtBJZ.js.map +1 -0
- package/dist/sources/index.d.ts +212 -205
- package/dist/sources/index.d.ts.map +1 -0
- package/dist/sources/index.js +614 -33
- package/dist/sources/index.js.map +1 -1
- package/dist/types-DcCCzreS.d.ts +175 -0
- package/dist/types-DcCCzreS.d.ts.map +1 -0
- package/dist/viz/index.d.ts +23 -22
- package/dist/viz/index.d.ts.map +1 -0
- package/dist/viz/index.js +134 -10
- package/dist/viz/index.js.map +1 -1
- package/package.json +22 -11
- package/dist/benchmarks/index.js.map +0 -1
- package/dist/chunk-46YPZHAX.js +0 -5443
- package/dist/chunk-46YPZHAX.js.map +0 -1
- package/dist/chunk-4PNXQ2NT.js +0 -147
- package/dist/chunk-4PNXQ2NT.js.map +0 -1
- package/dist/chunk-AKYJG2MR.js +0 -2183
- package/dist/chunk-AKYJG2MR.js.map +0 -1
- package/dist/chunk-DQ3PDMDP.js +0 -115
- package/dist/chunk-DQ3PDMDP.js.map +0 -1
- package/dist/chunk-MYFM6LKH.js +0 -551
- package/dist/chunk-MYFM6LKH.js.map +0 -1
- package/dist/chunk-PVCSESAF.js +0 -3153
- package/dist/chunk-PVCSESAF.js.map +0 -1
- package/dist/chunk-YMKHCTS2.js +0 -19
- package/dist/chunk-YMKHCTS2.js.map +0 -1
- package/dist/index-C--N5wQV.d.ts +0 -796
- package/dist/memory/index.js.map +0 -1
- package/dist/types-6x0OpfW6.d.ts +0 -173
- package/dist/types-BY-xLVw-.d.ts +0 -622
package/dist/types-BY-xLVw-.d.ts
DELETED
|
@@ -1,622 +0,0 @@
|
|
|
1
|
-
import { Scenario, DispatchContext, MutableSurface, JudgeConfig, CampaignStorage, CostLedgerHandle, CampaignResult } from '@tangle-network/agent-eval/campaign';
|
|
2
|
-
import { c as KnowledgeIndex, S as SourceRecord } from './types-6x0OpfW6.js';
|
|
3
|
-
import { AgentCandidateJsonValue } from '@tangle-network/agent-interface';
|
|
4
|
-
|
|
5
|
-
type RetrievalConfig = Record<string, AgentCandidateJsonValue>;
|
|
6
|
-
type RetrievalGoldTarget = {
|
|
7
|
-
kind: 'page';
|
|
8
|
-
pageId: string;
|
|
9
|
-
} | {
|
|
10
|
-
kind: 'page-path';
|
|
11
|
-
path: string;
|
|
12
|
-
} | {
|
|
13
|
-
kind: 'source';
|
|
14
|
-
sourceId: string;
|
|
15
|
-
} | {
|
|
16
|
-
kind: 'source-anchor';
|
|
17
|
-
sourceId: string;
|
|
18
|
-
anchorId: string;
|
|
19
|
-
} | {
|
|
20
|
-
kind: 'source-span';
|
|
21
|
-
sourceId: string;
|
|
22
|
-
charStart: number;
|
|
23
|
-
charEnd: number;
|
|
24
|
-
};
|
|
25
|
-
interface RetrievalEvalScenario extends Scenario {
|
|
26
|
-
kind: 'retrieval-eval';
|
|
27
|
-
query: string;
|
|
28
|
-
expected: RetrievalGoldTarget | readonly RetrievalGoldTarget[];
|
|
29
|
-
k?: number;
|
|
30
|
-
}
|
|
31
|
-
interface RetrievedSourceSpan {
|
|
32
|
-
sourceId: string;
|
|
33
|
-
anchorId?: string;
|
|
34
|
-
charStart?: number;
|
|
35
|
-
charEnd?: number;
|
|
36
|
-
}
|
|
37
|
-
interface RetrievedKnowledgeHit {
|
|
38
|
-
pageId: string;
|
|
39
|
-
path: string;
|
|
40
|
-
title?: string;
|
|
41
|
-
rank: number;
|
|
42
|
-
score?: number;
|
|
43
|
-
normalizedScore?: number;
|
|
44
|
-
sourceIds?: readonly string[];
|
|
45
|
-
sourceSpans?: readonly RetrievedSourceSpan[];
|
|
46
|
-
snippet?: string;
|
|
47
|
-
metadata?: Record<string, AgentCandidateJsonValue>;
|
|
48
|
-
}
|
|
49
|
-
interface RetrievalEvalArtifact {
|
|
50
|
-
config: RetrievalConfig;
|
|
51
|
-
query: string;
|
|
52
|
-
requestedK: number;
|
|
53
|
-
hits: readonly RetrievedKnowledgeHit[];
|
|
54
|
-
durationMs: number;
|
|
55
|
-
/** Informational copy. Billable retrievers account through context.cost.runPaidCall. */
|
|
56
|
-
costUsd?: number;
|
|
57
|
-
metadata?: Record<string, AgentCandidateJsonValue>;
|
|
58
|
-
}
|
|
59
|
-
interface RetrievalMetricSummary {
|
|
60
|
-
recall: number;
|
|
61
|
-
mrr: number;
|
|
62
|
-
ndcg: number;
|
|
63
|
-
precisionAtK: number;
|
|
64
|
-
expectedCount: number;
|
|
65
|
-
matchedCount: number;
|
|
66
|
-
relevantHitCount: number;
|
|
67
|
-
firstHitRank: number | null;
|
|
68
|
-
matchedTargetIds: readonly string[];
|
|
69
|
-
}
|
|
70
|
-
interface RetrievalEvalRetrieverInput {
|
|
71
|
-
index?: KnowledgeIndex;
|
|
72
|
-
config: RetrievalConfig;
|
|
73
|
-
scenario: RetrievalEvalScenario;
|
|
74
|
-
k: number;
|
|
75
|
-
signal: AbortSignal;
|
|
76
|
-
context: DispatchContext;
|
|
77
|
-
}
|
|
78
|
-
interface RetrievalEvalRetrieverResult {
|
|
79
|
-
hits: readonly RetrievedKnowledgeHit[];
|
|
80
|
-
/** Informational copy. Billable retrievers account through context.cost.runPaidCall. */
|
|
81
|
-
costUsd?: number;
|
|
82
|
-
metadata?: Record<string, AgentCandidateJsonValue>;
|
|
83
|
-
}
|
|
84
|
-
type RetrievalEvalRetriever = (input: RetrievalEvalRetrieverInput) => Promise<readonly RetrievedKnowledgeHit[] | RetrievalEvalRetrieverResult>;
|
|
85
|
-
interface BuildRetrievalEvalDispatchOptions {
|
|
86
|
-
index?: KnowledgeIndex;
|
|
87
|
-
defaultK?: number;
|
|
88
|
-
retrieve?: RetrievalEvalRetriever;
|
|
89
|
-
}
|
|
90
|
-
interface RetrievalMetricWeights {
|
|
91
|
-
recall?: number;
|
|
92
|
-
mrr?: number;
|
|
93
|
-
ndcg?: number;
|
|
94
|
-
precisionAtK?: number;
|
|
95
|
-
}
|
|
96
|
-
interface RetrievalRecallJudgeOptions {
|
|
97
|
-
name?: string;
|
|
98
|
-
weights?: RetrievalMetricWeights;
|
|
99
|
-
}
|
|
100
|
-
interface PartitionRetrievalScenariosOptions {
|
|
101
|
-
selectionFraction?: number;
|
|
102
|
-
finalFraction?: number;
|
|
103
|
-
seed?: number;
|
|
104
|
-
}
|
|
105
|
-
interface RetrievalScenarioPartitions {
|
|
106
|
-
trainScenarios: RetrievalEvalScenario[];
|
|
107
|
-
selectionScenarios: RetrievalEvalScenario[];
|
|
108
|
-
finalScenarios: RetrievalEvalScenario[];
|
|
109
|
-
}
|
|
110
|
-
declare function retrievalConfigSurface(config: RetrievalConfig): string;
|
|
111
|
-
declare function retrievalConfigFromSurface(surface: MutableSurface): RetrievalConfig;
|
|
112
|
-
declare function buildRetrievalEvalDispatch(options: BuildRetrievalEvalDispatchOptions): (surface: MutableSurface, scenario: RetrievalEvalScenario, context: DispatchContext) => Promise<RetrievalEvalArtifact>;
|
|
113
|
-
declare function retrievalRecallJudge(options?: RetrievalRecallJudgeOptions): JudgeConfig<RetrievalEvalArtifact, RetrievalEvalScenario>;
|
|
114
|
-
declare function scoreRetrievalArtifact(artifact: RetrievalEvalArtifact, scenario: RetrievalEvalScenario): RetrievalMetricSummary;
|
|
115
|
-
declare function partitionRetrievalScenarios(scenarios: readonly RetrievalEvalScenario[], options?: PartitionRetrievalScenariosOptions): RetrievalScenarioPartitions;
|
|
116
|
-
|
|
117
|
-
type AgentMemoryKind = 'message' | 'entity' | 'fact' | 'preference' | 'observation' | 'reasoning-trace';
|
|
118
|
-
interface AgentMemoryScope {
|
|
119
|
-
tenantId?: string;
|
|
120
|
-
userId?: string;
|
|
121
|
-
agentId?: string;
|
|
122
|
-
teamId?: string;
|
|
123
|
-
runId?: string;
|
|
124
|
-
sessionId?: string;
|
|
125
|
-
namespace?: string;
|
|
126
|
-
tags?: Record<string, string>;
|
|
127
|
-
}
|
|
128
|
-
interface AgentMemoryHit {
|
|
129
|
-
id: string;
|
|
130
|
-
uri: string;
|
|
131
|
-
kind: AgentMemoryKind;
|
|
132
|
-
text: string;
|
|
133
|
-
title?: string;
|
|
134
|
-
score?: number;
|
|
135
|
-
normalizedScore?: number;
|
|
136
|
-
confidence?: number;
|
|
137
|
-
createdAt?: string;
|
|
138
|
-
validUntil?: string;
|
|
139
|
-
lastVerifiedAt?: string;
|
|
140
|
-
metadata?: Record<string, unknown>;
|
|
141
|
-
}
|
|
142
|
-
interface AgentMemoryContext {
|
|
143
|
-
query: string;
|
|
144
|
-
text: string;
|
|
145
|
-
hits: AgentMemoryHit[];
|
|
146
|
-
sourceRecords: SourceRecord[];
|
|
147
|
-
metadata?: Record<string, unknown>;
|
|
148
|
-
}
|
|
149
|
-
interface AgentMemorySearchOptions {
|
|
150
|
-
scope?: AgentMemoryScope;
|
|
151
|
-
limit?: number;
|
|
152
|
-
minScore?: number;
|
|
153
|
-
kinds?: AgentMemoryKind[];
|
|
154
|
-
metadata?: Record<string, unknown>;
|
|
155
|
-
/**
|
|
156
|
-
* Opt-in randomized retrieval holdout (epsilon-dropout) for treatment-effect logging.
|
|
157
|
-
* Absent by default; when absent, retrieval behavior is unchanged. See ./holdout.
|
|
158
|
-
*/
|
|
159
|
-
holdout?: RetrievalHoldoutConfig;
|
|
160
|
-
}
|
|
161
|
-
interface AgentMemoryWriteInput {
|
|
162
|
-
kind: AgentMemoryKind;
|
|
163
|
-
text: string;
|
|
164
|
-
id?: string;
|
|
165
|
-
title?: string;
|
|
166
|
-
role?: 'system' | 'user' | 'assistant' | 'tool';
|
|
167
|
-
entityName?: string;
|
|
168
|
-
entityType?: string;
|
|
169
|
-
category?: string;
|
|
170
|
-
predicate?: string;
|
|
171
|
-
subject?: string;
|
|
172
|
-
object?: string;
|
|
173
|
-
confidence?: number;
|
|
174
|
-
scope?: AgentMemoryScope;
|
|
175
|
-
metadata?: Record<string, unknown>;
|
|
176
|
-
}
|
|
177
|
-
interface AgentMemoryWriteResult {
|
|
178
|
-
accepted: boolean;
|
|
179
|
-
id: string;
|
|
180
|
-
uri: string;
|
|
181
|
-
kind: AgentMemoryKind;
|
|
182
|
-
sourceRecord?: SourceRecord;
|
|
183
|
-
metadata?: Record<string, unknown>;
|
|
184
|
-
}
|
|
185
|
-
type AgentMemoryBranchIsolation = {
|
|
186
|
-
mode: 'scoped';
|
|
187
|
-
/** False when writes may outlive the worker process that issued them. */
|
|
188
|
-
processExitSafe?: boolean;
|
|
189
|
-
/** Wait before clearing an abandoned branch so accepted asynchronous writes become visible. */
|
|
190
|
-
recoveryDelayMs?: number;
|
|
191
|
-
} | {
|
|
192
|
-
mode: 'instance';
|
|
193
|
-
branchId: string;
|
|
194
|
-
/** True only when the dedicated instance also enforces every logical scope. */
|
|
195
|
-
supportsLogicalScopes?: boolean;
|
|
196
|
-
} | {
|
|
197
|
-
mode: 'unsupported';
|
|
198
|
-
reason: string;
|
|
199
|
-
};
|
|
200
|
-
interface AgentMemoryAdapter {
|
|
201
|
-
readonly id: string;
|
|
202
|
-
/** How this adapter prevents candidate branches from reading each other's state. */
|
|
203
|
-
readonly branchIsolation?: AgentMemoryBranchIsolation;
|
|
204
|
-
search(query: string, options?: AgentMemorySearchOptions): Promise<AgentMemoryHit[]>;
|
|
205
|
-
getContext(query: string, options?: AgentMemorySearchOptions): Promise<AgentMemoryContext>;
|
|
206
|
-
write(input: AgentMemoryWriteInput): Promise<AgentMemoryWriteResult>;
|
|
207
|
-
/** Delete exactly this scope. Repeated and concurrent calls for the same scope must be safe. */
|
|
208
|
-
clear?(scope?: AgentMemoryScope): Promise<void>;
|
|
209
|
-
flush?(): Promise<void>;
|
|
210
|
-
close?(): Promise<void>;
|
|
211
|
-
}
|
|
212
|
-
/**
|
|
213
|
-
* Optional session-level retrieval dropout for estimating whether delivered memories affect
|
|
214
|
-
* task outcomes. The feature is disabled unless configured, and consumers persist events
|
|
215
|
-
* through `onEvent`.
|
|
216
|
-
*/
|
|
217
|
-
interface RetrievalHoldoutConfig {
|
|
218
|
-
/** Per-session probability that one eligible watchlist item is suppressed. 0 logs the full schema without ever dropping. */
|
|
219
|
-
epsilon: number;
|
|
220
|
-
/** Item ids eligible for suppression. Empty or absent means no item can ever be dropped. */
|
|
221
|
-
watchlist?: string[];
|
|
222
|
-
/** Ties every event to the exact epsilon/watchlist in force, for audit and replay. */
|
|
223
|
-
configVersion?: string;
|
|
224
|
-
/** Copied onto every event so multi-adapter logs stay attributable. */
|
|
225
|
-
adapterId?: string;
|
|
226
|
-
/** Corpus/store version stamp; an edited item under the same id is a different treatment. */
|
|
227
|
-
corpusVersion?: string;
|
|
228
|
-
/**
|
|
229
|
-
* Emit plaintext sessionId and scope on events. Default false: events carry only
|
|
230
|
-
* sessionIdHash/scopeHash, so PII-bearing identifiers (tenantId/userId/tags) never reach a
|
|
231
|
-
* consumer-controlled sink unless the consumer explicitly owns that decision. Note that
|
|
232
|
-
* replaying assignment draws from logs alone needs the plaintext sessionId, so
|
|
233
|
-
* privacy-preserving logs require the consumer's own sessionId mapping for replay.
|
|
234
|
-
*/
|
|
235
|
-
includePlaintextIdentifiers?: boolean;
|
|
236
|
-
/**
|
|
237
|
-
* Cap on tracked sessions per experiment config in the sticky wrapper's registry.
|
|
238
|
-
* The default is 10,000.
|
|
239
|
-
*/
|
|
240
|
-
maxTrackedSessions?: number;
|
|
241
|
-
/**
|
|
242
|
-
* Uniform-[0,1) generator keyed by a string. Defaults to a sha256-derived deterministic
|
|
243
|
-
* generator so every assignment is replayable from the logged keys alone.
|
|
244
|
-
*/
|
|
245
|
-
rng?: (key: string) => number;
|
|
246
|
-
/** Receives one event per retrieval call, including calls where nothing is dropped. */
|
|
247
|
-
onEvent: (event: RetrievalHoldoutEvent) => void;
|
|
248
|
-
}
|
|
249
|
-
interface RetrievalHoldoutEligibleItem {
|
|
250
|
-
id: string;
|
|
251
|
-
/** 1-based position in the post-filter hit list. */
|
|
252
|
-
rank: number;
|
|
253
|
-
score?: number;
|
|
254
|
-
kind: string;
|
|
255
|
-
/** sha256(hit.text) prefix; effects are estimated per (id, contentHash) pair. */
|
|
256
|
-
contentHash: string;
|
|
257
|
-
}
|
|
258
|
-
interface RetrievalHoldoutEvent {
|
|
259
|
-
v: 1;
|
|
260
|
-
eventId: string;
|
|
261
|
-
ts: string;
|
|
262
|
-
adapterId?: string;
|
|
263
|
-
/** Plaintext session id, emitted only when `includePlaintextIdentifiers` is true. */
|
|
264
|
-
sessionId?: string;
|
|
265
|
-
/** Consumer-supplied experiment/outcome join id (scope.tags.taskId); deliberately plaintext. */
|
|
266
|
-
taskId?: string;
|
|
267
|
-
/** 1-based call counter within the session; 0 when the call is outside session randomization. */
|
|
268
|
-
callIndex: number;
|
|
269
|
-
/**
|
|
270
|
-
* sha256(sessionId) prefix used as the privacy-preserving join key and assignment seed.
|
|
271
|
-
*/
|
|
272
|
-
sessionIdHash?: string;
|
|
273
|
-
queryHash?: string;
|
|
274
|
-
/** Verbatim scope, emitted only when `includePlaintextIdentifiers` is true. */
|
|
275
|
-
scope?: AgentMemoryScope;
|
|
276
|
-
/** sha256 prefix of the canonical-JSON scope (keys sorted, undefined stripped). */
|
|
277
|
-
scopeHash?: string;
|
|
278
|
-
config: {
|
|
279
|
-
epsilon: number;
|
|
280
|
-
watchlist: string[];
|
|
281
|
-
configVersion?: string;
|
|
282
|
-
};
|
|
283
|
-
/**
|
|
284
|
-
* Value-hash of the experiment-defining knobs, sha256({epsilon, sorted watchlist}) prefix.
|
|
285
|
-
* The estimator groups events by it; the sticky-session registry is keyed by it.
|
|
286
|
-
*/
|
|
287
|
-
configHash: string;
|
|
288
|
-
/**
|
|
289
|
-
* False when no sessionId is available or the adapter answered without retrieval
|
|
290
|
-
* (see bypassReason), so the fraction-under-experiment denominator stays honest.
|
|
291
|
-
*/
|
|
292
|
-
holdoutEligible: boolean;
|
|
293
|
-
/** Present only on adapter paths that bypassed retrieval, where no suppression could apply. */
|
|
294
|
-
bypassReason?: RetrievalHoldoutBypassReason;
|
|
295
|
-
/** The full post-filter eligibility set E, logged on every call (control arm + interference probes). */
|
|
296
|
-
eligible: RetrievalHoldoutEligibleItem[];
|
|
297
|
-
/** Ids in watchlist ∩ E, in eligibility order. */
|
|
298
|
-
watchlistEligible: string[];
|
|
299
|
-
sessionHoldout: boolean;
|
|
300
|
-
/** The session's sticky drop target once drawn; distinguishes "target absent from E" from "not yet drawn". */
|
|
301
|
-
sessionTargetId: string | null;
|
|
302
|
-
/** The item suppressed in THIS call, or null. */
|
|
303
|
-
droppedId: string | null;
|
|
304
|
-
/** 1/|watchlist ∩ E| recorded at draw time; the exact inverse-propensity weight input. */
|
|
305
|
-
pickPropensity: number | null;
|
|
306
|
-
/** epsilon * pickPropensity, recorded at draw time so analysis never re-derives assignment probabilities. */
|
|
307
|
-
dropPropensity: number | null;
|
|
308
|
-
deliveredIds: string[];
|
|
309
|
-
corpusVersion?: string;
|
|
310
|
-
}
|
|
311
|
-
interface RetrievalHoldoutSessionState {
|
|
312
|
-
sessionId: string;
|
|
313
|
-
/** Calls observed so far in this session. */
|
|
314
|
-
callCount: number;
|
|
315
|
-
sessionHoldout: boolean;
|
|
316
|
-
/** Sticky drop target; drawn once at the first call whose eligibility set intersects the watchlist. */
|
|
317
|
-
targetId: string | null;
|
|
318
|
-
pickPropensity: number | null;
|
|
319
|
-
}
|
|
320
|
-
interface RetrievalHoldoutCallContext {
|
|
321
|
-
sessionId?: string;
|
|
322
|
-
taskId?: string;
|
|
323
|
-
/** Raw query; only its sha256 prefix is logged. */
|
|
324
|
-
query?: string;
|
|
325
|
-
scope?: AgentMemoryScope;
|
|
326
|
-
/** State returned by the previous call of this session; threading it is what makes suppression sticky. */
|
|
327
|
-
session?: RetrievalHoldoutSessionState;
|
|
328
|
-
}
|
|
329
|
-
interface RetrievalHoldoutResult {
|
|
330
|
-
delivered: AgentMemoryHit[];
|
|
331
|
-
event: RetrievalHoldoutEvent;
|
|
332
|
-
session?: RetrievalHoldoutSessionState;
|
|
333
|
-
}
|
|
334
|
-
/** Adapter context paths that answer without retrieval, so no holdout draw can happen. */
|
|
335
|
-
type RetrievalHoldoutBypassReason = 'short-term-context' | 'raw-string-context';
|
|
336
|
-
|
|
337
|
-
interface AgentMemoryRunLease {
|
|
338
|
-
assertOwned(): Promise<void> | void;
|
|
339
|
-
release(): Promise<void> | void;
|
|
340
|
-
}
|
|
341
|
-
type AgentMemoryAcquireRunLease = (input: {
|
|
342
|
-
experimentId: string;
|
|
343
|
-
runDir: string;
|
|
344
|
-
}) => AgentMemoryRunLease | Promise<AgentMemoryRunLease>;
|
|
345
|
-
type AgentMemoryControllerMode = 'process-local';
|
|
346
|
-
interface OwnedAgentMemoryRunLease {
|
|
347
|
-
assertOwned(): Promise<void>;
|
|
348
|
-
release(): Promise<void>;
|
|
349
|
-
}
|
|
350
|
-
declare function acquireAgentMemoryRunLease(input: {
|
|
351
|
-
experimentId: string;
|
|
352
|
-
runDir: string;
|
|
353
|
-
storage: CampaignStorage;
|
|
354
|
-
customStorage: boolean;
|
|
355
|
-
lockFileName: string;
|
|
356
|
-
label: string;
|
|
357
|
-
controllerMode?: AgentMemoryControllerMode;
|
|
358
|
-
acquireRunLease?: AgentMemoryAcquireRunLease;
|
|
359
|
-
}): Promise<OwnedAgentMemoryRunLease>;
|
|
360
|
-
|
|
361
|
-
type KnowledgeBenchmarkTaskKind = 'retrieval' | 'rag-answer' | 'hallucination' | 'kb-improvement' | 'memory-ingest' | 'memory-recall' | 'memory-temporal' | 'memory-update' | 'memory-forgetting' | 'memory-reasoning' | 'memory-summarization' | 'memory-recommendation' | 'memory-multiparty';
|
|
362
|
-
type KnowledgeAnswerBenchmarkTaskKind = 'rag-answer' | 'hallucination' | 'kb-improvement';
|
|
363
|
-
type KnowledgeMemoryBenchmarkTaskKind = Exclude<KnowledgeBenchmarkTaskKind, 'retrieval' | KnowledgeAnswerBenchmarkTaskKind>;
|
|
364
|
-
type KnowledgeBenchmarkFamily = 'beir' | 'mteb-retrieval' | 'msmarco' | 'trec-dl' | 'miracl' | 'lotte' | 'bright' | 'crag' | 'hotpotqa' | 'kilt' | 'ragtruth' | 'faithbench' | 'locomo' | 'longmemeval' | 'longmemeval-v2' | 'memora' | 'memoryagentbench' | 'memorybank' | 'groupmembench' | 'first-party' | 'custom';
|
|
365
|
-
type KnowledgeBenchmarkSplit = 'search' | 'dev' | 'holdout' | string;
|
|
366
|
-
interface KnowledgeBenchmarkSource {
|
|
367
|
-
name?: string;
|
|
368
|
-
url?: string;
|
|
369
|
-
version?: string;
|
|
370
|
-
license?: string;
|
|
371
|
-
citation?: string;
|
|
372
|
-
}
|
|
373
|
-
interface KnowledgeBenchmarkSpec {
|
|
374
|
-
id: string;
|
|
375
|
-
family: KnowledgeBenchmarkFamily;
|
|
376
|
-
taskKind: KnowledgeBenchmarkTaskKind;
|
|
377
|
-
primaryMetrics: readonly string[];
|
|
378
|
-
adapter: string;
|
|
379
|
-
notes: string;
|
|
380
|
-
}
|
|
381
|
-
interface KnowledgeBenchmarkCaseBase {
|
|
382
|
-
id: string;
|
|
383
|
-
family: KnowledgeBenchmarkFamily | string;
|
|
384
|
-
taskKind: KnowledgeBenchmarkTaskKind;
|
|
385
|
-
split?: KnowledgeBenchmarkSplit;
|
|
386
|
-
tags?: readonly string[];
|
|
387
|
-
source?: KnowledgeBenchmarkSource;
|
|
388
|
-
metadata?: Record<string, unknown>;
|
|
389
|
-
}
|
|
390
|
-
interface KnowledgeRetrievalBenchmarkCase extends KnowledgeBenchmarkCaseBase {
|
|
391
|
-
taskKind: 'retrieval';
|
|
392
|
-
query: string;
|
|
393
|
-
expected: RetrievalGoldTarget | readonly RetrievalGoldTarget[];
|
|
394
|
-
k?: number;
|
|
395
|
-
}
|
|
396
|
-
interface KnowledgeClaimMatcher {
|
|
397
|
-
id: string;
|
|
398
|
-
anyOf: readonly string[];
|
|
399
|
-
weight?: number;
|
|
400
|
-
}
|
|
401
|
-
interface KnowledgeMemoryEvent {
|
|
402
|
-
id: string;
|
|
403
|
-
text: string;
|
|
404
|
-
actorId?: string;
|
|
405
|
-
sessionId?: string;
|
|
406
|
-
timestamp?: string;
|
|
407
|
-
metadata?: Record<string, unknown>;
|
|
408
|
-
}
|
|
409
|
-
interface KnowledgeMemoryFactMatcher extends KnowledgeClaimMatcher {
|
|
410
|
-
sourceEventIds?: readonly string[];
|
|
411
|
-
validAt?: string;
|
|
412
|
-
obsolete?: boolean;
|
|
413
|
-
}
|
|
414
|
-
interface KnowledgeAnswerBenchmarkCase extends KnowledgeBenchmarkCaseBase {
|
|
415
|
-
taskKind: KnowledgeAnswerBenchmarkTaskKind;
|
|
416
|
-
prompt: string;
|
|
417
|
-
requiredClaims?: readonly KnowledgeClaimMatcher[];
|
|
418
|
-
forbiddenClaims?: readonly KnowledgeClaimMatcher[];
|
|
419
|
-
expectedSourceIds?: readonly string[];
|
|
420
|
-
referenceAnswer?: string;
|
|
421
|
-
}
|
|
422
|
-
interface KnowledgeMemoryBenchmarkCase extends KnowledgeBenchmarkCaseBase {
|
|
423
|
-
taskKind: KnowledgeMemoryBenchmarkTaskKind;
|
|
424
|
-
events: readonly KnowledgeMemoryEvent[];
|
|
425
|
-
prompt: string;
|
|
426
|
-
requiredFacts?: readonly KnowledgeMemoryFactMatcher[];
|
|
427
|
-
forbiddenFacts?: readonly KnowledgeMemoryFactMatcher[];
|
|
428
|
-
expectedEventIds?: readonly string[];
|
|
429
|
-
expectedActorIds?: readonly string[];
|
|
430
|
-
referenceAnswer?: string;
|
|
431
|
-
}
|
|
432
|
-
type KnowledgeBenchmarkCase = KnowledgeRetrievalBenchmarkCase | KnowledgeAnswerBenchmarkCase | KnowledgeMemoryBenchmarkCase;
|
|
433
|
-
interface KnowledgeBenchmarkArtifact {
|
|
434
|
-
answer?: string;
|
|
435
|
-
text?: string;
|
|
436
|
-
hits?: readonly RetrievedKnowledgeHit[];
|
|
437
|
-
citedSourceIds?: readonly string[];
|
|
438
|
-
rememberedFacts?: readonly string[];
|
|
439
|
-
citedEventIds?: readonly string[];
|
|
440
|
-
usedMemoryIds?: readonly string[];
|
|
441
|
-
actorIds?: readonly string[];
|
|
442
|
-
/** Informational copy. Billable responders account through context.cost.runPaidCall. */
|
|
443
|
-
costUsd?: number;
|
|
444
|
-
durationMs?: number;
|
|
445
|
-
metadata?: Record<string, unknown>;
|
|
446
|
-
}
|
|
447
|
-
interface KnowledgeBenchmarkEvaluation {
|
|
448
|
-
score: number;
|
|
449
|
-
passed: boolean;
|
|
450
|
-
dimensions: Record<string, number>;
|
|
451
|
-
/** Dimensions for which this case declared an actual target. */
|
|
452
|
-
applicableDimensions?: readonly string[];
|
|
453
|
-
notes: string;
|
|
454
|
-
raw: Record<string, unknown>;
|
|
455
|
-
}
|
|
456
|
-
interface KnowledgeBenchmarkScenario extends Scenario {
|
|
457
|
-
kind: 'knowledge-benchmark';
|
|
458
|
-
family: KnowledgeBenchmarkFamily | string;
|
|
459
|
-
taskKind: KnowledgeBenchmarkTaskKind;
|
|
460
|
-
splitTag: KnowledgeBenchmarkSplit;
|
|
461
|
-
case: KnowledgeBenchmarkCase;
|
|
462
|
-
}
|
|
463
|
-
type KnowledgeBenchmarkResponder<TArtifact = KnowledgeBenchmarkArtifact> = (input: {
|
|
464
|
-
case: KnowledgeBenchmarkCase;
|
|
465
|
-
scenario: KnowledgeBenchmarkScenario;
|
|
466
|
-
context: DispatchContext;
|
|
467
|
-
}) => Promise<TArtifact> | TArtifact;
|
|
468
|
-
interface RunKnowledgeBenchmarkSuiteOptions<TArtifact = KnowledgeBenchmarkArtifact> {
|
|
469
|
-
cases: readonly KnowledgeBenchmarkCase[];
|
|
470
|
-
respond: KnowledgeBenchmarkResponder<TArtifact>;
|
|
471
|
-
/** Versioned identity for the model, prompt, retrieval, and runtime behavior. */
|
|
472
|
-
respondRef?: string;
|
|
473
|
-
runDir: string;
|
|
474
|
-
splits?: readonly KnowledgeBenchmarkSplit[];
|
|
475
|
-
repo?: string;
|
|
476
|
-
seed?: number;
|
|
477
|
-
reps?: number;
|
|
478
|
-
resumable?: boolean;
|
|
479
|
-
costCeiling?: number;
|
|
480
|
-
/** Shared across nested benchmark suites when an outer run owns spend. */
|
|
481
|
-
costLedger?: CostLedgerHandle;
|
|
482
|
-
costPhase?: string;
|
|
483
|
-
maxConcurrency?: number;
|
|
484
|
-
dispatchTimeoutMs?: number;
|
|
485
|
-
expectUsage?: 'assert' | 'warn' | 'off';
|
|
486
|
-
storage?: CampaignStorage;
|
|
487
|
-
now?: () => Date;
|
|
488
|
-
}
|
|
489
|
-
interface KnowledgeBenchmarkDistribution {
|
|
490
|
-
n: number;
|
|
491
|
-
min: number;
|
|
492
|
-
mean: number;
|
|
493
|
-
median: number;
|
|
494
|
-
p90: number;
|
|
495
|
-
max: number;
|
|
496
|
-
}
|
|
497
|
-
interface KnowledgeBenchmarkSliceSummary {
|
|
498
|
-
n: number;
|
|
499
|
-
meanScore: number;
|
|
500
|
-
passRate: number;
|
|
501
|
-
score: KnowledgeBenchmarkDistribution;
|
|
502
|
-
}
|
|
503
|
-
interface KnowledgeBenchmarkReport {
|
|
504
|
-
totalCases: number;
|
|
505
|
-
totalCells: number;
|
|
506
|
-
cellsFailed: number;
|
|
507
|
-
cellsCached: number;
|
|
508
|
-
totalCostUsd: number;
|
|
509
|
-
bySplit: Record<string, KnowledgeBenchmarkSliceSummary>;
|
|
510
|
-
byFamily: Record<string, KnowledgeBenchmarkSliceSummary>;
|
|
511
|
-
byTaskKind: Record<string, KnowledgeBenchmarkSliceSummary>;
|
|
512
|
-
dimensions: Record<string, KnowledgeBenchmarkDistribution>;
|
|
513
|
-
score: KnowledgeBenchmarkDistribution;
|
|
514
|
-
}
|
|
515
|
-
interface RunKnowledgeBenchmarkSuiteResult<TArtifact = KnowledgeBenchmarkArtifact> {
|
|
516
|
-
scenarios: readonly KnowledgeBenchmarkScenario[];
|
|
517
|
-
campaign: CampaignResult<TArtifact, KnowledgeBenchmarkScenario>;
|
|
518
|
-
report: KnowledgeBenchmarkReport;
|
|
519
|
-
reportJsonPath: string;
|
|
520
|
-
reportMarkdownPath: string;
|
|
521
|
-
}
|
|
522
|
-
interface MemoryAdapterBenchmarkCandidate {
|
|
523
|
-
id: string;
|
|
524
|
-
/** Versioned adapter and configuration identity used by resumable caches. */
|
|
525
|
-
ref: string;
|
|
526
|
-
/** Expected adapter.id. Defaults to candidate id and permits lazy no-work resume. */
|
|
527
|
-
adapterId?: string;
|
|
528
|
-
label?: string;
|
|
529
|
-
/** Local construction is free; call markExternalCall before billable provisioning or reconnects. */
|
|
530
|
-
createAdapter: (input: {
|
|
531
|
-
purpose: 'execute' | 'recovery';
|
|
532
|
-
signal: AbortSignal;
|
|
533
|
-
markExternalCall(): void;
|
|
534
|
-
}) => AgentMemoryAdapter | Promise<AgentMemoryAdapter>;
|
|
535
|
-
/** Conservative charge for one billable adapter provisioning or reconnect call. */
|
|
536
|
-
adapterCreationCostUsd?: number;
|
|
537
|
-
searchLimit?: number;
|
|
538
|
-
costUsdPerCase?: number;
|
|
539
|
-
/** Conservative extra provider charge for recovering one interrupted case. */
|
|
540
|
-
recoveryCostUsdPerAttempt?: number;
|
|
541
|
-
scope?: AgentMemoryScope;
|
|
542
|
-
}
|
|
543
|
-
interface RunMemoryAdapterBenchmarkOptions {
|
|
544
|
-
cases: readonly KnowledgeMemoryBenchmarkCase[];
|
|
545
|
-
candidates: readonly MemoryAdapterBenchmarkCandidate[];
|
|
546
|
-
/** Retired candidates retained only so interrupted scopes can be cleaned on resume. */
|
|
547
|
-
recoveryCandidates?: readonly MemoryAdapterBenchmarkCandidate[];
|
|
548
|
-
runDir: string;
|
|
549
|
-
storage?: CampaignStorage;
|
|
550
|
-
repo?: string;
|
|
551
|
-
seed?: number;
|
|
552
|
-
reps?: number;
|
|
553
|
-
resumable?: boolean;
|
|
554
|
-
costCeiling?: number;
|
|
555
|
-
/** Shared with nested benchmark suites so the dollar limit applies to the whole comparison. */
|
|
556
|
-
costLedger?: CostLedgerHandle;
|
|
557
|
-
costPhase?: string;
|
|
558
|
-
maxConcurrency?: number;
|
|
559
|
-
dispatchTimeoutMs?: number;
|
|
560
|
-
cleanupTimeoutMs?: number;
|
|
561
|
-
/** Refuse a damaged run with more unfinished attempts than this. Default 1000. */
|
|
562
|
-
maxRecoveryAttempts?: number;
|
|
563
|
-
/** Bound repeated provider cleanup after process crashes. Default 3 per attempt. */
|
|
564
|
-
maxRecoveryRetriesPerAttempt?: number;
|
|
565
|
-
expectUsage?: 'assert' | 'warn' | 'off';
|
|
566
|
-
now?: () => Date;
|
|
567
|
-
/** Required with custom storage when all controllers are confined to one process. */
|
|
568
|
-
controllerMode?: AgentMemoryControllerMode;
|
|
569
|
-
/** Required for distributed controllers that share custom storage. */
|
|
570
|
-
acquireRunLease?: AgentMemoryAcquireRunLease;
|
|
571
|
-
}
|
|
572
|
-
interface MemoryAdapterBenchmarkRankingRow {
|
|
573
|
-
rank: number;
|
|
574
|
-
candidateId: string;
|
|
575
|
-
label: string;
|
|
576
|
-
adapterId: string;
|
|
577
|
-
scoreMean: number;
|
|
578
|
-
passRate: number;
|
|
579
|
-
totalCases: number;
|
|
580
|
-
totalCells: number;
|
|
581
|
-
cellsFailed: number;
|
|
582
|
-
totalCostUsd: number;
|
|
583
|
-
reportJsonPath: string;
|
|
584
|
-
reportMarkdownPath: string;
|
|
585
|
-
report: KnowledgeBenchmarkReport;
|
|
586
|
-
}
|
|
587
|
-
interface RunMemoryAdapterBenchmarkResult {
|
|
588
|
-
rows: readonly MemoryAdapterBenchmarkRankingRow[];
|
|
589
|
-
totalCostUsd: number;
|
|
590
|
-
/** Recovery spend for retired candidates, excluded from ranking rows but included in totalCostUsd. */
|
|
591
|
-
unrankedRecoveryCostUsd: number;
|
|
592
|
-
rankingJsonPath: string;
|
|
593
|
-
rankingMarkdownPath: string;
|
|
594
|
-
attemptLogPath: string;
|
|
595
|
-
recoveryLogPath: string;
|
|
596
|
-
}
|
|
597
|
-
interface KnowledgeRetrievalBenchmarkQuery {
|
|
598
|
-
id: string;
|
|
599
|
-
text: string;
|
|
600
|
-
split?: KnowledgeBenchmarkSplit;
|
|
601
|
-
tags?: readonly string[];
|
|
602
|
-
metadata?: Record<string, unknown>;
|
|
603
|
-
}
|
|
604
|
-
interface KnowledgeRetrievalBenchmarkQrel {
|
|
605
|
-
queryId: string;
|
|
606
|
-
documentId: string;
|
|
607
|
-
score: number;
|
|
608
|
-
}
|
|
609
|
-
interface BuildRetrievalBenchmarkCasesFromQrelsOptions {
|
|
610
|
-
benchmarkId: string;
|
|
611
|
-
family: KnowledgeBenchmarkFamily | string;
|
|
612
|
-
queries: readonly KnowledgeRetrievalBenchmarkQuery[];
|
|
613
|
-
qrels: readonly KnowledgeRetrievalBenchmarkQrel[];
|
|
614
|
-
source?: KnowledgeBenchmarkSource;
|
|
615
|
-
tags?: readonly string[];
|
|
616
|
-
k?: number;
|
|
617
|
-
targetKind?: 'page' | 'page-path' | 'source';
|
|
618
|
-
documentTarget?: (documentId: string, qrel: KnowledgeRetrievalBenchmarkQrel) => RetrievalGoldTarget;
|
|
619
|
-
splitOf?: (queryId: string) => KnowledgeBenchmarkSplit;
|
|
620
|
-
}
|
|
621
|
-
|
|
622
|
-
export { type RetrievalHoldoutConfig as $, type AgentMemoryAcquireRunLease as A, type BuildRetrievalBenchmarkCasesFromQrelsOptions as B, type KnowledgeBenchmarkScenario as C, type KnowledgeBenchmarkSliceSummary as D, type KnowledgeBenchmarkSource as E, type KnowledgeBenchmarkSpec as F, type KnowledgeBenchmarkSplit as G, type KnowledgeBenchmarkTaskKind as H, type KnowledgeClaimMatcher as I, type KnowledgeMemoryBenchmarkCase as J, type KnowledgeAnswerBenchmarkCase as K, type KnowledgeMemoryBenchmarkTaskKind as L, type KnowledgeMemoryEvent as M, type KnowledgeMemoryFactMatcher as N, type KnowledgeRetrievalBenchmarkCase as O, type KnowledgeRetrievalBenchmarkQrel as P, type KnowledgeRetrievalBenchmarkQuery as Q, type RetrievalConfig as R, type MemoryAdapterBenchmarkCandidate as S, type MemoryAdapterBenchmarkRankingRow as T, type OwnedAgentMemoryRunLease as U, type PartitionRetrievalScenariosOptions as V, type RetrievalEvalRetrieverInput as W, type RetrievalEvalRetrieverResult as X, type RetrievalGoldTarget as Y, type RetrievalHoldoutBypassReason as Z, type RetrievalHoldoutCallContext as _, type RetrievalEvalScenario as a, type RetrievalHoldoutEligibleItem as a0, type RetrievalHoldoutEvent as a1, type RetrievalHoldoutResult as a2, type RetrievalHoldoutSessionState as a3, type RetrievalMetricSummary as a4, type RetrievalRecallJudgeOptions as a5, type RetrievalScenarioPartitions as a6, type RetrievedSourceSpan as a7, type RunKnowledgeBenchmarkSuiteOptions as a8, type RunKnowledgeBenchmarkSuiteResult as a9, type RunMemoryAdapterBenchmarkOptions as aa, type RunMemoryAdapterBenchmarkResult as ab, acquireAgentMemoryRunLease as ac, buildRetrievalEvalDispatch as ad, partitionRetrievalScenarios as ae, retrievalConfigFromSurface as af, retrievalConfigSurface as ag, retrievalRecallJudge as ah, scoreRetrievalArtifact as ai, type RetrievalEvalArtifact as b, type RetrievalEvalRetriever as c, type RetrievalMetricWeights as d, type RetrievedKnowledgeHit as e, type AgentMemoryAdapter as f, type AgentMemoryBranchIsolation as g, type AgentMemoryContext as h, type AgentMemoryControllerMode as i, type AgentMemoryHit as j, type AgentMemoryKind as k, type AgentMemoryRunLease as l, type AgentMemoryScope as m, type AgentMemorySearchOptions as n, type AgentMemoryWriteInput as o, type AgentMemoryWriteResult as p, type BuildRetrievalEvalDispatchOptions as q, type KnowledgeAnswerBenchmarkTaskKind as r, type KnowledgeBenchmarkArtifact as s, type KnowledgeBenchmarkCase as t, type KnowledgeBenchmarkCaseBase as u, type KnowledgeBenchmarkDistribution as v, type KnowledgeBenchmarkEvaluation as w, type KnowledgeBenchmarkFamily as x, type KnowledgeBenchmarkReport as y, type KnowledgeBenchmarkResponder as z };
|