@tangle-network/agent-eval 0.120.0 → 0.120.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +12 -0
- package/package.json +1 -1
- package/dist/analyst/index.d.ts +0 -3111
- package/dist/analyst/index.js +0 -403
- package/dist/analyst/index.js.map +0 -1
- package/dist/authenticity/index.d.ts +0 -161
- package/dist/authenticity/index.js +0 -215
- package/dist/authenticity/index.js.map +0 -1
- package/dist/belief-state/index.d.ts +0 -1301
- package/dist/belief-state/index.js +0 -2152
- package/dist/belief-state/index.js.map +0 -1
- package/dist/benchmarks/index.d.ts +0 -974
- package/dist/benchmarks/index.js +0 -60
- package/dist/benchmarks/index.js.map +0 -1
- package/dist/builder-eval/index.d.ts +0 -695
- package/dist/builder-eval/index.js +0 -366
- package/dist/builder-eval/index.js.map +0 -1
- package/dist/campaign/index.d.ts +0 -7454
- package/dist/campaign/index.js +0 -272
- package/dist/campaign/index.js.map +0 -1
- package/dist/chunk-32BZXMSO.js +0 -3878
- package/dist/chunk-32BZXMSO.js.map +0 -1
- package/dist/chunk-3A246TSA.js +0 -998
- package/dist/chunk-3A246TSA.js.map +0 -1
- package/dist/chunk-3RF76KTD.js +0 -84
- package/dist/chunk-3RF76KTD.js.map +0 -1
- package/dist/chunk-3YYRZDON.js +0 -45
- package/dist/chunk-3YYRZDON.js.map +0 -1
- package/dist/chunk-4I2E3LLO.js +0 -1030
- package/dist/chunk-4I2E3LLO.js.map +0 -1
- package/dist/chunk-ARU2PZFM.js +0 -312
- package/dist/chunk-ARU2PZFM.js.map +0 -1
- package/dist/chunk-BOD4O7OF.js +0 -40
- package/dist/chunk-BOD4O7OF.js.map +0 -1
- package/dist/chunk-DPZAEKA6.js +0 -880
- package/dist/chunk-DPZAEKA6.js.map +0 -1
- package/dist/chunk-DTJ6QUQB.js +0 -131
- package/dist/chunk-DTJ6QUQB.js.map +0 -1
- package/dist/chunk-GGE4NNQT.js +0 -65
- package/dist/chunk-GGE4NNQT.js.map +0 -1
- package/dist/chunk-H5UD2323.js +0 -286
- package/dist/chunk-H5UD2323.js.map +0 -1
- package/dist/chunk-HHWE3POT.js +0 -94
- package/dist/chunk-HHWE3POT.js.map +0 -1
- package/dist/chunk-HKUCJ437.js +0 -787
- package/dist/chunk-HKUCJ437.js.map +0 -1
- package/dist/chunk-JHCHEVET.js +0 -274
- package/dist/chunk-JHCHEVET.js.map +0 -1
- package/dist/chunk-JHOJHHU7.js +0 -867
- package/dist/chunk-JHOJHHU7.js.map +0 -1
- package/dist/chunk-JM2SKQMS.js +0 -750
- package/dist/chunk-JM2SKQMS.js.map +0 -1
- package/dist/chunk-JN2FCO5W.js +0 -7958
- package/dist/chunk-JN2FCO5W.js.map +0 -1
- package/dist/chunk-K4DBDHLK.js +0 -158
- package/dist/chunk-K4DBDHLK.js.map +0 -1
- package/dist/chunk-K6N6XJJX.js +0 -306
- package/dist/chunk-K6N6XJJX.js.map +0 -1
- package/dist/chunk-MA6HLL3S.js +0 -65
- package/dist/chunk-MA6HLL3S.js.map +0 -1
- package/dist/chunk-MAZ26DC7.js +0 -99
- package/dist/chunk-MAZ26DC7.js.map +0 -1
- package/dist/chunk-MOXWMGPC.js +0 -577
- package/dist/chunk-MOXWMGPC.js.map +0 -1
- package/dist/chunk-NJC7U437.js +0 -626
- package/dist/chunk-NJC7U437.js.map +0 -1
- package/dist/chunk-NPCTHQIO.js +0 -91
- package/dist/chunk-NPCTHQIO.js.map +0 -1
- package/dist/chunk-ONWEPEDO.js +0 -57
- package/dist/chunk-ONWEPEDO.js.map +0 -1
- package/dist/chunk-OYZAPX5G.js +0 -1526
- package/dist/chunk-OYZAPX5G.js.map +0 -1
- package/dist/chunk-PC4UYEBM.js +0 -166
- package/dist/chunk-PC4UYEBM.js.map +0 -1
- package/dist/chunk-PICTDURQ.js +0 -766
- package/dist/chunk-PICTDURQ.js.map +0 -1
- package/dist/chunk-PJQFMIOX.js +0 -1182
- package/dist/chunk-PJQFMIOX.js.map +0 -1
- package/dist/chunk-PXD6ZFNY.js +0 -1107
- package/dist/chunk-PXD6ZFNY.js.map +0 -1
- package/dist/chunk-PXE2VKMX.js +0 -140
- package/dist/chunk-PXE2VKMX.js.map +0 -1
- package/dist/chunk-PZ5AY32C.js +0 -10
- package/dist/chunk-PZ5AY32C.js.map +0 -1
- package/dist/chunk-QBRSJK47.js +0 -622
- package/dist/chunk-QBRSJK47.js.map +0 -1
- package/dist/chunk-QWMPPZ3X.js +0 -550
- package/dist/chunk-QWMPPZ3X.js.map +0 -1
- package/dist/chunk-S3UZOQ5Y.js +0 -328
- package/dist/chunk-S3UZOQ5Y.js.map +0 -1
- package/dist/chunk-S5TT5R3L.js +0 -2668
- package/dist/chunk-S5TT5R3L.js.map +0 -1
- package/dist/chunk-T4SQEITX.js +0 -95
- package/dist/chunk-T4SQEITX.js.map +0 -1
- package/dist/chunk-TT4KNT67.js +0 -124
- package/dist/chunk-TT4KNT67.js.map +0 -1
- package/dist/chunk-U5CHZ5M3.js +0 -357
- package/dist/chunk-U5CHZ5M3.js.map +0 -1
- package/dist/chunk-ULOKLHIQ.js +0 -1937
- package/dist/chunk-ULOKLHIQ.js.map +0 -1
- package/dist/chunk-VI2UW6B6.js +0 -162
- package/dist/chunk-VI2UW6B6.js.map +0 -1
- package/dist/chunk-VQMK5FMP.js +0 -247
- package/dist/chunk-VQMK5FMP.js.map +0 -1
- package/dist/chunk-VSMTAMNK.js +0 -53
- package/dist/chunk-VSMTAMNK.js.map +0 -1
- package/dist/chunk-VZSRQ272.js +0 -149
- package/dist/chunk-VZSRQ272.js.map +0 -1
- package/dist/chunk-WW2A73HW.js +0 -159
- package/dist/chunk-WW2A73HW.js.map +0 -1
- package/dist/chunk-X4UCIOTZ.js +0 -136
- package/dist/chunk-X4UCIOTZ.js.map +0 -1
- package/dist/chunk-XDIRG3TO.js +0 -1266
- package/dist/chunk-XDIRG3TO.js.map +0 -1
- package/dist/chunk-XJYR7XFV.js +0 -317
- package/dist/chunk-XJYR7XFV.js.map +0 -1
- package/dist/chunk-ZET2UAYW.js +0 -89
- package/dist/chunk-ZET2UAYW.js.map +0 -1
- package/dist/chunk-ZZUXHH3R.js +0 -99
- package/dist/chunk-ZZUXHH3R.js.map +0 -1
- package/dist/cli.d.ts +0 -1
- package/dist/cli.js +0 -112
- package/dist/cli.js.map +0 -1
- package/dist/contract/index.d.ts +0 -4969
- package/dist/contract/index.js +0 -1653
- package/dist/contract/index.js.map +0 -1
- package/dist/control.d.ts +0 -1013
- package/dist/control.js +0 -34
- package/dist/control.js.map +0 -1
- package/dist/fuzz.d.ts +0 -759
- package/dist/fuzz.js +0 -714
- package/dist/fuzz.js.map +0 -1
- package/dist/hosted/index.d.ts +0 -730
- package/dist/hosted/index.js +0 -14
- package/dist/hosted/index.js.map +0 -1
- package/dist/index.d.ts +0 -16780
- package/dist/index.js +0 -12168
- package/dist/index.js.map +0 -1
- package/dist/matrix/index.d.ts +0 -155
- package/dist/matrix/index.js +0 -8
- package/dist/matrix/index.js.map +0 -1
- package/dist/meta-eval/index.d.ts +0 -1030
- package/dist/meta-eval/index.js +0 -417
- package/dist/meta-eval/index.js.map +0 -1
- package/dist/multishot/index.d.ts +0 -579
- package/dist/multishot/index.js +0 -589
- package/dist/multishot/index.js.map +0 -1
- package/dist/openapi.json +0 -992
- package/dist/pipelines/index.d.ts +0 -567
- package/dist/pipelines/index.js +0 -515
- package/dist/pipelines/index.js.map +0 -1
- package/dist/reporting.d.ts +0 -1277
- package/dist/reporting.js +0 -48
- package/dist/reporting.js.map +0 -1
- package/dist/rl.d.ts +0 -4092
- package/dist/rl.js +0 -1724
- package/dist/rl.js.map +0 -1
- package/dist/run-campaign-HNFPJET4.js +0 -14
- package/dist/run-campaign-HNFPJET4.js.map +0 -1
- package/dist/storyboard/index.d.ts +0 -279
- package/dist/storyboard/index.js +0 -767
- package/dist/storyboard/index.js.map +0 -1
- package/dist/trace-attributes.d.ts +0 -52
- package/dist/trace-attributes.js +0 -62
- package/dist/trace-attributes.js.map +0 -1
- package/dist/traces.d.ts +0 -2343
- package/dist/traces.js +0 -249
- package/dist/traces.js.map +0 -1
- package/dist/wire/index.d.ts +0 -1252
- package/dist/wire/index.js +0 -81
- package/dist/wire/index.js.map +0 -1
|
@@ -1,579 +0,0 @@
|
|
|
1
|
-
import { AgentProfile } from '@tangle-network/agent-interface';
|
|
2
|
-
|
|
3
|
-
interface MultishotMessage {
|
|
4
|
-
role: 'user' | 'assistant' | 'tool';
|
|
5
|
-
content: string;
|
|
6
|
-
toolCallId?: string;
|
|
7
|
-
toolCalls?: Array<{
|
|
8
|
-
id: string;
|
|
9
|
-
name: string;
|
|
10
|
-
args: Record<string, unknown>;
|
|
11
|
-
}>;
|
|
12
|
-
}
|
|
13
|
-
interface MultishotArtifact {
|
|
14
|
-
type: string;
|
|
15
|
-
turn: number;
|
|
16
|
-
invocation: {
|
|
17
|
-
name: string;
|
|
18
|
-
args: Record<string, unknown>;
|
|
19
|
-
};
|
|
20
|
-
content: string;
|
|
21
|
-
}
|
|
22
|
-
interface MultishotResult {
|
|
23
|
-
transcript: MultishotMessage[];
|
|
24
|
-
artifacts: MultishotArtifact[];
|
|
25
|
-
toolCalls: number;
|
|
26
|
-
durationMs: number;
|
|
27
|
-
costUsd: number;
|
|
28
|
-
}
|
|
29
|
-
interface MultishotToolDefinition {
|
|
30
|
-
type: 'function';
|
|
31
|
-
function: {
|
|
32
|
-
name: string;
|
|
33
|
-
description: string;
|
|
34
|
-
parameters: Record<string, unknown>;
|
|
35
|
-
};
|
|
36
|
-
}
|
|
37
|
-
/** One chat-completion request the multishot loop issues for a single agent
|
|
38
|
-
* (or driver) inference step. Mirrors the OpenAI-compat body the loop would
|
|
39
|
-
* otherwise POST to the Tangle router. */
|
|
40
|
-
interface MultishotTransportRequest {
|
|
41
|
-
model: string;
|
|
42
|
-
messages: Array<Record<string, unknown>>;
|
|
43
|
-
tools?: MultishotToolDefinition[];
|
|
44
|
-
temperature?: number;
|
|
45
|
-
maxTokens?: number;
|
|
46
|
-
signal?: AbortSignal;
|
|
47
|
-
}
|
|
48
|
-
interface MultishotTransportToolCall {
|
|
49
|
-
id: string;
|
|
50
|
-
type: 'function';
|
|
51
|
-
function: {
|
|
52
|
-
name: string;
|
|
53
|
-
arguments: string;
|
|
54
|
-
};
|
|
55
|
-
}
|
|
56
|
-
interface MultishotTransportResponse {
|
|
57
|
-
message: {
|
|
58
|
-
content?: string | null;
|
|
59
|
-
tool_calls?: MultishotTransportToolCall[];
|
|
60
|
-
};
|
|
61
|
-
usage?: {
|
|
62
|
-
prompt_tokens?: number;
|
|
63
|
-
completion_tokens?: number;
|
|
64
|
-
};
|
|
65
|
-
/** Actual spend for this call. When omitted, the loop meters cost from
|
|
66
|
-
* `usage` via the per-model router estimator (estimateRouterCost). */
|
|
67
|
-
costUsd?: number;
|
|
68
|
-
}
|
|
69
|
-
/** Execution seam for one leg of the multishot loop. When provided, it
|
|
70
|
-
* replaces the internal router HTTP call for that leg — the loop still owns
|
|
71
|
-
* turn scheduling, tool dispatch, transcript capture, and cost metering.
|
|
72
|
-
* agent-eval has no dependency on agent-runtime; adapt agent-runtime's
|
|
73
|
-
* resolveAgentBackend (or any sandbox/cli-bridge/router client) into this
|
|
74
|
-
* signature product-side. */
|
|
75
|
-
type MultishotTransport = (req: MultishotTransportRequest) => Promise<MultishotTransportResponse>;
|
|
76
|
-
type MultishotToolExecutor = (args: Record<string, unknown>, ctx: {
|
|
77
|
-
apiKey: string;
|
|
78
|
-
baseUrl: string;
|
|
79
|
-
signal?: AbortSignal;
|
|
80
|
-
}) => Promise<{
|
|
81
|
-
content: string;
|
|
82
|
-
costUsd: number;
|
|
83
|
-
}>;
|
|
84
|
-
interface MultishotPersona {
|
|
85
|
-
/** Stable identifier — used for per-cell artifact paths + matrix axis keys. */
|
|
86
|
-
id: string;
|
|
87
|
-
/** Per-domain payload (income/profile/voice/etc.) shaped by the consumer. */
|
|
88
|
-
[k: string]: unknown;
|
|
89
|
-
}
|
|
90
|
-
interface MultishotShape<TPersona extends MultishotPersona> {
|
|
91
|
-
/** Opening user message (turn 0) — the persona's first ask. */
|
|
92
|
-
buildOpener: (persona: TPersona) => string;
|
|
93
|
-
/** System prompt the driver LLM uses to roleplay the persona. Should set
|
|
94
|
-
* voice, goals, constraints, time-pressure, and the "never go silent" rule. */
|
|
95
|
-
buildDriverSystemPrompt: (persona: TPersona) => string;
|
|
96
|
-
}
|
|
97
|
-
declare class MultishotDriverEmptyError extends Error {
|
|
98
|
-
readonly turn: number;
|
|
99
|
-
constructor(turn: number);
|
|
100
|
-
}
|
|
101
|
-
declare class MultishotFatalToolError extends Error {
|
|
102
|
-
constructor(message: string);
|
|
103
|
-
}
|
|
104
|
-
|
|
105
|
-
interface RouterCompletionRequest {
|
|
106
|
-
apiKey: string;
|
|
107
|
-
baseUrl: string;
|
|
108
|
-
model: string;
|
|
109
|
-
messages: Array<Record<string, unknown>>;
|
|
110
|
-
tools?: MultishotToolDefinition[];
|
|
111
|
-
temperature?: number;
|
|
112
|
-
maxTokens?: number;
|
|
113
|
-
signal?: AbortSignal;
|
|
114
|
-
}
|
|
115
|
-
interface RouterToolCall {
|
|
116
|
-
id: string;
|
|
117
|
-
type: 'function';
|
|
118
|
-
function: {
|
|
119
|
-
name: string;
|
|
120
|
-
arguments: string;
|
|
121
|
-
};
|
|
122
|
-
}
|
|
123
|
-
interface RouterCompletionResponse {
|
|
124
|
-
message: {
|
|
125
|
-
content?: string | null;
|
|
126
|
-
tool_calls?: RouterToolCall[];
|
|
127
|
-
};
|
|
128
|
-
usage?: {
|
|
129
|
-
prompt_tokens?: number;
|
|
130
|
-
completion_tokens?: number;
|
|
131
|
-
};
|
|
132
|
-
}
|
|
133
|
-
declare function routerCompletion(req: RouterCompletionRequest): Promise<RouterCompletionResponse>;
|
|
134
|
-
declare function estimateRouterCost(model: string, usage?: {
|
|
135
|
-
prompt_tokens?: number;
|
|
136
|
-
completion_tokens?: number;
|
|
137
|
-
}): number;
|
|
138
|
-
declare function defaultRouterBaseUrl(): string;
|
|
139
|
-
declare function requireRouterApiKey(): string;
|
|
140
|
-
|
|
141
|
-
declare const DEFAULT_RESEARCHER_MODEL = "openai/gpt-4o-mini";
|
|
142
|
-
declare const DEFAULT_CODER_MODEL = "openai/gpt-4o-mini";
|
|
143
|
-
interface DefaultResearcherConfig {
|
|
144
|
-
/** Replace the system prompt to bias the researcher toward a domain's
|
|
145
|
-
* citation style. Defaults to a generic "cite sources by name" prompt. */
|
|
146
|
-
systemPrompt?: string;
|
|
147
|
-
model?: string;
|
|
148
|
-
}
|
|
149
|
-
interface DefaultCoderConfig {
|
|
150
|
-
/** Replace the system prompt to bias the coder toward a language /
|
|
151
|
-
* framework / artifact style. */
|
|
152
|
-
systemPrompt?: string;
|
|
153
|
-
model?: string;
|
|
154
|
-
}
|
|
155
|
-
declare const DEFAULT_DELEGATE_RESEARCH_TOOL: MultishotToolDefinition;
|
|
156
|
-
declare const DEFAULT_DELEGATE_CODE_TOOL: MultishotToolDefinition;
|
|
157
|
-
declare function createResearchExecutor(config?: DefaultResearcherConfig): MultishotToolExecutor;
|
|
158
|
-
declare function createCodeExecutor(config?: DefaultCoderConfig): MultishotToolExecutor;
|
|
159
|
-
interface DefaultToolsConfig {
|
|
160
|
-
research?: DefaultResearcherConfig;
|
|
161
|
-
code?: DefaultCoderConfig;
|
|
162
|
-
/** When true (default), each tool result is recorded as a typed artifact:
|
|
163
|
-
* research → type='research', code → type='code'. */
|
|
164
|
-
recordArtifacts?: boolean;
|
|
165
|
-
}
|
|
166
|
-
interface DefaultToolsBundle {
|
|
167
|
-
tools: MultishotToolDefinition[];
|
|
168
|
-
executors: Record<string, MultishotToolExecutor>;
|
|
169
|
-
artifactTypeFor: (toolName: string) => string | undefined;
|
|
170
|
-
}
|
|
171
|
-
declare function defaultDelegationTools(config?: DefaultToolsConfig): DefaultToolsBundle;
|
|
172
|
-
|
|
173
|
-
/**
|
|
174
|
-
* LLM client with graceful degrade.
|
|
175
|
-
*
|
|
176
|
-
* OpenAI-compatible `/v1/chat/completions` client with:
|
|
177
|
-
* - Exponential-backoff retry on 429 + 5xx gateway errors (502/503/504).
|
|
178
|
-
* - Retry on transient network errors (fetch failed, AbortError, ECONNRESET).
|
|
179
|
-
* - Graceful json_schema → json_object degrade on 400 with schema-reject body.
|
|
180
|
-
* - Fenced-JSON stripping (```json ... ```) for models that wrap structured output.
|
|
181
|
-
* - Configurable base URL + api key / bearer, works with LiteLLM proxies, OpenAI
|
|
182
|
-
* directly, cli-bridge subscriptions, and any router that speaks the spec.
|
|
183
|
-
*
|
|
184
|
-
* Usage:
|
|
185
|
-
* const { value, result } = await callLlmJson<MyType>(
|
|
186
|
-
* { model: 'gpt-4o', messages: [...], jsonSchema: { name: 'x', schema: {...} } },
|
|
187
|
-
* { baseUrl: 'https://router.tangle.tools/v1', apiKey: process.env.KEY },
|
|
188
|
-
* )
|
|
189
|
-
*
|
|
190
|
-
* This is THE llm-calling seam for agent-eval primitives that need structured
|
|
191
|
-
* output (semantic concept judge, reviewer directives, critic scores). Primitives
|
|
192
|
-
* that need free-form text use `callLlm` and parse output themselves.
|
|
193
|
-
*/
|
|
194
|
-
|
|
195
|
-
interface LlmUsage {
|
|
196
|
-
promptTokens: number;
|
|
197
|
-
completionTokens: number;
|
|
198
|
-
totalTokens: number;
|
|
199
|
-
/** False when the provider omitted or malformed prompt/completion usage. */
|
|
200
|
-
captured?: boolean;
|
|
201
|
-
/** Proxies populate this when prompt caching is on. */
|
|
202
|
-
cachedPromptTokens?: number;
|
|
203
|
-
}
|
|
204
|
-
interface LlmCallResult {
|
|
205
|
-
/** The text content of the first choice. Empty string if none. */
|
|
206
|
-
content: string;
|
|
207
|
-
usage: LlmUsage;
|
|
208
|
-
/**
|
|
209
|
-
* Cost in USD. Pulled from proxy's `_response_cost` field when present;
|
|
210
|
-
* `null` when neither the proxy nor the caller can derive it.
|
|
211
|
-
*/
|
|
212
|
-
costUsd: number | null;
|
|
213
|
-
/** Model name actually used (echoed from response). */
|
|
214
|
-
model: string;
|
|
215
|
-
/** Wall-clock duration of the HTTP call (last attempt, if retried). */
|
|
216
|
-
durationMs: number;
|
|
217
|
-
/**
|
|
218
|
-
* `finish_reason` echoed from the first choice (`stop`, `length`,
|
|
219
|
-
* `content_filter`, `tool_calls`, ...). `null` when the provider omits it.
|
|
220
|
-
* Exposed so a free-form `callLlm` caller CAN detect a truncated answer
|
|
221
|
-
* (`length`) instead of treating a cut-off completion as complete. Note:
|
|
222
|
-
* `callLlm` does not itself reject on it — acting on this signal is the
|
|
223
|
-
* caller's responsibility (in-repo free-form drivers do not yet enforce it).
|
|
224
|
-
*/
|
|
225
|
-
finishReason?: string | null;
|
|
226
|
-
/**
|
|
227
|
-
* True when `content.trim()` is empty. An empty completion is a silent zero
|
|
228
|
-
* for free-form `callLlm` callers; this flag is the signal a caller can
|
|
229
|
-
* inspect to fail loud rather than proceed on an empty string. `callLlm`
|
|
230
|
-
* surfaces it but does not throw on it.
|
|
231
|
-
*/
|
|
232
|
-
contentEmpty?: boolean;
|
|
233
|
-
/** Raw response body. */
|
|
234
|
-
raw: Record<string, unknown>;
|
|
235
|
-
}
|
|
236
|
-
type LlmCallMetadata = Pick<LlmCallResult, 'usage' | 'costUsd' | 'model' | 'durationMs'>;
|
|
237
|
-
|
|
238
|
-
/**
|
|
239
|
-
* Pass A substrate types — `runCampaign` is the one primitive every
|
|
240
|
-
* eval flow composes from. Three contracts in this file:
|
|
241
|
-
*
|
|
242
|
-
* - `Scenario` input set
|
|
243
|
-
* - `DispatchFn` how to run one scenario → artifact
|
|
244
|
-
* - `CampaignResult` defined output schema (the contract downstream tools depend on)
|
|
245
|
-
*
|
|
246
|
-
* Three more lifted from earlier substrate work (re-exported):
|
|
247
|
-
*
|
|
248
|
-
* - `JudgeConfig` pluggable dimensional scorer (0.38)
|
|
249
|
-
* - `Mutator` optimization-loop surface mutator
|
|
250
|
-
* - `Gate` promotion gate (`HeldOutGate` and friends adapt to this)
|
|
251
|
-
*
|
|
252
|
-
* No new architecture vs 0.38 — Pass A formalizes the shapes so consumers
|
|
253
|
-
* can build dashboards / CI gates / regression diffs against a stable schema.
|
|
254
|
-
*/
|
|
255
|
-
|
|
256
|
-
/** The canonical judge verdict shape — one declaration, shared by campaign
|
|
257
|
-
* judges and the multishot judge runner (which re-exports this type).
|
|
258
|
-
*
|
|
259
|
-
* Scale is PRODUCER-DEFINED: campaign convention is [0,1]; the legacy
|
|
260
|
-
* multishot runner emits 0-10. Cross-scale comparison must go through
|
|
261
|
-
* `detectScale` (src/campaign/gates/statistical-heldout.ts, used by
|
|
262
|
-
* promotion-policy) — never renormalize a producer's values in place, as
|
|
263
|
-
* downstream thresholds (`composite >= 5` in multishot/matrix.ts, live-soak
|
|
264
|
-
* `>= 7` gates) key on the producer's native scale. */
|
|
265
|
-
interface JudgeScore {
|
|
266
|
-
dimensions: Record<string, number>;
|
|
267
|
-
composite: number;
|
|
268
|
-
notes: string;
|
|
269
|
-
/** Provider metadata for display and diagnostics; accounting uses CostLedger receipts. */
|
|
270
|
-
llmCall?: LlmCallMetadata;
|
|
271
|
-
/** Set when the judge itself failed (call error, unparseable output).
|
|
272
|
-
* `composite`/`dimensions` carry no signal — aggregators MUST exclude
|
|
273
|
-
* failed scores from means instead of folding them into zeros. */
|
|
274
|
-
failed?: true;
|
|
275
|
-
/** Ensemble extras (populated by `ensembleJudge`): max per-dimension
|
|
276
|
-
* spread across surviving judges — the inter-rater signal. */
|
|
277
|
-
maxDisagreement?: number;
|
|
278
|
-
/** Ensemble extras: judge identities whose verdict failed. */
|
|
279
|
-
failedJudges?: string[];
|
|
280
|
-
/** Ensemble extras: each surviving judge's per-dimension scores. */
|
|
281
|
-
perJudge?: Record<string, Record<string, number>>;
|
|
282
|
-
}
|
|
283
|
-
|
|
284
|
-
declare const DEFAULT_JUDGE_MODEL = "openai/gpt-4o-mini";
|
|
285
|
-
interface JudgeDimension {
|
|
286
|
-
/** JSON field name + score key. */
|
|
287
|
-
key: string;
|
|
288
|
-
/** Description shown in the judge's user prompt. */
|
|
289
|
-
description: string;
|
|
290
|
-
}
|
|
291
|
-
interface JudgeConfig<TInput> {
|
|
292
|
-
/** Display name (for trace + log). */
|
|
293
|
-
name: string;
|
|
294
|
-
/** Model used for this judge. */
|
|
295
|
-
model?: string;
|
|
296
|
-
/** 0-10 scored dimensions. */
|
|
297
|
-
dimensions: JudgeDimension[];
|
|
298
|
-
/** Judge system prompt — sets persona + JSON-only constraint. */
|
|
299
|
-
systemPrompt: string;
|
|
300
|
-
/** Build the user prompt from the typed input. Must include "Respond with
|
|
301
|
-
* ONLY this JSON: { ... }" listing each dimension key. */
|
|
302
|
-
buildPrompt: (input: TInput) => string;
|
|
303
|
-
/** Optional model + api overrides. */
|
|
304
|
-
apiKey?: string;
|
|
305
|
-
baseUrl?: string;
|
|
306
|
-
/** Maximum output tokens for the judge response. Defaults to 1500. */
|
|
307
|
-
maxTokens?: number;
|
|
308
|
-
}
|
|
309
|
-
declare function runJudge<TInput>(judge: JudgeConfig<TInput>, input: TInput): Promise<JudgeScore>;
|
|
310
|
-
/** Convenience: stringified dimension list for inclusion in a judge prompt.
|
|
311
|
-
* Returns lines like `- audience_fit: Does this match what the audience cares about? (0-10)`. */
|
|
312
|
-
declare function renderDimensions(dims: readonly JudgeDimension[]): string;
|
|
313
|
-
/** Convenience: build the "Respond with ONLY this JSON" footer for a judge prompt. */
|
|
314
|
-
declare function renderJsonFooter(dims: readonly JudgeDimension[]): string;
|
|
315
|
-
|
|
316
|
-
/**
|
|
317
|
-
* Validator-output verdict — substrate primitive for "did this output pass,
|
|
318
|
-
* and how well?"
|
|
319
|
-
*
|
|
320
|
-
* Used by:
|
|
321
|
-
* - `@tangle-network/agent-eval/matrix` — verdict per cell in the cartesian.
|
|
322
|
-
* - `@tangle-network/agent-runtime` — Validator<Output, Verdict = DefaultVerdict>.
|
|
323
|
-
* Runtime keeps `Validator` because it's coupled to runtime-shaped
|
|
324
|
-
* `ValidationCtx` (iteration, signal, traceEmitter); the verdict TYPE
|
|
325
|
-
* itself is a substrate concept and lives here.
|
|
326
|
-
*
|
|
327
|
-
* Repo layering: agent-eval is the substrate (no upward deps). Both
|
|
328
|
-
* agent-runtime and agent-knowledge consume this type FROM agent-eval —
|
|
329
|
-
* never the other way around. See CLAUDE.md "Repo layering" for the rule.
|
|
330
|
-
*/
|
|
331
|
-
/**
|
|
332
|
-
* Minimal verdict shape — `valid` + `score` are required; `scores` +
|
|
333
|
-
* `notes` are optional surface. Validators that need richer shapes
|
|
334
|
-
* parameterise `Validator<Output, MyVerdict>` with their own type.
|
|
335
|
-
*
|
|
336
|
-
* Need structured extras? Extend DefaultVerdict with typed fields — never
|
|
337
|
-
* serialize extras into `notes`.
|
|
338
|
-
*/
|
|
339
|
-
interface DefaultVerdict {
|
|
340
|
-
/** Whether the output meets the validator's pass criteria. */
|
|
341
|
-
valid: boolean;
|
|
342
|
-
/** Aggregate score in [0, 1]. Drivers use this for winner selection. */
|
|
343
|
-
score: number;
|
|
344
|
-
/** Per-dimension scores. Free-form; weighted into `score` by the validator. */
|
|
345
|
-
scores?: Record<string, number>;
|
|
346
|
-
/** Human-readable rationale; surfaces in trace + final-result `winner.verdict`. */
|
|
347
|
-
notes?: string;
|
|
348
|
-
}
|
|
349
|
-
|
|
350
|
-
/**
|
|
351
|
-
* N-axis cartesian matrix over substrate types — types module.
|
|
352
|
-
*
|
|
353
|
-
* The matrix is a runner + aggregator. It iterates the cartesian product of
|
|
354
|
-
* caller-provided axes (any value type — `AgentProfile` from agent-interface,
|
|
355
|
-
* `Driver` / `Validator` from agent-runtime, rubric records, thinking levels, anything)
|
|
356
|
-
* and aggregates per-axis pass/score/cost summaries. Substrate types are
|
|
357
|
-
* imported at the boundary by JSDoc only; the matrix never wraps them.
|
|
358
|
-
*/
|
|
359
|
-
|
|
360
|
-
/** A cell carries one picked value from each axis, keyed by axis name. */
|
|
361
|
-
interface MatrixCell {
|
|
362
|
-
axes: Record<string, {
|
|
363
|
-
id: string;
|
|
364
|
-
value: unknown;
|
|
365
|
-
}>;
|
|
366
|
-
/** 0-based replicate index within the same axis combination. */
|
|
367
|
-
rep: number;
|
|
368
|
-
/** Stable sort key — preserves cartesian order across concurrent execution. */
|
|
369
|
-
ordinal: number;
|
|
370
|
-
}
|
|
371
|
-
interface CellResult<Output> {
|
|
372
|
-
output: Output;
|
|
373
|
-
verdict: DefaultVerdict;
|
|
374
|
-
costUsd: number;
|
|
375
|
-
durationMs: number;
|
|
376
|
-
runId?: string;
|
|
377
|
-
/** Populated when `runCell` threw. The cell contributes 0 to passRate AND
|
|
378
|
-
* meanScore regardless of `verdict`. */
|
|
379
|
-
error?: {
|
|
380
|
-
message: string;
|
|
381
|
-
kind: string;
|
|
382
|
-
};
|
|
383
|
-
}
|
|
384
|
-
interface AxisSummary {
|
|
385
|
-
axisName: string;
|
|
386
|
-
axisValue: string;
|
|
387
|
-
cells: number;
|
|
388
|
-
passRate: number;
|
|
389
|
-
meanScore: number;
|
|
390
|
-
p50Score: number;
|
|
391
|
-
p90Score: number;
|
|
392
|
-
totalCostUsd: number;
|
|
393
|
-
meanDurationMs: number;
|
|
394
|
-
}
|
|
395
|
-
interface MatrixResult<Output> {
|
|
396
|
-
cells: Array<{
|
|
397
|
-
cell: MatrixCell;
|
|
398
|
-
runs: CellResult<Output>[];
|
|
399
|
-
}>;
|
|
400
|
-
/** `byAxis[axisName][axisValueId] = summary`. Populated only for axes
|
|
401
|
-
* named in `aggregateBy` (default = every axis in `axes`). */
|
|
402
|
-
byAxis: Record<string, Record<string, AxisSummary>>;
|
|
403
|
-
summary: {
|
|
404
|
-
totalCells: number;
|
|
405
|
-
runsExecuted: number;
|
|
406
|
-
/** Cells removed by `filter` plus cells unscheduled after the cost
|
|
407
|
-
* ceiling or abort signal tripped. */
|
|
408
|
-
cellsSkipped: number;
|
|
409
|
-
overallPassRate: number;
|
|
410
|
-
overallMeanScore: number;
|
|
411
|
-
totalCostUsd: number;
|
|
412
|
-
durationMs: number;
|
|
413
|
-
};
|
|
414
|
-
/** Stable id-like string generated at the end of the run. */
|
|
415
|
-
matrixId: string;
|
|
416
|
-
}
|
|
417
|
-
|
|
418
|
-
interface ConversationJudgeInput<TPersona extends MultishotPersona> {
|
|
419
|
-
transcript: MultishotMessage[];
|
|
420
|
-
persona: TPersona;
|
|
421
|
-
}
|
|
422
|
-
interface ArtifactJudgeInput<TPersona extends MultishotPersona> {
|
|
423
|
-
artifact: MultishotArtifact;
|
|
424
|
-
persona: TPersona;
|
|
425
|
-
}
|
|
426
|
-
interface MultishotJudges<TPersona extends MultishotPersona> {
|
|
427
|
-
/** Scores the full transcript end-to-end (always runs). */
|
|
428
|
-
conversation: JudgeConfig<ConversationJudgeInput<TPersona>>;
|
|
429
|
-
/** Scores each code-type artifact. Optional — omit when domain has no code artifacts. */
|
|
430
|
-
codeReview?: JudgeConfig<ArtifactJudgeInput<TPersona>>;
|
|
431
|
-
/** Scores each non-code (research/content/template) artifact. Optional. */
|
|
432
|
-
contentQuality?: JudgeConfig<ArtifactJudgeInput<TPersona>>;
|
|
433
|
-
/** Which artifact types route to codeReview. Defaults to ['code']. */
|
|
434
|
-
codeArtifactTypes?: string[];
|
|
435
|
-
/** Which artifact types route to contentQuality. Defaults to ['research']. */
|
|
436
|
-
contentArtifactTypes?: string[];
|
|
437
|
-
}
|
|
438
|
-
interface CellCompositeScore {
|
|
439
|
-
composite: number;
|
|
440
|
-
conversation: JudgeScore;
|
|
441
|
-
codeReview?: {
|
|
442
|
-
perArtifact: Array<JudgeScore & {
|
|
443
|
-
turn: number;
|
|
444
|
-
type: string;
|
|
445
|
-
}>;
|
|
446
|
-
composite: number;
|
|
447
|
-
};
|
|
448
|
-
contentQuality?: {
|
|
449
|
-
perArtifact: Array<JudgeScore & {
|
|
450
|
-
turn: number;
|
|
451
|
-
type: string;
|
|
452
|
-
}>;
|
|
453
|
-
composite: number;
|
|
454
|
-
};
|
|
455
|
-
}
|
|
456
|
-
interface RunMultishotMatrixOptions<TPersona extends MultishotPersona> {
|
|
457
|
-
/** AgentProfile axis (matrix primary). */
|
|
458
|
-
profiles: Array<{
|
|
459
|
-
id: string;
|
|
460
|
-
value: AgentProfile;
|
|
461
|
-
}>;
|
|
462
|
-
/** Persona axis. */
|
|
463
|
-
personas: TPersona[];
|
|
464
|
-
/** Persona-shaping callbacks. */
|
|
465
|
-
shape: MultishotShape<TPersona>;
|
|
466
|
-
/** Judge configurations. */
|
|
467
|
-
judges: MultishotJudges<TPersona>;
|
|
468
|
-
/** Tool definitions advertised to the agent. Defaults to delegate_research + delegate_code. */
|
|
469
|
-
tools?: MultishotToolDefinition[];
|
|
470
|
-
/** Map from tool name → inline executor. Must align with `tools`. */
|
|
471
|
-
toolExecutors?: Record<string, MultishotToolExecutor>;
|
|
472
|
-
/** Tool name → artifact type label. Defaults to research/code mapping. */
|
|
473
|
-
artifactTypeFor?: (toolName: string) => string | undefined;
|
|
474
|
-
/** Where per-cell artifacts land. Cells write to `<runDir>/<profileId>/<personaId>/rep-N/`. */
|
|
475
|
-
runDir: string;
|
|
476
|
-
/** Replicates per (profile, persona) cell. */
|
|
477
|
-
reps?: number;
|
|
478
|
-
/** Max conversation turns per cell. */
|
|
479
|
-
maxTurns?: number;
|
|
480
|
-
/** Maximum tool calls the agent may dispatch inside one assistant turn. */
|
|
481
|
-
maxToolDispatches?: number;
|
|
482
|
-
/** Max concurrent cells. */
|
|
483
|
-
maxConcurrency?: number;
|
|
484
|
-
/** Total $ ceiling across the matrix; cells aborted past this. */
|
|
485
|
-
costCeiling?: number;
|
|
486
|
-
/** Agent model. */
|
|
487
|
-
agentModel?: string;
|
|
488
|
-
/** Driver model. */
|
|
489
|
-
driverModel?: string;
|
|
490
|
-
/** Fallback driver models tried when the primary simulated-user model returns empty twice. */
|
|
491
|
-
driverFallbackModels?: string[];
|
|
492
|
-
/** Maximum output tokens for the first agent call in each assistant turn. */
|
|
493
|
-
agentMaxTokens?: number;
|
|
494
|
-
/** Maximum output tokens for agent follow-up calls after tool results. */
|
|
495
|
-
toolFollowupMaxTokens?: number;
|
|
496
|
-
/** Maximum output tokens for each simulated-user driver response. */
|
|
497
|
-
driverMaxTokens?: number;
|
|
498
|
-
/** Maximum output tokens for each judge response. */
|
|
499
|
-
judgeMaxTokens?: number;
|
|
500
|
-
/** Execution seam for the agent leg of every cell — replaces the router
|
|
501
|
-
* HTTP call when provided (see RunMultishotOptions.agentTransport).
|
|
502
|
-
* Judges are unaffected; configure those via MultishotJudges. */
|
|
503
|
-
agentTransport?: MultishotTransport;
|
|
504
|
-
/** Execution seam for the simulated-user driver leg of every cell. */
|
|
505
|
-
driverTransport?: MultishotTransport;
|
|
506
|
-
/** Pass-thru fields. */
|
|
507
|
-
apiKey?: string;
|
|
508
|
-
baseUrl?: string;
|
|
509
|
-
}
|
|
510
|
-
interface CellOutput {
|
|
511
|
-
turns: number;
|
|
512
|
-
toolCalls: number;
|
|
513
|
-
artifactCount: number;
|
|
514
|
-
}
|
|
515
|
-
interface CellCompositeInput {
|
|
516
|
-
conversation: JudgeScore;
|
|
517
|
-
/** Present iff the codeReview judge is configured. */
|
|
518
|
-
codeReviews?: ReadonlyArray<JudgeScore>;
|
|
519
|
-
/** Present iff the contentQuality judge is configured. */
|
|
520
|
-
contentReviews?: ReadonlyArray<JudgeScore>;
|
|
521
|
-
}
|
|
522
|
-
/** Cell composite = mean over configured judge slots, excluding failed
|
|
523
|
-
* scores: a failed conversation judge or an all-failed artifact slot carries
|
|
524
|
-
* no signal and is dropped from the mean. `composite` is 0 only when EVERY
|
|
525
|
-
* configured slot failed (`allJudgesFailed` distinguishes that from a real
|
|
526
|
-
* zero). Pure — exported for deterministic testing. */
|
|
527
|
-
declare function computeCellComposite(input: CellCompositeInput): {
|
|
528
|
-
composite: number;
|
|
529
|
-
codeComposite: number;
|
|
530
|
-
contentComposite: number;
|
|
531
|
-
allJudgesFailed: boolean;
|
|
532
|
-
};
|
|
533
|
-
interface RunMultishotMatrixResult {
|
|
534
|
-
matrix: MatrixResult<CellOutput>;
|
|
535
|
-
}
|
|
536
|
-
declare function runMultishotMatrix<TPersona extends MultishotPersona>(opts: RunMultishotMatrixOptions<TPersona>): Promise<RunMultishotMatrixResult>;
|
|
537
|
-
|
|
538
|
-
interface RunMultishotOptions<TPersona extends MultishotPersona> {
|
|
539
|
-
profile: AgentProfile;
|
|
540
|
-
persona: TPersona;
|
|
541
|
-
shape: MultishotShape<TPersona>;
|
|
542
|
-
/** Tool definitions advertised to the agent. Defaults to delegate_research + delegate_code. */
|
|
543
|
-
tools?: MultishotToolDefinition[];
|
|
544
|
-
/** Map from tool name → executor invoked inline when the agent emits a tool_call. */
|
|
545
|
-
toolExecutors?: Record<string, MultishotToolExecutor>;
|
|
546
|
-
/** Map from tool name → artifact type label written into MultishotArtifact.type.
|
|
547
|
-
* Tools without a mapping still execute, but their results aren't surfaced as
|
|
548
|
-
* typed artifacts (only as tool messages in the transcript). */
|
|
549
|
-
artifactTypeFor?: (toolName: string) => string | undefined;
|
|
550
|
-
maxTurns?: number;
|
|
551
|
-
agentModel?: string;
|
|
552
|
-
driverModel?: string;
|
|
553
|
-
/** Fallback driver models tried when the primary simulated-user model returns empty twice. */
|
|
554
|
-
driverFallbackModels?: string[];
|
|
555
|
-
/** Maximum output tokens for the first agent call in each assistant turn. */
|
|
556
|
-
agentMaxTokens?: number;
|
|
557
|
-
/** Maximum output tokens for agent follow-up calls after tool results. */
|
|
558
|
-
toolFollowupMaxTokens?: number;
|
|
559
|
-
/** Maximum output tokens for each simulated-user driver response. */
|
|
560
|
-
driverMaxTokens?: number;
|
|
561
|
-
/** Maximum tool calls the agent may dispatch inside one assistant turn. */
|
|
562
|
-
maxToolDispatches?: number;
|
|
563
|
-
/** Execution seam for the agent leg. When provided, every agent inference
|
|
564
|
-
* step goes through this function instead of the router HTTP call; the
|
|
565
|
-
* string levers (agentModel, apiKey, baseUrl) stop applying to that leg.
|
|
566
|
-
* apiKey/baseUrl are still resolved for tool executors and any leg
|
|
567
|
-
* without an injected transport. */
|
|
568
|
-
agentTransport?: MultishotTransport;
|
|
569
|
-
/** Execution seam for the simulated-user driver leg (symmetric to
|
|
570
|
-
* agentTransport). Driver model fallback rotation still applies — the
|
|
571
|
-
* transport receives each candidate model in turn. */
|
|
572
|
-
driverTransport?: MultishotTransport;
|
|
573
|
-
apiKey?: string;
|
|
574
|
-
baseUrl?: string;
|
|
575
|
-
signal?: AbortSignal;
|
|
576
|
-
}
|
|
577
|
-
declare function runMultishot<TPersona extends MultishotPersona>(opts: RunMultishotOptions<TPersona>): Promise<MultishotResult>;
|
|
578
|
-
|
|
579
|
-
export { type ArtifactJudgeInput, type CellCompositeInput, type CellCompositeScore, type ConversationJudgeInput, DEFAULT_CODER_MODEL, DEFAULT_DELEGATE_CODE_TOOL, DEFAULT_DELEGATE_RESEARCH_TOOL, DEFAULT_JUDGE_MODEL, DEFAULT_RESEARCHER_MODEL, type DefaultCoderConfig, type DefaultResearcherConfig, type DefaultToolsBundle, type DefaultToolsConfig, type JudgeConfig, type JudgeDimension, type JudgeScore, type MultishotArtifact, MultishotDriverEmptyError, MultishotFatalToolError, type MultishotJudges, type MultishotMessage, type MultishotPersona, type MultishotResult, type MultishotShape, type MultishotToolDefinition, type MultishotToolExecutor, type MultishotTransport, type MultishotTransportRequest, type MultishotTransportResponse, type MultishotTransportToolCall, type RouterCompletionRequest, type RouterCompletionResponse, type RouterToolCall, type RunMultishotMatrixOptions, type RunMultishotMatrixResult, type RunMultishotOptions, computeCellComposite, createCodeExecutor, createResearchExecutor, defaultDelegationTools, defaultRouterBaseUrl, estimateRouterCost, renderDimensions, renderJsonFooter, requireRouterApiKey, routerCompletion, runJudge, runMultishot, runMultishotMatrix };
|