@tangle-network/agent-eval 0.128.1 → 0.129.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +271 -0
- package/README.md +18 -0
- package/dist/analyst/index.d.ts +107 -165
- package/dist/analyst/index.js +5 -9
- package/dist/analyst/index.js.map +1 -1
- package/dist/belief-state/index.d.ts +2 -19
- package/dist/belief-state/index.js +30 -31
- package/dist/belief-state/index.js.map +1 -1
- package/dist/benchmarks/index.d.ts +5 -8
- package/dist/benchmarks/index.js +12 -11
- package/dist/builder-eval/index.js +1 -1
- package/dist/campaign/index.d.ts +30 -39
- package/dist/campaign/index.js +11 -10
- package/dist/{chunk-NKAGIDE2.js → chunk-2QU3YOPR.js} +15 -274
- package/dist/chunk-2QU3YOPR.js.map +1 -0
- package/dist/{chunk-EJGRPCO3.js → chunk-3OCR4R5I.js} +245 -134
- package/dist/chunk-3OCR4R5I.js.map +1 -0
- package/dist/{chunk-2JX3CFMB.js → chunk-56TAVBOK.js} +5 -2
- package/dist/chunk-56TAVBOK.js.map +1 -0
- package/dist/{chunk-DPUHNQLN.js → chunk-7FO3TNPI.js} +2 -2
- package/dist/{chunk-DJKY2TSY.js → chunk-BSO5JDQH.js} +27 -120
- package/dist/chunk-BSO5JDQH.js.map +1 -0
- package/dist/{chunk-EZJEIH2R.js → chunk-C6LXANRU.js} +11 -20
- package/dist/chunk-C6LXANRU.js.map +1 -0
- package/dist/{chunk-ZUUWPZCV.js → chunk-DODXQREJ.js} +4 -4
- package/dist/{chunk-2MKQIFS4.js → chunk-E7QXT7SX.js} +2 -2
- package/dist/{chunk-NYLOYM6N.js → chunk-EG66UGL4.js} +37 -28
- package/dist/chunk-EG66UGL4.js.map +1 -0
- package/dist/{chunk-P5W7RQKK.js → chunk-FXTVJPYD.js} +2 -2
- package/dist/{chunk-VZSRQ272.js → chunk-G7MGMCZD.js} +6 -2
- package/dist/chunk-G7MGMCZD.js.map +1 -0
- package/dist/{chunk-IHQDPH7D.js → chunk-H23X7XKK.js} +85 -75
- package/dist/chunk-H23X7XKK.js.map +1 -0
- package/dist/{chunk-VBQ3CRKH.js → chunk-HPWUNB47.js} +4 -6
- package/dist/chunk-HPWUNB47.js.map +1 -0
- package/dist/{chunk-XPRT64IE.js → chunk-IYCLP2N2.js} +3 -3
- package/dist/chunk-IYCLP2N2.js.map +1 -0
- package/dist/{chunk-XDWDC2MP.js → chunk-JQSF5DQT.js} +11 -5
- package/dist/chunk-JQSF5DQT.js.map +1 -0
- package/dist/{chunk-NACAGYSY.js → chunk-M4YBQKIJ.js} +11 -11
- package/dist/chunk-M4YBQKIJ.js.map +1 -0
- package/dist/{chunk-YJBNWCAA.js → chunk-NY44NC4A.js} +3 -3
- package/dist/chunk-OIUOT4QD.js +44 -0
- package/dist/chunk-OIUOT4QD.js.map +1 -0
- package/dist/chunk-OWN5NPMC.js +152 -0
- package/dist/chunk-OWN5NPMC.js.map +1 -0
- package/dist/chunk-PC5DOSM7.js +579 -0
- package/dist/chunk-PC5DOSM7.js.map +1 -0
- package/dist/{chunk-UB2LOJ6Q.js → chunk-QB6BDBP2.js} +23 -20
- package/dist/chunk-QB6BDBP2.js.map +1 -0
- package/dist/chunk-RXHCETDZ.js +536 -0
- package/dist/chunk-RXHCETDZ.js.map +1 -0
- package/dist/{chunk-PBE2LOSS.js → chunk-SFLLL76A.js} +7 -7
- package/dist/chunk-SFLLL76A.js.map +1 -0
- package/dist/{chunk-VGRCHJON.js → chunk-T6RLYGAD.js} +3 -8
- package/dist/chunk-T6RLYGAD.js.map +1 -0
- package/dist/{chunk-VLOATJQ2.js → chunk-TJVT4QFF.js} +21 -18
- package/dist/chunk-TJVT4QFF.js.map +1 -0
- package/dist/{chunk-S5YLIBFX.js → chunk-TQ7LNKZ3.js} +2 -2
- package/dist/{chunk-EOSZT7PL.js → chunk-U4L7JRPZ.js} +2 -297
- package/dist/chunk-U4L7JRPZ.js.map +1 -0
- package/dist/chunk-U4PHLT2N.js +419 -0
- package/dist/chunk-U4PHLT2N.js.map +1 -0
- package/dist/{chunk-WS3NZZQQ.js → chunk-VCZ5FQYW.js} +3 -4
- package/dist/chunk-VCZ5FQYW.js.map +1 -0
- package/dist/{chunk-BYT7ELPS.js → chunk-WVATSFCP.js} +2 -2
- package/dist/{chunk-TSN7JT6D.js → chunk-X4YIBDER.js} +21 -5
- package/dist/{chunk-TSN7JT6D.js.map → chunk-X4YIBDER.js.map} +1 -1
- package/dist/{chunk-TBL77AUT.js → chunk-YQN4ICPP.js} +5 -5
- package/dist/{chunk-MHELPNRP.js → chunk-ZHTZ4EYI.js} +1 -1
- package/dist/chunk-ZHTZ4EYI.js.map +1 -0
- package/dist/cli.js +6 -5
- package/dist/cli.js.map +1 -1
- package/dist/contract/index.d.ts +47 -87
- package/dist/contract/index.js +14 -13
- package/dist/contract/index.js.map +1 -1
- package/dist/control.js +3 -2
- package/dist/fuzz.js +3 -2
- package/dist/fuzz.js.map +1 -1
- package/dist/index.d.ts +659 -203
- package/dist/index.js +145 -117
- package/dist/index.js.map +1 -1
- package/dist/meta-eval/index.js +2 -2
- package/dist/multishot/index.d.ts +3 -4
- package/dist/multishot/index.js.map +1 -1
- package/dist/openapi.json +1 -1
- package/dist/pipelines/index.js +5 -5
- package/dist/reporting.d.ts +14 -0
- package/dist/reporting.js +7 -6
- package/dist/rl.d.ts +652 -82
- package/dist/rl.js +415 -171
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +1071 -32
- package/dist/rollout/index.js +68 -10
- package/dist/run-campaign-OJJ7CZF4.js +18 -0
- package/dist/supervisor-run/index.d.ts +114 -4
- package/dist/supervisor-run/index.js +4 -3
- package/dist/traces.d.ts +1 -1
- package/dist/traces.js +6 -5
- package/dist/wire/index.d.ts +10 -11
- package/dist/wire/index.js +3 -3
- package/docs/feature-guide.md +1 -1
- package/docs/rollout.md +116 -2
- package/package.json +4 -4
- package/dist/chunk-2JX3CFMB.js.map +0 -1
- package/dist/chunk-DJKY2TSY.js.map +0 -1
- package/dist/chunk-EJGRPCO3.js.map +0 -1
- package/dist/chunk-EOSZT7PL.js.map +0 -1
- package/dist/chunk-EZJEIH2R.js.map +0 -1
- package/dist/chunk-IHQDPH7D.js.map +0 -1
- package/dist/chunk-MHELPNRP.js.map +0 -1
- package/dist/chunk-NACAGYSY.js.map +0 -1
- package/dist/chunk-NKAGIDE2.js.map +0 -1
- package/dist/chunk-NYLOYM6N.js.map +0 -1
- package/dist/chunk-PBE2LOSS.js.map +0 -1
- package/dist/chunk-TT4KNT67.js +0 -124
- package/dist/chunk-TT4KNT67.js.map +0 -1
- package/dist/chunk-UB2LOJ6Q.js.map +0 -1
- package/dist/chunk-UWZZKKU7.js +0 -237
- package/dist/chunk-UWZZKKU7.js.map +0 -1
- package/dist/chunk-VBQ3CRKH.js.map +0 -1
- package/dist/chunk-VGRCHJON.js.map +0 -1
- package/dist/chunk-VLOATJQ2.js.map +0 -1
- package/dist/chunk-VZSRQ272.js.map +0 -1
- package/dist/chunk-WS3NZZQQ.js.map +0 -1
- package/dist/chunk-XDWDC2MP.js.map +0 -1
- package/dist/chunk-XPRT64IE.js.map +0 -1
- package/dist/run-campaign-ISHFZ7FJ.js +0 -17
- /package/dist/{chunk-DPUHNQLN.js.map → chunk-7FO3TNPI.js.map} +0 -0
- /package/dist/{chunk-ZUUWPZCV.js.map → chunk-DODXQREJ.js.map} +0 -0
- /package/dist/{chunk-2MKQIFS4.js.map → chunk-E7QXT7SX.js.map} +0 -0
- /package/dist/{chunk-P5W7RQKK.js.map → chunk-FXTVJPYD.js.map} +0 -0
- /package/dist/{chunk-YJBNWCAA.js.map → chunk-NY44NC4A.js.map} +0 -0
- /package/dist/{chunk-S5YLIBFX.js.map → chunk-TQ7LNKZ3.js.map} +0 -0
- /package/dist/{chunk-BYT7ELPS.js.map → chunk-WVATSFCP.js.map} +0 -0
- /package/dist/{chunk-TBL77AUT.js.map → chunk-YQN4ICPP.js.map} +0 -0
- /package/dist/{run-campaign-ISHFZ7FJ.js.map → run-campaign-OJJ7CZF4.js.map} +0 -0
package/dist/analyst/index.d.ts
CHANGED
|
@@ -1,4 +1,3 @@
|
|
|
1
|
-
import { TCloud } from '@tangle-network/tcloud';
|
|
2
1
|
import { AxAIArgs, AxAIService, AxFunction } from '@ax-llm/ax';
|
|
3
2
|
import { z } from 'zod';
|
|
4
3
|
|
|
@@ -983,9 +982,8 @@ type CostLedgerHandle = Pick<CostLedger, Exclude<keyof CostLedger, 'listPending'
|
|
|
983
982
|
* { baseUrl: 'https://router.tangle.tools/v1', apiKey: process.env.KEY },
|
|
984
983
|
* )
|
|
985
984
|
*
|
|
986
|
-
*
|
|
987
|
-
*
|
|
988
|
-
* that need free-form text use `callLlm` and parse output themselves.
|
|
985
|
+
* `createChatClient` wraps this implementation for provider-neutral package
|
|
986
|
+
* entry points. Direct callers can use `callLlm` or `callLlmJson`.
|
|
989
987
|
*/
|
|
990
988
|
|
|
991
989
|
interface LlmMessage {
|
|
@@ -1096,8 +1094,8 @@ interface LlmClientOptions {
|
|
|
1096
1094
|
* total attempts × `timeoutMs`.
|
|
1097
1095
|
*/
|
|
1098
1096
|
deadlineMs?: number;
|
|
1099
|
-
/** Total provider attempts.
|
|
1100
|
-
|
|
1097
|
+
/** Total provider attempts. Default 3. */
|
|
1098
|
+
maximumAttempts?: number;
|
|
1101
1099
|
/** Token rates used when the provider omits cost or package pricing does not cover the model. */
|
|
1102
1100
|
customTokenPricing?: CustomTokenPricing;
|
|
1103
1101
|
/**
|
|
@@ -1243,6 +1241,91 @@ interface SemanticConceptJudgeOptions {
|
|
|
1243
1241
|
complexityWeights?: Partial<Record<ConceptComplexity, number>>;
|
|
1244
1242
|
}
|
|
1245
1243
|
|
|
1244
|
+
/**
|
|
1245
|
+
* Provider-neutral chat contract for every model call made by agent-eval.
|
|
1246
|
+
*
|
|
1247
|
+
* Callers choose the transport at the package boundary with `createChatClient`.
|
|
1248
|
+
* Evaluation code receives canonical requests and results without importing a
|
|
1249
|
+
* provider SDK.
|
|
1250
|
+
*/
|
|
1251
|
+
|
|
1252
|
+
/**
|
|
1253
|
+
* Unified chat interface using the package's canonical LLM request and result.
|
|
1254
|
+
*/
|
|
1255
|
+
interface ChatClient {
|
|
1256
|
+
/** Display name of the bound transport, included in telemetry. */
|
|
1257
|
+
readonly transport: ChatTransport;
|
|
1258
|
+
/** Default model when the caller omits one. */
|
|
1259
|
+
readonly defaultModel?: string;
|
|
1260
|
+
/** Total provider attempts this transport can make for one chat call. */
|
|
1261
|
+
readonly maximumAttempts?: number;
|
|
1262
|
+
/** Implementations must enforce `req.maxTokens` when it is present. */
|
|
1263
|
+
chat(req: ChatRequest, opts?: ChatCallOpts): Promise<ChatResponse>;
|
|
1264
|
+
}
|
|
1265
|
+
type ChatTransport = 'router' | 'sandbox-sdk' | 'cli-bridge' | 'direct-provider' | 'custom' | 'mock';
|
|
1266
|
+
interface ChatRequest extends Omit<LlmCallRequest, 'model'> {
|
|
1267
|
+
/** Optional — falls back to ChatClient.defaultModel. */
|
|
1268
|
+
model?: string;
|
|
1269
|
+
}
|
|
1270
|
+
type ChatResponse = LlmCallResult;
|
|
1271
|
+
interface ChatCallOpts {
|
|
1272
|
+
/** Cancel the in-flight request. */
|
|
1273
|
+
signal?: AbortSignal;
|
|
1274
|
+
/** Hard USD ceiling for this single call (informational; the underlying transport may not enforce). */
|
|
1275
|
+
maxCostUsd?: number;
|
|
1276
|
+
/** Correlation tag carried into request headers when the transport allows. */
|
|
1277
|
+
correlationId?: string;
|
|
1278
|
+
/** Stable provider idempotency key for retries/redrives of one paid call. */
|
|
1279
|
+
idempotencyKey?: string;
|
|
1280
|
+
}
|
|
1281
|
+
type CreateChatClientOpts = RouterTransportOpts | CliBridgeTransportOpts | DirectProviderTransportOpts | SandboxSdkTransportOpts | CustomTransportOpts | MockTransportOpts;
|
|
1282
|
+
interface BaseTransportOpts {
|
|
1283
|
+
defaultModel?: string;
|
|
1284
|
+
/** Total provider attempts. Required for opaque transports used in capped runs. */
|
|
1285
|
+
maximumAttempts?: number;
|
|
1286
|
+
}
|
|
1287
|
+
interface RouterTransportOpts extends BaseTransportOpts {
|
|
1288
|
+
transport: 'router';
|
|
1289
|
+
baseUrl?: string;
|
|
1290
|
+
apiKey: string;
|
|
1291
|
+
}
|
|
1292
|
+
interface CliBridgeTransportOpts extends BaseTransportOpts {
|
|
1293
|
+
transport: 'cli-bridge';
|
|
1294
|
+
baseUrl?: string;
|
|
1295
|
+
bearer?: string;
|
|
1296
|
+
}
|
|
1297
|
+
interface DirectProviderTransportOpts extends BaseTransportOpts {
|
|
1298
|
+
transport: 'direct-provider';
|
|
1299
|
+
baseUrl: string;
|
|
1300
|
+
apiKey: string;
|
|
1301
|
+
}
|
|
1302
|
+
/**
|
|
1303
|
+
* Sandbox-SDK transport. The caller supplies a canonical chat function for an
|
|
1304
|
+
* already-configured Sandbox handle, so agent-eval does not import the SDK.
|
|
1305
|
+
*/
|
|
1306
|
+
interface SandboxSdkTransportOpts extends BaseTransportOpts {
|
|
1307
|
+
transport: 'sandbox-sdk';
|
|
1308
|
+
chat: (req: ChatRequest, opts?: ChatCallOpts) => Promise<ChatResponse>;
|
|
1309
|
+
}
|
|
1310
|
+
/** Caller-adapted SDK or transport returning the canonical ChatResponse shape. */
|
|
1311
|
+
interface CustomTransportOpts extends BaseTransportOpts {
|
|
1312
|
+
transport: 'custom';
|
|
1313
|
+
chat: (req: ChatRequest, opts?: ChatCallOpts) => Promise<ChatResponse>;
|
|
1314
|
+
}
|
|
1315
|
+
/**
|
|
1316
|
+
* Mock transport for tests. The handler receives the request and returns
|
|
1317
|
+
* whatever the test wants. No retries, no JSON-schema degrade.
|
|
1318
|
+
*/
|
|
1319
|
+
interface MockTransportOpts extends BaseTransportOpts {
|
|
1320
|
+
transport: 'mock';
|
|
1321
|
+
handler: (req: ChatRequest, opts?: ChatCallOpts) => Promise<ChatResponse>;
|
|
1322
|
+
}
|
|
1323
|
+
/**
|
|
1324
|
+
* Build a ChatClient bound to a specific transport. The returned client
|
|
1325
|
+
* is safe to share across analysts in a single registry run.
|
|
1326
|
+
*/
|
|
1327
|
+
declare function createChatClient(opts: CreateChatClientOpts): ChatClient;
|
|
1328
|
+
|
|
1246
1329
|
interface Scenario {
|
|
1247
1330
|
id: string;
|
|
1248
1331
|
persona: string;
|
|
@@ -1309,10 +1392,8 @@ interface JudgeInput {
|
|
|
1309
1392
|
costPhase?: string;
|
|
1310
1393
|
costTags?: Record<string, string>;
|
|
1311
1394
|
signal?: AbortSignal;
|
|
1312
|
-
/** Exact maximum provider attempts configured on the supplied TCloud client. */
|
|
1313
|
-
tcloudMaximumAttempts?: number;
|
|
1314
1395
|
}
|
|
1315
|
-
type JudgeFn = (
|
|
1396
|
+
type JudgeFn = (chat: ChatClient, input: JudgeInput) => Promise<JudgeScore[]>;
|
|
1316
1397
|
|
|
1317
1398
|
/**
|
|
1318
1399
|
* Shared types for the trace-analyst module.
|
|
@@ -1542,103 +1623,6 @@ interface TraceAnalysisStore {
|
|
|
1542
1623
|
}): Promise<SearchSpanResult>;
|
|
1543
1624
|
}
|
|
1544
1625
|
|
|
1545
|
-
/**
|
|
1546
|
-
* ChatClient — the single LLM abstraction analysts call.
|
|
1547
|
-
*
|
|
1548
|
-
* agent-eval already ships an `LlmClient` (OpenAI-compatible, retry,
|
|
1549
|
-
* graceful JSON-schema degrade) and judges that talk to `TCloud`. Two
|
|
1550
|
-
* mixed patterns force every analyst author to pick a transport, which
|
|
1551
|
-
* couples analyst code to runtime concerns (cli-bridge vs router vs
|
|
1552
|
-
* sandbox-sdk) it shouldn't know about.
|
|
1553
|
-
*
|
|
1554
|
-
* `ChatClient` is one interface every analyst takes via `AnalystContext.chat`.
|
|
1555
|
-
* The operator decides at the registry boundary which transport binds
|
|
1556
|
-
* to it. Analyst code stays transport-agnostic; swapping production
|
|
1557
|
-
* (sandbox-sdk) for local dev (cli-bridge) or tests (mock) is a one-
|
|
1558
|
-
* line factory call.
|
|
1559
|
-
*
|
|
1560
|
-
* Designed to coexist: existing `LlmClient` callers and existing
|
|
1561
|
-
* `TCloud`-based judges keep working untouched. New analyst code uses
|
|
1562
|
-
* `ChatClient`. When old call sites migrate, they pick up budgeting,
|
|
1563
|
-
* cancellation, and unified telemetry for free.
|
|
1564
|
-
*/
|
|
1565
|
-
|
|
1566
|
-
/**
|
|
1567
|
-
* Unified chat interface. Mirrors LlmCallRequest/Result so the OpenAI-
|
|
1568
|
-
* compatible mental model stays. Two methods: a one-shot `chat()` and
|
|
1569
|
-
* an `streamChat()` for future agentic loops (not yet exposed).
|
|
1570
|
-
*/
|
|
1571
|
-
interface ChatClient {
|
|
1572
|
-
/** Display name of the bound transport — included in telemetry. */
|
|
1573
|
-
readonly transport: ChatTransport;
|
|
1574
|
-
/** Default model when caller omits — operators bind this per environment. */
|
|
1575
|
-
readonly defaultModel?: string;
|
|
1576
|
-
/** Total provider attempts this transport can make for one chat call. */
|
|
1577
|
-
readonly maximumAttempts?: number;
|
|
1578
|
-
/** Implementations must enforce `req.maxTokens` when it is present. */
|
|
1579
|
-
chat(req: ChatRequest, opts?: ChatCallOpts): Promise<ChatResponse>;
|
|
1580
|
-
}
|
|
1581
|
-
type ChatTransport = 'router' | 'sandbox-sdk' | 'cli-bridge' | 'direct-provider' | 'mock';
|
|
1582
|
-
interface ChatRequest extends Omit<LlmCallRequest, 'model'> {
|
|
1583
|
-
/** Optional — falls back to ChatClient.defaultModel. */
|
|
1584
|
-
model?: string;
|
|
1585
|
-
}
|
|
1586
|
-
type ChatResponse = LlmCallResult;
|
|
1587
|
-
interface ChatCallOpts {
|
|
1588
|
-
/** Cancel the in-flight request. */
|
|
1589
|
-
signal?: AbortSignal;
|
|
1590
|
-
/** Hard USD ceiling for this single call (informational; the underlying transport may not enforce). */
|
|
1591
|
-
maxCostUsd?: number;
|
|
1592
|
-
/** Correlation tag carried into request headers when the transport allows. */
|
|
1593
|
-
correlationId?: string;
|
|
1594
|
-
/** Stable provider idempotency key for retries/redrives of one paid call. */
|
|
1595
|
-
idempotencyKey?: string;
|
|
1596
|
-
}
|
|
1597
|
-
type CreateChatClientOpts = RouterTransportOpts | CliBridgeTransportOpts | DirectProviderTransportOpts | SandboxSdkTransportOpts | MockTransportOpts;
|
|
1598
|
-
interface BaseTransportOpts {
|
|
1599
|
-
defaultModel?: string;
|
|
1600
|
-
/** Total provider attempts. Required for opaque transports used in capped runs. */
|
|
1601
|
-
maximumAttempts?: number;
|
|
1602
|
-
}
|
|
1603
|
-
interface RouterTransportOpts extends BaseTransportOpts {
|
|
1604
|
-
transport: 'router';
|
|
1605
|
-
baseUrl?: string;
|
|
1606
|
-
apiKey: string;
|
|
1607
|
-
}
|
|
1608
|
-
interface CliBridgeTransportOpts extends BaseTransportOpts {
|
|
1609
|
-
transport: 'cli-bridge';
|
|
1610
|
-
baseUrl?: string;
|
|
1611
|
-
bearer?: string;
|
|
1612
|
-
}
|
|
1613
|
-
interface DirectProviderTransportOpts extends BaseTransportOpts {
|
|
1614
|
-
transport: 'direct-provider';
|
|
1615
|
-
baseUrl: string;
|
|
1616
|
-
apiKey: string;
|
|
1617
|
-
}
|
|
1618
|
-
/**
|
|
1619
|
-
* Sandbox-SDK transport. Provided as a thin pass-through: the caller
|
|
1620
|
-
* supplies a callable that mimics LlmClient.chat() against an already-
|
|
1621
|
-
* configured Sandbox handle. We don't import the SDK here to keep
|
|
1622
|
-
* agent-eval dep-free of @tangle-network/sandbox.
|
|
1623
|
-
*/
|
|
1624
|
-
interface SandboxSdkTransportOpts extends BaseTransportOpts {
|
|
1625
|
-
transport: 'sandbox-sdk';
|
|
1626
|
-
chat: (req: ChatRequest, opts?: ChatCallOpts) => Promise<ChatResponse>;
|
|
1627
|
-
}
|
|
1628
|
-
/**
|
|
1629
|
-
* Mock transport for tests. The handler receives the request and returns
|
|
1630
|
-
* whatever the test wants. No retries, no JSON-schema degrade.
|
|
1631
|
-
*/
|
|
1632
|
-
interface MockTransportOpts extends BaseTransportOpts {
|
|
1633
|
-
transport: 'mock';
|
|
1634
|
-
handler: (req: ChatRequest, opts?: ChatCallOpts) => Promise<ChatResponse>;
|
|
1635
|
-
}
|
|
1636
|
-
/**
|
|
1637
|
-
* Build a ChatClient bound to a specific transport. The returned client
|
|
1638
|
-
* is safe to share across analysts in a single registry run.
|
|
1639
|
-
*/
|
|
1640
|
-
declare function createChatClient(opts: CreateChatClientOpts): ChatClient;
|
|
1641
|
-
|
|
1642
1626
|
/**
|
|
1643
1627
|
* Analyst contract — the missing orchestration layer over agent-eval's
|
|
1644
1628
|
* existing analyzers (analyzeTraces, MultiLayerVerifier, RunCritic,
|
|
@@ -1859,13 +1843,8 @@ interface AnalystRunSummary {
|
|
|
1859
1843
|
reason?: string;
|
|
1860
1844
|
findings_count: number;
|
|
1861
1845
|
latency_ms: number;
|
|
1862
|
-
|
|
1863
|
-
|
|
1864
|
-
* Additive receipt for model usage. Registry-produced summaries populate it
|
|
1865
|
-
* even when the analyst emits no findings. `cost_usd` remains the legacy
|
|
1866
|
-
* numeric field; inspect `usage.cost` before treating zero as observed.
|
|
1867
|
-
*/
|
|
1868
|
-
usage?: AnalystUsageReceipt;
|
|
1846
|
+
/** Additive model usage and cost provenance for this analyst. */
|
|
1847
|
+
usage: AnalystUsageReceipt;
|
|
1869
1848
|
/** When `status='failed'`: the error class + message, never the full stack. */
|
|
1870
1849
|
error?: {
|
|
1871
1850
|
class: string;
|
|
@@ -1967,8 +1946,8 @@ interface JudgeAdapterOpts {
|
|
|
1967
1946
|
id?: string;
|
|
1968
1947
|
area?: string;
|
|
1969
1948
|
judge: JudgeFn;
|
|
1970
|
-
/**
|
|
1971
|
-
|
|
1949
|
+
/** Chat client passed to the JudgeFn. */
|
|
1950
|
+
chat: ChatClient;
|
|
1972
1951
|
/** Optional cost classification — most judges call an LLM. */
|
|
1973
1952
|
cost?: Analyst['cost'];
|
|
1974
1953
|
/** Optional threshold below which a JudgeScore becomes a finding. Default 6 (on 0-10 scale). */
|
|
@@ -2088,12 +2067,9 @@ declare function behavioralAnalyst(): Analyst<TraceAnalysisStore>;
|
|
|
2088
2067
|
/**
|
|
2089
2068
|
* Typed Ax output for analyst findings.
|
|
2090
2069
|
*
|
|
2091
|
-
*
|
|
2092
|
-
*
|
|
2093
|
-
*
|
|
2094
|
-
* native structured output; at the kind-factory boundary we Zod-validate
|
|
2095
|
-
* each emitted finding so malformed rows fail loud instead of being
|
|
2096
|
-
* silently lifted with default severity.
|
|
2070
|
+
* Ax binds the field as `findings:json[]` so the provider emits native
|
|
2071
|
+
* structured output. At the kind-factory boundary every row is validated
|
|
2072
|
+
* before it becomes an `AnalystFinding`.
|
|
2097
2073
|
*
|
|
2098
2074
|
* Why not `f.object().array()` directly in the signature? The Ax
|
|
2099
2075
|
* signature string `question:string -> findings:json[]` already lets
|
|
@@ -2108,40 +2084,16 @@ declare const RawAnalystEvidenceSchema: z.ZodObject<{
|
|
|
2108
2084
|
excerpt: z.ZodOptional<z.ZodString>;
|
|
2109
2085
|
}, z.core.$strict>;
|
|
2110
2086
|
type RawAnalystEvidence = z.infer<typeof RawAnalystEvidenceSchema>;
|
|
2111
|
-
/** Original public schema retained for stored rows and callback contracts. */
|
|
2112
2087
|
declare const RawAnalystFindingSchema: z.ZodObject<{
|
|
2113
|
-
evidence_uri: z.ZodString;
|
|
2114
|
-
evidence_excerpt: z.ZodOptional<z.ZodString>;
|
|
2115
|
-
severity: z.ZodEnum<{
|
|
2116
|
-
critical: "critical";
|
|
2117
|
-
info: "info";
|
|
2118
|
-
low: "low";
|
|
2119
|
-
high: "high";
|
|
2120
|
-
medium: "medium";
|
|
2121
|
-
}>;
|
|
2122
|
-
claim: z.ZodString;
|
|
2123
|
-
subject: z.ZodOptional<z.ZodString>;
|
|
2124
|
-
confidence: z.ZodNumber;
|
|
2125
|
-
rationale: z.ZodOptional<z.ZodString>;
|
|
2126
|
-
recommended_action: z.ZodOptional<z.ZodString>;
|
|
2127
|
-
}, z.core.$strict>;
|
|
2128
|
-
type RawAnalystFinding = z.infer<typeof RawAnalystFindingSchema>;
|
|
2129
|
-
/**
|
|
2130
|
-
* Canonical plural-evidence contract. The preprocessor accepts the original
|
|
2131
|
-
* `evidence_uri` / `evidence_excerpt` pair and normalizes it into one evidence
|
|
2132
|
-
* item so persisted rows and older model fixtures remain readable. New output
|
|
2133
|
-
* always receives the plural shape.
|
|
2134
|
-
*/
|
|
2135
|
-
declare const CanonicalRawAnalystFindingSchema: z.ZodPreprocess<z.ZodObject<{
|
|
2136
2088
|
evidence: z.ZodArray<z.ZodObject<{
|
|
2137
2089
|
uri: z.ZodString;
|
|
2138
2090
|
excerpt: z.ZodOptional<z.ZodString>;
|
|
2139
2091
|
}, z.core.$strict>>;
|
|
2140
2092
|
severity: z.ZodEnum<{
|
|
2141
|
-
critical: "critical";
|
|
2142
|
-
info: "info";
|
|
2143
2093
|
low: "low";
|
|
2144
2094
|
high: "high";
|
|
2095
|
+
critical: "critical";
|
|
2096
|
+
info: "info";
|
|
2145
2097
|
medium: "medium";
|
|
2146
2098
|
}>;
|
|
2147
2099
|
claim: z.ZodString;
|
|
@@ -2149,25 +2101,17 @@ declare const CanonicalRawAnalystFindingSchema: z.ZodPreprocess<z.ZodObject<{
|
|
|
2149
2101
|
confidence: z.ZodNumber;
|
|
2150
2102
|
rationale: z.ZodOptional<z.ZodString>;
|
|
2151
2103
|
recommended_action: z.ZodOptional<z.ZodString>;
|
|
2152
|
-
}, z.core.$strict
|
|
2153
|
-
type
|
|
2104
|
+
}, z.core.$strict>;
|
|
2105
|
+
type RawAnalystFinding = z.infer<typeof RawAnalystFindingSchema>;
|
|
2154
2106
|
/**
|
|
2155
2107
|
* Description embedded into the actor prompt so the LLM knows what
|
|
2156
2108
|
* shape to emit. Kept here so kinds share one source of truth rather
|
|
2157
2109
|
* than restating the schema in every prompt.
|
|
2158
2110
|
*/
|
|
2159
2111
|
declare const RAW_FINDING_SCHEMA_PROMPT = "Each finding MUST be a strict JSON object with:\n - severity: \"critical\" | \"high\" | \"medium\" | \"low\" | \"info\"\n - claim: one-sentence statement (max 2000 chars)\n - subject?: one exact subject form listed by this kind; omit rather than guess\n - evidence: REQUIRED non-empty array of {\"uri\": string, \"excerpt\"?: string}. Use real identifiers with span://, event://, artifact://, metric://, or finding://. Include a short exact quote in excerpt when available. If nothing is citable, do not emit the finding.\n - confidence: number 0..1 (0.9+ exact evidence; 0.6-0.8 inferred pattern; <0.5 speculative)\n - rationale?: one or two reasoning sentences\n - recommended_action?: concrete imperative change; omit for descriptive findings\n\nUnknown fields are rejected. Do not emit area; the factory assigns it. Emit [] when there are no findings. Never fabricate evidence.";
|
|
2160
|
-
/** Convert
|
|
2161
|
-
declare function evidenceRefsFromRawFinding(finding:
|
|
2162
|
-
/**
|
|
2163
|
-
* Validate the original singular-evidence shape. This public parser retains
|
|
2164
|
-
* its pre-canonicalization result type so existing callback code and stored
|
|
2165
|
-
* rows continue to receive exactly the object accepted by
|
|
2166
|
-
* {@link RawAnalystFindingSchema}.
|
|
2167
|
-
*/
|
|
2112
|
+
/** Convert raw citations into the public finding evidence envelope. */
|
|
2113
|
+
declare function evidenceRefsFromRawFinding(finding: RawAnalystFinding): EvidenceRef[];
|
|
2168
2114
|
declare function parseRawFinding(row: unknown, log?: (msg: string, fields?: Record<string, unknown>) => void): RawAnalystFinding | null;
|
|
2169
|
-
/** Validate model output and normalize original singular citations. */
|
|
2170
|
-
declare function parseCanonicalRawFinding(row: unknown, log?: (msg: string, fields?: Record<string, unknown>) => void): CanonicalRawAnalystFinding | null;
|
|
2171
2115
|
|
|
2172
2116
|
/**
|
|
2173
2117
|
* Analyst-kind factory — the typed way to define trace analysts.
|
|
@@ -2176,7 +2120,7 @@ declare function parseCanonicalRawFinding(row: unknown, log?: (msg: string, fiel
|
|
|
2176
2120
|
* and bounded Ax subqueries target one failure-mode lens (failure-mode
|
|
2177
2121
|
* classification, knowledge gap discovery, knowledge poisoning,
|
|
2178
2122
|
* self-improvement, ...). Kinds emit findings in the typed
|
|
2179
|
-
* `
|
|
2123
|
+
* `RawAnalystFinding` shape via a JSON-array Ax output; the factory
|
|
2180
2124
|
* validates each row with Zod and lifts it into `AnalystFinding[]`.
|
|
2181
2125
|
*
|
|
2182
2126
|
* Composition rules:
|
|
@@ -2239,7 +2183,7 @@ interface TraceAnalystKindSpec {
|
|
|
2239
2183
|
*/
|
|
2240
2184
|
interface TraceAnalystGolden {
|
|
2241
2185
|
question: string;
|
|
2242
|
-
expected: ReadonlyArray<Omit<
|
|
2186
|
+
expected: ReadonlyArray<Omit<RawAnalystFinding, 'confidence'>>;
|
|
2243
2187
|
}
|
|
2244
2188
|
interface CreateTraceAnalystKindOpts {
|
|
2245
2189
|
/** AxAIService bound at registration time. */
|
|
@@ -2907,7 +2851,7 @@ declare function coerceJson(text: string): unknown;
|
|
|
2907
2851
|
* Coerce arbitrary actor/structurer output into an array of candidate finding
|
|
2908
2852
|
* rows: a JSON string → parse; a single object → 1-element array; an array →
|
|
2909
2853
|
* as-is; anything else → []. Callers still run each row through Zod
|
|
2910
|
-
* (`
|
|
2854
|
+
* (`parseRawFinding`) — this only fixes the shape and never invents fields.
|
|
2911
2855
|
*/
|
|
2912
2856
|
declare function coerceToFindingRows(raw: unknown): unknown[];
|
|
2913
2857
|
|
|
@@ -2980,8 +2924,6 @@ interface StructureFindingsOptions {
|
|
|
2980
2924
|
maxReasks?: number;
|
|
2981
2925
|
/** Apply the caller's normal finding rules before a recovered row is lifted. */
|
|
2982
2926
|
processRow?: (row: RawAnalystFinding) => RawAnalystFinding | null;
|
|
2983
|
-
/** Apply canonical multi-citation rules after any original callback. */
|
|
2984
|
-
processCanonicalRow?: (row: CanonicalRawAnalystFinding) => CanonicalRawAnalystFinding | null;
|
|
2985
2927
|
/** Provenance copied onto every recovered finding. */
|
|
2986
2928
|
findingMetadata?: Record<string, unknown>;
|
|
2987
2929
|
/** Test seam: inject a fetch (no network in unit tests). */
|
|
@@ -3026,4 +2968,4 @@ type TraceToolGroupName =
|
|
|
3026
2968
|
*/
|
|
3027
2969
|
declare function buildTraceToolsForGroup(group: TraceToolGroupName, store: TraceAnalysisStore): AxFunction[];
|
|
3028
2970
|
|
|
3029
|
-
export { ANALYST_SEVERITIES, type Analyst, type AnalystContext, type AnalystCost, type AnalystFinding, type AnalystHooks, type AnalystInputKind, AnalystRegistry, type AnalystRegistryOptions, type AnalystRequirements, type AnalystRunEvent, type AnalystRunInputs, type AnalystRunResult, type AnalystRunSummary, type AnalystSeverity, type AnalystUsageReceipt, type BudgetPolicy, type
|
|
2971
|
+
export { ANALYST_SEVERITIES, type Analyst, type AnalystContext, type AnalystCost, type AnalystFinding, type AnalystHooks, type AnalystInputKind, AnalystRegistry, type AnalystRegistryOptions, type AnalystRequirements, type AnalystRunEvent, type AnalystRunInputs, type AnalystRunResult, type AnalystRunSummary, type AnalystSeverity, type AnalystUsageReceipt, type BudgetPolicy, type ChatCallOpts, type ChatClient, type ChatRequest, type ChatResponse, type ChatTransport, type CliBridgeTransportOpts, type CreateAnalystAiConfig, type CreateChatClientOpts, type CreateTraceAnalystKindOpts, type CustomTransportOpts, DEFAULT_TRACE_ANALYST_KINDS, type DefaultAnalystRegistryOptions, type DiffPolicy, type DirectProviderTransportOpts, type EvidenceRef, FAILURE_MODE_KIND_SPEC, FINDING_SUBJECT_GRAMMAR_PROMPT, FINDING_SUBJECT_KINDS, FINDING_SUBJECT_SYNTAX, type FindingSubject, type FindingSubjectKind, FindingSubjectStringSchema, type FindingsDiff, FindingsStore, IMPROVEMENT_KIND_SPEC, type JudgeAdapterOpts, KIND_EXPECTED_SUBJECTS, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, type MockTransportOpts, type PersistedFinding, RAW_FINDING_SCHEMA_PROMPT, type RawAnalystEvidence, RawAnalystEvidenceSchema, type RawAnalystFinding, RawAnalystFindingSchema, type RegistryRunOpts, type RouterTransportOpts, type RunCriticAdapterOpts, SKILL_USAGE_ANALYST, type SandboxSdkTransportOpts, type SemanticConceptJudgeAdapterOpts, SkillUsageAnalyst, type SkillUsageRecord, type SkillUsageReport, type SkillUsageScanConfig, type StructureFindingsOptions, type StructureFindingsResult, type TraceAnalystGolden, type TraceAnalystKindSpec, type TraceToolGroupName, type VerifierAdapterOpts, assertNoJudgeVerdict, behavioralAnalyst, buildDefaultAnalystRegistry, buildSkillUsageReport, buildTraceToolsForGroup, coerceJson, coerceToFindingRows, computeFindingId, createAnalystAi, createChatClient, createJudgeAdapter, createRunCriticAdapter, createSemanticConceptJudgeAdapter, createTraceAnalystKind, createVerifierAdapter, defaultIsMaterial, deriveEfficiencyFindings, diffFindings, emitSkillUsageFindings, evidenceRefsFromRawFinding, findingSubjectGrammarPromptFor, isJudgeVerdict, isTraceObservable, liftSeverity, makeFinding, parseFindingSubject, parseRawFinding, renderFindingSubject, renderPriorFindings, renderUpstreamFindings, stripCodeFences, structureFindings };
|
package/dist/analyst/index.js
CHANGED
|
@@ -9,11 +9,10 @@ import {
|
|
|
9
9
|
diffFindings,
|
|
10
10
|
emitSkillUsageFindings,
|
|
11
11
|
runSemanticConceptJudge
|
|
12
|
-
} from "../chunk-
|
|
12
|
+
} from "../chunk-DODXQREJ.js";
|
|
13
13
|
import {
|
|
14
14
|
ANALYST_SEVERITIES,
|
|
15
15
|
AnalystRegistry,
|
|
16
|
-
CanonicalRawAnalystFindingSchema,
|
|
17
16
|
DEFAULT_TRACE_ANALYST_KINDS,
|
|
18
17
|
FAILURE_MODE_KIND_SPEC,
|
|
19
18
|
FINDING_SUBJECT_GRAMMAR_PROMPT,
|
|
@@ -40,7 +39,6 @@ import {
|
|
|
40
39
|
evidenceRefsFromRawFinding,
|
|
41
40
|
findingSubjectGrammarPromptFor,
|
|
42
41
|
makeFinding,
|
|
43
|
-
parseCanonicalRawFinding,
|
|
44
42
|
parseFindingSubject,
|
|
45
43
|
parseRawFinding,
|
|
46
44
|
renderFindingSubject,
|
|
@@ -50,13 +48,13 @@ import {
|
|
|
50
48
|
stripCodeFences,
|
|
51
49
|
structureFindings,
|
|
52
50
|
validateUsageSettlementTimeout
|
|
53
|
-
} from "../chunk-
|
|
51
|
+
} from "../chunk-BSO5JDQH.js";
|
|
54
52
|
import "../chunk-HHWE3POT.js";
|
|
55
53
|
import "../chunk-WGXIEX7P.js";
|
|
56
|
-
import "../chunk-
|
|
54
|
+
import "../chunk-SFLLL76A.js";
|
|
57
55
|
import {
|
|
58
56
|
CostLedger
|
|
59
|
-
} from "../chunk-
|
|
57
|
+
} from "../chunk-VCZ5FQYW.js";
|
|
60
58
|
import "../chunk-VI2UW6B6.js";
|
|
61
59
|
import "../chunk-P6FYH6K4.js";
|
|
62
60
|
import "../chunk-PC4UYEBM.js";
|
|
@@ -206,7 +204,7 @@ function createJudgeAdapter(opts) {
|
|
|
206
204
|
cost: opts.cost ?? { kind: "llm" },
|
|
207
205
|
version: `judge-${ADAPTER_REV}`,
|
|
208
206
|
async analyze(input) {
|
|
209
|
-
const scores = await opts.judge(opts.
|
|
207
|
+
const scores = await opts.judge(opts.chat, input);
|
|
210
208
|
return scores.filter((s) => normalize10(s.score) < threshold).map((s) => liftJudgeScore(id, area, s));
|
|
211
209
|
}
|
|
212
210
|
};
|
|
@@ -334,7 +332,6 @@ function assertNoJudgeVerdict(findings, context = "steer") {
|
|
|
334
332
|
export {
|
|
335
333
|
ANALYST_SEVERITIES,
|
|
336
334
|
AnalystRegistry,
|
|
337
|
-
CanonicalRawAnalystFindingSchema,
|
|
338
335
|
DEFAULT_TRACE_ANALYST_KINDS,
|
|
339
336
|
FAILURE_MODE_KIND_SPEC,
|
|
340
337
|
FINDING_SUBJECT_GRAMMAR_PROMPT,
|
|
@@ -376,7 +373,6 @@ export {
|
|
|
376
373
|
isTraceObservable,
|
|
377
374
|
liftSeverity,
|
|
378
375
|
makeFinding,
|
|
379
|
-
parseCanonicalRawFinding,
|
|
380
376
|
parseFindingSubject,
|
|
381
377
|
parseRawFinding,
|
|
382
378
|
renderFindingSubject,
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"sources":["../../src/analyst/adapters.ts","../../src/analyst/steer-firewall.ts"],"sourcesContent":["/**\n * Adapter factories — lift each existing agent-eval primitive into the\n * Analyst contract without re-implementing it.\n *\n * Five primitives, five factories. Each one:\n * - Builds an Analyst with a stable id (caller chooses; defaults\n * given), a sensible default `inputKind`, a version derived from\n * the wrapped primitive's version + an adapter revision, and an\n * `analyze()` that calls the primitive and lifts its output to\n * AnalystFinding[] using `makeFinding()`.\n * - Maps severities: the existing `Severity` ('critical' | 'major' |\n * 'minor' | 'info') projects onto AnalystSeverity ('critical' |\n * 'high' | 'medium' | 'low' | 'info'); 'major' → 'high', 'minor' →\n * 'medium'. Domain analysts that want finer-grained mapping override.\n *\n * Adapters never own state. Calling the same factory twice with the\n * same primitive instance is safe.\n */\n\nimport { CostLedger } from '../cost-ledger'\nimport type {\n Finding as LayerFinding,\n Severity as LayerSeverity,\n MultiLayerVerifier,\n VerifyOptions,\n} from '../multi-layer-verifier'\nimport { RunCritic, type RunTrace } from '../run-critic'\nimport {\n runSemanticConceptJudge,\n SEMANTIC_CONCEPT_JUDGE_VERSION,\n type SemanticConceptJudgeInput,\n type SemanticConceptJudgeOptions,\n type SemanticConceptJudgeResult,\n} from '../semantic-concept-judge'\nimport type { JudgeFn, JudgeInput, JudgeScore, TCloud } from '../types'\nimport type { Analyst, AnalystFinding, AnalystSeverity } from './types'\nimport { makeFinding } from './types'\nimport { settleUsageReceiptFromCostLedger, validateUsageSettlementTimeout } from './usage-receipt'\n\nconst ADAPTER_REV = '1'\n\n// ── Severity bridges ───────────────────────────────────────────────\n\nexport function liftSeverity(s: LayerSeverity): AnalystSeverity {\n switch (s) {\n case 'critical':\n return 'critical'\n case 'major':\n return 'high'\n case 'minor':\n return 'medium'\n case 'info':\n return 'info'\n }\n}\n\n// ── 1. MultiLayerVerifier → Analyst ─────────────────────────────────\n\nexport interface VerifierAdapterOpts<Env> {\n id?: string\n area?: string\n verifier: MultiLayerVerifier<Env>\n /**\n * The verifier expects an `env` per run. Adapters take it from\n * `AnalystRunInputs.custom[<id>]` via the registry's 'custom' routing.\n */\n options?: Omit<VerifyOptions<Env>, 'env'>\n}\n\nexport function createVerifierAdapter<Env>(opts: VerifierAdapterOpts<Env>): Analyst<Env> {\n const id = opts.id ?? 'multi-layer-verifier'\n const area = opts.area ?? 'verification'\n return {\n id,\n description:\n \"Runs a MultiLayerVerifier and lifts each layer's findings into the analyst envelope.\",\n inputKind: 'custom',\n cost: { kind: 'deterministic' },\n version: `verifier-${ADAPTER_REV}`,\n async analyze(env, ctx) {\n const report = await opts.verifier.run({ env, ...opts.options })\n const out: AnalystFinding[] = []\n for (const layer of report.layers) {\n for (const finding of layer.findings) {\n out.push(liftLayerFinding(id, area, layer.layer, finding))\n }\n // Layer-level signal: a failed/error layer is itself a finding\n // even if it didn't emit per-finding rows.\n if (layer.status === 'fail' || layer.status === 'error' || layer.status === 'timeout') {\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: layer.layer,\n claim: `layer \"${layer.layer}\" ${layer.status}: ${layer.reason ?? 'no reason given'}`,\n severity:\n layer.status === 'error' ? 'high' : layer.status === 'timeout' ? 'medium' : 'high',\n confidence: 1,\n evidence_refs: [],\n metadata: {\n layer_status: layer.status,\n duration_ms: layer.durationMs,\n score: layer.score,\n diagnostics: layer.diagnostics,\n },\n }),\n )\n }\n }\n ctx.log?.('verifier complete', {\n layers: report.layers.length,\n blended: report.blendedScore,\n all_pass: report.allPass,\n })\n return out\n },\n }\n}\n\nfunction liftLayerFinding(\n analyst_id: string,\n area: string,\n layer: string,\n f: LayerFinding,\n): AnalystFinding {\n return makeFinding({\n analyst_id,\n area,\n subject: f.layer ?? layer,\n claim: f.message,\n severity: liftSeverity(f.severity),\n confidence: 0.85,\n evidence_refs: f.evidence\n ? [{ kind: 'artifact', uri: 'inline:evidence', excerpt: f.evidence }]\n : [],\n metadata: f.detail,\n })\n}\n\n// ── 2. RunCritic → Analyst ──────────────────────────────────────────\n\nexport interface RunCriticAdapterOpts {\n id?: string\n area?: string\n critic?: RunCritic\n /** Optional threshold below which a dimension is reported as a finding. Default 0.5. */\n threshold?: number\n}\n\nexport function createRunCriticAdapter(opts: RunCriticAdapterOpts = {}): Analyst<RunTrace> {\n const id = opts.id ?? 'run-critic'\n const area = opts.area ?? 'run-quality'\n const critic = opts.critic ?? new RunCritic()\n const threshold = opts.threshold ?? 0.5\n return {\n id,\n description:\n 'Scores a single run across success / grounding / drift / tool-quality and surfaces below-threshold dimensions.',\n inputKind: 'custom',\n cost: { kind: 'deterministic' },\n version: `run-critic-${ADAPTER_REV}`,\n async analyze(trace) {\n const score = critic.scoreTrace(trace)\n const out: AnalystFinding[] = []\n const dims: Array<[keyof typeof score, AnalystSeverity, string]> = [\n ['success', 'critical', 'run did not complete successfully'],\n ['goalProgress', 'high', 'goal progress is low'],\n ['repoGroundedness', 'high', 'output is poorly grounded in the repository'],\n ['toolUseQuality', 'medium', 'tool use quality is low'],\n ['patchQuality', 'medium', 'no real patch/edit evidence'],\n ['testReality', 'high', 'no real test/build evidence'],\n ['finalGate', 'critical', 'final gate is blocking'],\n ]\n for (const [dim, sev, msg] of dims) {\n const value = score[dim] as number\n if (typeof value === 'number' && value < threshold) {\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: dim,\n claim: msg,\n rationale: `${dim}=${value.toFixed(2)} below threshold ${threshold}`,\n severity: sev,\n confidence: 1,\n evidence_refs: [],\n metadata: { dimension: dim, value, threshold, run_id: trace.run.runId },\n }),\n )\n }\n }\n // Drift penalty is high → surface as a finding (inverse threshold).\n if (score.driftPenalty > 1 - threshold) {\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: 'drift',\n claim: 'agent output drifted from repository signal',\n rationale: `driftPenalty=${score.driftPenalty.toFixed(2)}`,\n severity: 'medium',\n confidence: 0.9,\n evidence_refs: [],\n metadata: { drift_penalty: score.driftPenalty, notes: score.notes },\n }),\n )\n }\n return out\n },\n }\n}\n\n// ── 3. JudgeFn → Analyst ────────────────────────────────────────────\n\nexport interface JudgeAdapterOpts {\n id?: string\n area?: string\n judge: JudgeFn\n /** TCloud handle the JudgeFn calls. */\n tcloud: TCloud\n /** Optional cost classification — most judges call an LLM. */\n cost?: Analyst['cost']\n /** Optional threshold below which a JudgeScore becomes a finding. Default 6 (on 0-10 scale). */\n threshold?: number\n}\n\nexport function createJudgeAdapter(opts: JudgeAdapterOpts): Analyst<JudgeInput> {\n const id = opts.id ?? 'judge'\n const area = opts.area ?? 'judge'\n const threshold = opts.threshold ?? 6\n return {\n id,\n description:\n 'Wraps an agent-eval JudgeFn into an analyst; below-threshold dimensions surface as findings.',\n inputKind: 'judge-input',\n cost: opts.cost ?? { kind: 'llm' },\n version: `judge-${ADAPTER_REV}`,\n async analyze(input) {\n const scores = await opts.judge(opts.tcloud, input)\n return scores\n .filter((s) => normalize10(s.score) < threshold)\n .map((s) => liftJudgeScore(id, area, s))\n },\n }\n}\n\nfunction normalize10(s: number): number {\n // JudgeScore convention is 0-10 but some judges emit 0-1. Coerce to 0-10.\n return s <= 1 ? s * 10 : s\n}\n\nfunction liftJudgeScore(analyst_id: string, area: string, s: JudgeScore): AnalystFinding {\n const score10 = normalize10(s.score)\n const severity: AnalystSeverity =\n score10 < 3 ? 'critical' : score10 < 5 ? 'high' : score10 < 7 ? 'medium' : 'low'\n return makeFinding({\n analyst_id,\n area,\n subject: s.dimension,\n claim: `${s.judgeName}/${s.dimension} scored ${score10.toFixed(1)}/10`,\n rationale: s.reasoning,\n severity,\n confidence: 0.8,\n evidence_refs: s.evidence\n ? [{ kind: 'artifact', uri: 'inline:evidence', excerpt: s.evidence }]\n : [],\n // Provenance: this finding IS a judge verdict (an acceptance score), not an\n // observation of behavior. The steer firewall (assertNoJudgeVerdict) rejects\n // it from steering — even when it cites an artifact above — because letting a\n // verdict steer the next attempt is the held-out judge leaking into the loop.\n derived_from_judge: true,\n metadata: { judge_name: s.judgeName, dimension: s.dimension, score_10: score10 },\n })\n}\n\n// ── 4. SemanticConceptJudge → Analyst ──────────────────────────────\n\nexport interface SemanticConceptJudgeAdapterOpts {\n id?: string\n area?: string\n /** Registry context owns cancellation and the per-analyst cost ledger. */\n options?: Omit<SemanticConceptJudgeOptions, 'costLedger' | 'signal'>\n /** Maximum post-cancellation wait for a provider receipt. Default 5 seconds. */\n settlementTimeoutMs?: number\n}\n\nexport function createSemanticConceptJudgeAdapter(\n opts: SemanticConceptJudgeAdapterOpts = {},\n): Analyst<SemanticConceptJudgeInput> {\n const id = opts.id ?? 'semantic-concept-judge'\n const area = opts.area ?? 'concept-coverage'\n const settlementTimeoutMs = validateUsageSettlementTimeout(opts.settlementTimeoutMs)\n return {\n id,\n description:\n 'Runs the semantic-concept judge and surfaces missing / weak concepts as findings.',\n inputKind: 'custom',\n cost: {\n kind: 'llm',\n models: opts.options?.model ? [opts.options.model] : undefined,\n settlement_timeout_ms: settlementTimeoutMs,\n },\n version: `${SEMANTIC_CONCEPT_JUDGE_VERSION}-adapter-${ADAPTER_REV}`,\n async analyze(input, ctx) {\n const costLedger = new CostLedger(ctx.budgetUsd)\n let result: SemanticConceptJudgeResult\n try {\n result = await runSemanticConceptJudge(input, {\n ...opts.options,\n costLedger,\n signal: ctx.signal,\n })\n } finally {\n const usage = await settleUsageReceiptFromCostLedger(costLedger, {\n channel: 'judge',\n timeoutMs: settlementTimeoutMs,\n })\n if (!usage.settled) {\n ctx.log?.('semantic-concept judge provider settlement timed out', {\n pending_calls: usage.pendingCalls,\n timeout_ms: settlementTimeoutMs,\n })\n }\n ctx.recordUsage?.(usage.receipt)\n }\n if (!result.available) {\n return [\n makeFinding({\n analyst_id: id,\n area,\n claim: 'semantic-concept judge unavailable',\n rationale: result.error,\n severity: 'info',\n confidence: 1,\n evidence_refs: [],\n metadata: { reason: result.error },\n }),\n ]\n }\n const out: AnalystFinding[] = []\n for (const f of result.findings) {\n // Only surface gaps: missing concepts or low scores. Concepts at\n // 7+/10 with present=true are not findings — they're successes.\n if (f.present && f.score >= 7) continue\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: f.concept,\n claim: f.present\n ? `concept \"${f.concept}\" is weak (${f.score}/10)`\n : `concept \"${f.concept}\" is missing`,\n rationale: f.evidence,\n severity: liftSeverity(f.severity),\n confidence: 0.85,\n evidence_refs: [{ kind: 'artifact', uri: 'inline:evidence', excerpt: f.evidence }],\n metadata: {\n concept: f.concept,\n present: f.present,\n score_10: f.score,\n },\n }),\n )\n }\n return out\n },\n }\n}\n","// The realness-oracle firewall (docs/learning-flywheel.md, \"The steer is f(trace)\").\n//\n// A realness/authenticity signal has TWO legitimate roles that must stay\n// separated by a firewall:\n// (a) anchor judge J — write-only: scores the chosen output, gates promotion,\n// NEVER seen by the worker/optimizer mid-run (else the loop games it).\n// (b) steer f(trace) — an analyst observes the agent's OWN behavior in the\n// trace (\"imported a stub\", \"used a non-crypto PRNG where encryption was\n// required\") and steers the next attempt. Legitimate, because it is derived\n// from OBSERVABLE BEHAVIOR, not from J's held-out verdict.\n//\n// The correct discriminator is PROVENANCE, not evidence presence. A judge verdict\n// lifted into a finding (createJudgeAdapter → liftJudgeScore) is a verdict even\n// when it cites an artifact; an evidence-less trace-analyst bullet is an\n// observation even though it cites nothing. So the firewall keys on\n// `AnalystFinding.derived_from_judge` (set at the judge lift site), NOT on whether\n// evidence_refs is populated. The instant a verdict steers the next attempt it is\n// a back-channel for J and the loop Goodharts realness exactly as it would\n// Goodhart pass-rate.\n\nimport type { AnalystFinding, EvidenceRef } from './types'\n\n/** Evidence grounded in the agent's OWN execution: OTLP trace elements\n * (`span`/`event`) or the artifact it produced (`artifact`). */\nconst OBSERVABLE_KINDS: ReadonlySet<EvidenceRef['kind']> = new Set<EvidenceRef['kind']>([\n 'span',\n 'event',\n 'artifact',\n])\n\n/** DESCRIPTIVE predicate: does the finding cite at least one observable\n * (span/event/artifact) evidence ref. Useful for ranking evidence quality or\n * rendering — it is NOT the steer gate. Evidence presence is the WRONG\n * discriminator for steering: a legitimate trace-analyst observation may cite\n * nothing (it would be wrongly rejected), and a judge verdict may cite an\n * artifact (it would be wrongly admitted). Use `assertNoJudgeVerdict` to gate\n * steering; use this only where \"is this grounded in observable evidence\" is the\n * literal question. */\nexport function isTraceObservable(finding: AnalystFinding): boolean {\n return finding.evidence_refs.some((ref) => OBSERVABLE_KINDS.has(ref.kind))\n}\n\n/** True iff the finding is a JUDGE VERDICT (an acceptance score lifted into a\n * finding), identified by provenance set at the lift site — independent of\n * whatever evidence it cites. */\nexport function isJudgeVerdict(finding: AnalystFinding): boolean {\n return finding.derived_from_judge === true\n}\n\n/**\n * THE steer firewall. Fail-loud guard for any path that admits analyst findings\n * as STEERING input (the `f(trace)` role): rejects — naming the offenders — any\n * finding whose provenance is a judge verdict, rather than let `J` leak into the\n * loop. Returns the findings unchanged for chaining.\n *\n * Call this at the chokepoint where a detector that ALSO scores/gates has its\n * findings turned into a steer (the judge-and-steer dual-role case). It keys on\n * provenance, so it correctly admits evidence-less trace-analyst observations and\n * correctly rejects an artifact-citing judge verdict — the cases an evidence\n * check gets backwards.\n *\n * It is necessary, not sufficient: it stops PROVENANCE-tagged verdicts. A judge\n * whose output is laundered through a hand-built finding with no provenance flag\n * is out of its reach — provenance must be honestly set at every judge→finding\n * lift (today: createJudgeAdapter). That is why the integrity rule lives at the\n * lift site, and why ProposeContext.judgeScores?: never is the complementary\n * compile-time tripwire on the obvious direct channel.\n */\nexport function assertNoJudgeVerdict(\n findings: ReadonlyArray<AnalystFinding>,\n context = 'steer',\n): ReadonlyArray<AnalystFinding> {\n const leaks = findings.filter(isJudgeVerdict)\n if (leaks.length > 0) {\n throw new Error(\n `${context}: a judge verdict cannot be admitted as steering input — that is the ` +\n `held-out judge leaking into the loop. Offending judge-derived findings: [${leaks\n .map((f) => f.finding_id)\n .join(', ')}]. Steering consumes observations of behavior, never acceptance verdicts.`,\n )\n }\n return findings\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAuCA,IAAM,cAAc;AAIb,SAAS,aAAa,GAAmC;AAC9D,UAAQ,GAAG;AAAA,IACT,KAAK;AACH,aAAO;AAAA,IACT,KAAK;AACH,aAAO;AAAA,IACT,KAAK;AACH,aAAO;AAAA,IACT,KAAK;AACH,aAAO;AAAA,EACX;AACF;AAeO,SAAS,sBAA2B,MAA8C;AACvF,QAAM,KAAK,KAAK,MAAM;AACtB,QAAM,OAAO,KAAK,QAAQ;AAC1B,SAAO;AAAA,IACL;AAAA,IACA,aACE;AAAA,IACF,WAAW;AAAA,IACX,MAAM,EAAE,MAAM,gBAAgB;AAAA,IAC9B,SAAS,YAAY,WAAW;AAAA,IAChC,MAAM,QAAQ,KAAK,KAAK;AACtB,YAAM,SAAS,MAAM,KAAK,SAAS,IAAI,EAAE,KAAK,GAAG,KAAK,QAAQ,CAAC;AAC/D,YAAM,MAAwB,CAAC;AAC/B,iBAAW,SAAS,OAAO,QAAQ;AACjC,mBAAW,WAAW,MAAM,UAAU;AACpC,cAAI,KAAK,iBAAiB,IAAI,MAAM,MAAM,OAAO,OAAO,CAAC;AAAA,QAC3D;AAGA,YAAI,MAAM,WAAW,UAAU,MAAM,WAAW,WAAW,MAAM,WAAW,WAAW;AACrF,cAAI;AAAA,YACF,YAAY;AAAA,cACV,YAAY;AAAA,cACZ;AAAA,cACA,SAAS,MAAM;AAAA,cACf,OAAO,UAAU,MAAM,KAAK,KAAK,MAAM,MAAM,KAAK,MAAM,UAAU,iBAAiB;AAAA,cACnF,UACE,MAAM,WAAW,UAAU,SAAS,MAAM,WAAW,YAAY,WAAW;AAAA,cAC9E,YAAY;AAAA,cACZ,eAAe,CAAC;AAAA,cAChB,UAAU;AAAA,gBACR,cAAc,MAAM;AAAA,gBACpB,aAAa,MAAM;AAAA,gBACnB,OAAO,MAAM;AAAA,gBACb,aAAa,MAAM;AAAA,cACrB;AAAA,YACF,CAAC;AAAA,UACH;AAAA,QACF;AAAA,MACF;AACA,UAAI,MAAM,qBAAqB;AAAA,QAC7B,QAAQ,OAAO,OAAO;AAAA,QACtB,SAAS,OAAO;AAAA,QAChB,UAAU,OAAO;AAAA,MACnB,CAAC;AACD,aAAO;AAAA,IACT;AAAA,EACF;AACF;AAEA,SAAS,iBACP,YACA,MACA,OACA,GACgB;AAChB,SAAO,YAAY;AAAA,IACjB;AAAA,IACA;AAAA,IACA,SAAS,EAAE,SAAS;AAAA,IACpB,OAAO,EAAE;AAAA,IACT,UAAU,aAAa,EAAE,QAAQ;AAAA,IACjC,YAAY;AAAA,IACZ,eAAe,EAAE,WACb,CAAC,EAAE,MAAM,YAAY,KAAK,mBAAmB,SAAS,EAAE,SAAS,CAAC,IAClE,CAAC;AAAA,IACL,UAAU,EAAE;AAAA,EACd,CAAC;AACH;AAYO,SAAS,uBAAuB,OAA6B,CAAC,GAAsB;AACzF,QAAM,KAAK,KAAK,MAAM;AACtB,QAAM,OAAO,KAAK,QAAQ;AAC1B,QAAM,SAAS,KAAK,UAAU,IAAI,UAAU;AAC5C,QAAM,YAAY,KAAK,aAAa;AACpC,SAAO;AAAA,IACL;AAAA,IACA,aACE;AAAA,IACF,WAAW;AAAA,IACX,MAAM,EAAE,MAAM,gBAAgB;AAAA,IAC9B,SAAS,cAAc,WAAW;AAAA,IAClC,MAAM,QAAQ,OAAO;AACnB,YAAM,QAAQ,OAAO,WAAW,KAAK;AACrC,YAAM,MAAwB,CAAC;AAC/B,YAAM,OAA6D;AAAA,QACjE,CAAC,WAAW,YAAY,mCAAmC;AAAA,QAC3D,CAAC,gBAAgB,QAAQ,sBAAsB;AAAA,QAC/C,CAAC,oBAAoB,QAAQ,6CAA6C;AAAA,QAC1E,CAAC,kBAAkB,UAAU,yBAAyB;AAAA,QACtD,CAAC,gBAAgB,UAAU,6BAA6B;AAAA,QACxD,CAAC,eAAe,QAAQ,6BAA6B;AAAA,QACrD,CAAC,aAAa,YAAY,wBAAwB;AAAA,MACpD;AACA,iBAAW,CAAC,KAAK,KAAK,GAAG,KAAK,MAAM;AAClC,cAAM,QAAQ,MAAM,GAAG;AACvB,YAAI,OAAO,UAAU,YAAY,QAAQ,WAAW;AAClD,cAAI;AAAA,YACF,YAAY;AAAA,cACV,YAAY;AAAA,cACZ;AAAA,cACA,SAAS;AAAA,cACT,OAAO;AAAA,cACP,WAAW,GAAG,GAAG,IAAI,MAAM,QAAQ,CAAC,CAAC,oBAAoB,SAAS;AAAA,cAClE,UAAU;AAAA,cACV,YAAY;AAAA,cACZ,eAAe,CAAC;AAAA,cAChB,UAAU,EAAE,WAAW,KAAK,OAAO,WAAW,QAAQ,MAAM,IAAI,MAAM;AAAA,YACxE,CAAC;AAAA,UACH;AAAA,QACF;AAAA,MACF;AAEA,UAAI,MAAM,eAAe,IAAI,WAAW;AACtC,YAAI;AAAA,UACF,YAAY;AAAA,YACV,YAAY;AAAA,YACZ;AAAA,YACA,SAAS;AAAA,YACT,OAAO;AAAA,YACP,WAAW,gBAAgB,MAAM,aAAa,QAAQ,CAAC,CAAC;AAAA,YACxD,UAAU;AAAA,YACV,YAAY;AAAA,YACZ,eAAe,CAAC;AAAA,YAChB,UAAU,EAAE,eAAe,MAAM,cAAc,OAAO,MAAM,MAAM;AAAA,UACpE,CAAC;AAAA,QACH;AAAA,MACF;AACA,aAAO;AAAA,IACT;AAAA,EACF;AACF;AAgBO,SAAS,mBAAmB,MAA6C;AAC9E,QAAM,KAAK,KAAK,MAAM;AACtB,QAAM,OAAO,KAAK,QAAQ;AAC1B,QAAM,YAAY,KAAK,aAAa;AACpC,SAAO;AAAA,IACL;AAAA,IACA,aACE;AAAA,IACF,WAAW;AAAA,IACX,MAAM,KAAK,QAAQ,EAAE,MAAM,MAAM;AAAA,IACjC,SAAS,SAAS,WAAW;AAAA,IAC7B,MAAM,QAAQ,OAAO;AACnB,YAAM,SAAS,MAAM,KAAK,MAAM,KAAK,QAAQ,KAAK;AAClD,aAAO,OACJ,OAAO,CAAC,MAAM,YAAY,EAAE,KAAK,IAAI,SAAS,EAC9C,IAAI,CAAC,MAAM,eAAe,IAAI,MAAM,CAAC,CAAC;AAAA,IAC3C;AAAA,EACF;AACF;AAEA,SAAS,YAAY,GAAmB;AAEtC,SAAO,KAAK,IAAI,IAAI,KAAK;AAC3B;AAEA,SAAS,eAAe,YAAoB,MAAc,GAA+B;AACvF,QAAM,UAAU,YAAY,EAAE,KAAK;AACnC,QAAM,WACJ,UAAU,IAAI,aAAa,UAAU,IAAI,SAAS,UAAU,IAAI,WAAW;AAC7E,SAAO,YAAY;AAAA,IACjB;AAAA,IACA;AAAA,IACA,SAAS,EAAE;AAAA,IACX,OAAO,GAAG,EAAE,SAAS,IAAI,EAAE,SAAS,WAAW,QAAQ,QAAQ,CAAC,CAAC;AAAA,IACjE,WAAW,EAAE;AAAA,IACb;AAAA,IACA,YAAY;AAAA,IACZ,eAAe,EAAE,WACb,CAAC,EAAE,MAAM,YAAY,KAAK,mBAAmB,SAAS,EAAE,SAAS,CAAC,IAClE,CAAC;AAAA;AAAA;AAAA;AAAA;AAAA,IAKL,oBAAoB;AAAA,IACpB,UAAU,EAAE,YAAY,EAAE,WAAW,WAAW,EAAE,WAAW,UAAU,QAAQ;AAAA,EACjF,CAAC;AACH;AAaO,SAAS,kCACd,OAAwC,CAAC,GACL;AACpC,QAAM,KAAK,KAAK,MAAM;AACtB,QAAM,OAAO,KAAK,QAAQ;AAC1B,QAAM,sBAAsB,+BAA+B,KAAK,mBAAmB;AACnF,SAAO;AAAA,IACL;AAAA,IACA,aACE;AAAA,IACF,WAAW;AAAA,IACX,MAAM;AAAA,MACJ,MAAM;AAAA,MACN,QAAQ,KAAK,SAAS,QAAQ,CAAC,KAAK,QAAQ,KAAK,IAAI;AAAA,MACrD,uBAAuB;AAAA,IACzB;AAAA,IACA,SAAS,GAAG,8BAA8B,YAAY,WAAW;AAAA,IACjE,MAAM,QAAQ,OAAO,KAAK;AACxB,YAAM,aAAa,IAAI,WAAW,IAAI,SAAS;AAC/C,UAAI;AACJ,UAAI;AACF,iBAAS,MAAM,wBAAwB,OAAO;AAAA,UAC5C,GAAG,KAAK;AAAA,UACR;AAAA,UACA,QAAQ,IAAI;AAAA,QACd,CAAC;AAAA,MACH,UAAE;AACA,cAAM,QAAQ,MAAM,iCAAiC,YAAY;AAAA,UAC/D,SAAS;AAAA,UACT,WAAW;AAAA,QACb,CAAC;AACD,YAAI,CAAC,MAAM,SAAS;AAClB,cAAI,MAAM,wDAAwD;AAAA,YAChE,eAAe,MAAM;AAAA,YACrB,YAAY;AAAA,UACd,CAAC;AAAA,QACH;AACA,YAAI,cAAc,MAAM,OAAO;AAAA,MACjC;AACA,UAAI,CAAC,OAAO,WAAW;AACrB,eAAO;AAAA,UACL,YAAY;AAAA,YACV,YAAY;AAAA,YACZ;AAAA,YACA,OAAO;AAAA,YACP,WAAW,OAAO;AAAA,YAClB,UAAU;AAAA,YACV,YAAY;AAAA,YACZ,eAAe,CAAC;AAAA,YAChB,UAAU,EAAE,QAAQ,OAAO,MAAM;AAAA,UACnC,CAAC;AAAA,QACH;AAAA,MACF;AACA,YAAM,MAAwB,CAAC;AAC/B,iBAAW,KAAK,OAAO,UAAU;AAG/B,YAAI,EAAE,WAAW,EAAE,SAAS,EAAG;AAC/B,YAAI;AAAA,UACF,YAAY;AAAA,YACV,YAAY;AAAA,YACZ;AAAA,YACA,SAAS,EAAE;AAAA,YACX,OAAO,EAAE,UACL,YAAY,EAAE,OAAO,cAAc,EAAE,KAAK,SAC1C,YAAY,EAAE,OAAO;AAAA,YACzB,WAAW,EAAE;AAAA,YACb,UAAU,aAAa,EAAE,QAAQ;AAAA,YACjC,YAAY;AAAA,YACZ,eAAe,CAAC,EAAE,MAAM,YAAY,KAAK,mBAAmB,SAAS,EAAE,SAAS,CAAC;AAAA,YACjF,UAAU;AAAA,cACR,SAAS,EAAE;AAAA,cACX,SAAS,EAAE;AAAA,cACX,UAAU,EAAE;AAAA,YACd;AAAA,UACF,CAAC;AAAA,QACH;AAAA,MACF;AACA,aAAO;AAAA,IACT;AAAA,EACF;AACF;;;ACvVA,IAAM,mBAAqD,oBAAI,IAAyB;AAAA,EACtF;AAAA,EACA;AAAA,EACA;AACF,CAAC;AAUM,SAAS,kBAAkB,SAAkC;AAClE,SAAO,QAAQ,cAAc,KAAK,CAAC,QAAQ,iBAAiB,IAAI,IAAI,IAAI,CAAC;AAC3E;AAKO,SAAS,eAAe,SAAkC;AAC/D,SAAO,QAAQ,uBAAuB;AACxC;AAqBO,SAAS,qBACd,UACA,UAAU,SACqB;AAC/B,QAAM,QAAQ,SAAS,OAAO,cAAc;AAC5C,MAAI,MAAM,SAAS,GAAG;AACpB,UAAM,IAAI;AAAA,MACR,GAAG,OAAO,sJACoE,MACzE,IAAI,CAAC,MAAM,EAAE,UAAU,EACvB,KAAK,IAAI,CAAC;AAAA,IACjB;AAAA,EACF;AACA,SAAO;AACT;","names":[]}
|
|
1
|
+
{"version":3,"sources":["../../src/analyst/adapters.ts","../../src/analyst/steer-firewall.ts"],"sourcesContent":["/**\n * Adapter factories — lift each existing agent-eval primitive into the\n * Analyst contract without re-implementing it.\n *\n * Five primitives, five factories. Each one:\n * - Builds an Analyst with a stable id (caller chooses; defaults\n * given), a sensible default `inputKind`, a version derived from\n * the wrapped primitive's version + an adapter revision, and an\n * `analyze()` that calls the primitive and lifts its output to\n * AnalystFinding[] using `makeFinding()`.\n * - Maps severities: the existing `Severity` ('critical' | 'major' |\n * 'minor' | 'info') projects onto AnalystSeverity ('critical' |\n * 'high' | 'medium' | 'low' | 'info'); 'major' → 'high', 'minor' →\n * 'medium'. Domain analysts that want finer-grained mapping override.\n *\n * Adapters never own state. Calling the same factory twice with the\n * same primitive instance is safe.\n */\n\nimport { CostLedger } from '../cost-ledger'\nimport type {\n Finding as LayerFinding,\n Severity as LayerSeverity,\n MultiLayerVerifier,\n VerifyOptions,\n} from '../multi-layer-verifier'\nimport { RunCritic, type RunTrace } from '../run-critic'\nimport {\n runSemanticConceptJudge,\n SEMANTIC_CONCEPT_JUDGE_VERSION,\n type SemanticConceptJudgeInput,\n type SemanticConceptJudgeOptions,\n type SemanticConceptJudgeResult,\n} from '../semantic-concept-judge'\nimport type { JudgeFn, JudgeInput, JudgeScore } from '../types'\nimport type { ChatClient } from './chat-client'\nimport type { Analyst, AnalystFinding, AnalystSeverity } from './types'\nimport { makeFinding } from './types'\nimport { settleUsageReceiptFromCostLedger, validateUsageSettlementTimeout } from './usage-receipt'\n\nconst ADAPTER_REV = '1'\n\n// ── Severity bridges ───────────────────────────────────────────────\n\nexport function liftSeverity(s: LayerSeverity): AnalystSeverity {\n switch (s) {\n case 'critical':\n return 'critical'\n case 'major':\n return 'high'\n case 'minor':\n return 'medium'\n case 'info':\n return 'info'\n }\n}\n\n// ── 1. MultiLayerVerifier → Analyst ─────────────────────────────────\n\nexport interface VerifierAdapterOpts<Env> {\n id?: string\n area?: string\n verifier: MultiLayerVerifier<Env>\n /**\n * The verifier expects an `env` per run. Adapters take it from\n * `AnalystRunInputs.custom[<id>]` via the registry's 'custom' routing.\n */\n options?: Omit<VerifyOptions<Env>, 'env'>\n}\n\nexport function createVerifierAdapter<Env>(opts: VerifierAdapterOpts<Env>): Analyst<Env> {\n const id = opts.id ?? 'multi-layer-verifier'\n const area = opts.area ?? 'verification'\n return {\n id,\n description:\n \"Runs a MultiLayerVerifier and lifts each layer's findings into the analyst envelope.\",\n inputKind: 'custom',\n cost: { kind: 'deterministic' },\n version: `verifier-${ADAPTER_REV}`,\n async analyze(env, ctx) {\n const report = await opts.verifier.run({ env, ...opts.options })\n const out: AnalystFinding[] = []\n for (const layer of report.layers) {\n for (const finding of layer.findings) {\n out.push(liftLayerFinding(id, area, layer.layer, finding))\n }\n // Layer-level signal: a failed/error layer is itself a finding\n // even if it didn't emit per-finding rows.\n if (layer.status === 'fail' || layer.status === 'error' || layer.status === 'timeout') {\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: layer.layer,\n claim: `layer \"${layer.layer}\" ${layer.status}: ${layer.reason ?? 'no reason given'}`,\n severity:\n layer.status === 'error' ? 'high' : layer.status === 'timeout' ? 'medium' : 'high',\n confidence: 1,\n evidence_refs: [],\n metadata: {\n layer_status: layer.status,\n duration_ms: layer.durationMs,\n score: layer.score,\n diagnostics: layer.diagnostics,\n },\n }),\n )\n }\n }\n ctx.log?.('verifier complete', {\n layers: report.layers.length,\n blended: report.blendedScore,\n all_pass: report.allPass,\n })\n return out\n },\n }\n}\n\nfunction liftLayerFinding(\n analyst_id: string,\n area: string,\n layer: string,\n f: LayerFinding,\n): AnalystFinding {\n return makeFinding({\n analyst_id,\n area,\n subject: f.layer ?? layer,\n claim: f.message,\n severity: liftSeverity(f.severity),\n confidence: 0.85,\n evidence_refs: f.evidence\n ? [{ kind: 'artifact', uri: 'inline:evidence', excerpt: f.evidence }]\n : [],\n metadata: f.detail,\n })\n}\n\n// ── 2. RunCritic → Analyst ──────────────────────────────────────────\n\nexport interface RunCriticAdapterOpts {\n id?: string\n area?: string\n critic?: RunCritic\n /** Optional threshold below which a dimension is reported as a finding. Default 0.5. */\n threshold?: number\n}\n\nexport function createRunCriticAdapter(opts: RunCriticAdapterOpts = {}): Analyst<RunTrace> {\n const id = opts.id ?? 'run-critic'\n const area = opts.area ?? 'run-quality'\n const critic = opts.critic ?? new RunCritic()\n const threshold = opts.threshold ?? 0.5\n return {\n id,\n description:\n 'Scores a single run across success / grounding / drift / tool-quality and surfaces below-threshold dimensions.',\n inputKind: 'custom',\n cost: { kind: 'deterministic' },\n version: `run-critic-${ADAPTER_REV}`,\n async analyze(trace) {\n const score = critic.scoreTrace(trace)\n const out: AnalystFinding[] = []\n const dims: Array<[keyof typeof score, AnalystSeverity, string]> = [\n ['success', 'critical', 'run did not complete successfully'],\n ['goalProgress', 'high', 'goal progress is low'],\n ['repoGroundedness', 'high', 'output is poorly grounded in the repository'],\n ['toolUseQuality', 'medium', 'tool use quality is low'],\n ['patchQuality', 'medium', 'no real patch/edit evidence'],\n ['testReality', 'high', 'no real test/build evidence'],\n ['finalGate', 'critical', 'final gate is blocking'],\n ]\n for (const [dim, sev, msg] of dims) {\n const value = score[dim] as number\n if (typeof value === 'number' && value < threshold) {\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: dim,\n claim: msg,\n rationale: `${dim}=${value.toFixed(2)} below threshold ${threshold}`,\n severity: sev,\n confidence: 1,\n evidence_refs: [],\n metadata: { dimension: dim, value, threshold, run_id: trace.run.runId },\n }),\n )\n }\n }\n // Drift penalty is high → surface as a finding (inverse threshold).\n if (score.driftPenalty > 1 - threshold) {\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: 'drift',\n claim: 'agent output drifted from repository signal',\n rationale: `driftPenalty=${score.driftPenalty.toFixed(2)}`,\n severity: 'medium',\n confidence: 0.9,\n evidence_refs: [],\n metadata: { drift_penalty: score.driftPenalty, notes: score.notes },\n }),\n )\n }\n return out\n },\n }\n}\n\n// ── 3. JudgeFn → Analyst ────────────────────────────────────────────\n\nexport interface JudgeAdapterOpts {\n id?: string\n area?: string\n judge: JudgeFn\n /** Chat client passed to the JudgeFn. */\n chat: ChatClient\n /** Optional cost classification — most judges call an LLM. */\n cost?: Analyst['cost']\n /** Optional threshold below which a JudgeScore becomes a finding. Default 6 (on 0-10 scale). */\n threshold?: number\n}\n\nexport function createJudgeAdapter(opts: JudgeAdapterOpts): Analyst<JudgeInput> {\n const id = opts.id ?? 'judge'\n const area = opts.area ?? 'judge'\n const threshold = opts.threshold ?? 6\n return {\n id,\n description:\n 'Wraps an agent-eval JudgeFn into an analyst; below-threshold dimensions surface as findings.',\n inputKind: 'judge-input',\n cost: opts.cost ?? { kind: 'llm' },\n version: `judge-${ADAPTER_REV}`,\n async analyze(input) {\n const scores = await opts.judge(opts.chat, input)\n return scores\n .filter((s) => normalize10(s.score) < threshold)\n .map((s) => liftJudgeScore(id, area, s))\n },\n }\n}\n\nfunction normalize10(s: number): number {\n // JudgeScore convention is 0-10 but some judges emit 0-1. Coerce to 0-10.\n return s <= 1 ? s * 10 : s\n}\n\nfunction liftJudgeScore(analyst_id: string, area: string, s: JudgeScore): AnalystFinding {\n const score10 = normalize10(s.score)\n const severity: AnalystSeverity =\n score10 < 3 ? 'critical' : score10 < 5 ? 'high' : score10 < 7 ? 'medium' : 'low'\n return makeFinding({\n analyst_id,\n area,\n subject: s.dimension,\n claim: `${s.judgeName}/${s.dimension} scored ${score10.toFixed(1)}/10`,\n rationale: s.reasoning,\n severity,\n confidence: 0.8,\n evidence_refs: s.evidence\n ? [{ kind: 'artifact', uri: 'inline:evidence', excerpt: s.evidence }]\n : [],\n // Provenance: this finding IS a judge verdict (an acceptance score), not an\n // observation of behavior. The steer firewall (assertNoJudgeVerdict) rejects\n // it from steering — even when it cites an artifact above — because letting a\n // verdict steer the next attempt is the held-out judge leaking into the loop.\n derived_from_judge: true,\n metadata: { judge_name: s.judgeName, dimension: s.dimension, score_10: score10 },\n })\n}\n\n// ── 4. SemanticConceptJudge → Analyst ──────────────────────────────\n\nexport interface SemanticConceptJudgeAdapterOpts {\n id?: string\n area?: string\n /** Registry context owns cancellation and the per-analyst cost ledger. */\n options?: Omit<SemanticConceptJudgeOptions, 'costLedger' | 'signal'>\n /** Maximum post-cancellation wait for a provider receipt. Default 5 seconds. */\n settlementTimeoutMs?: number\n}\n\nexport function createSemanticConceptJudgeAdapter(\n opts: SemanticConceptJudgeAdapterOpts = {},\n): Analyst<SemanticConceptJudgeInput> {\n const id = opts.id ?? 'semantic-concept-judge'\n const area = opts.area ?? 'concept-coverage'\n const settlementTimeoutMs = validateUsageSettlementTimeout(opts.settlementTimeoutMs)\n return {\n id,\n description:\n 'Runs the semantic-concept judge and surfaces missing / weak concepts as findings.',\n inputKind: 'custom',\n cost: {\n kind: 'llm',\n models: opts.options?.model ? [opts.options.model] : undefined,\n settlement_timeout_ms: settlementTimeoutMs,\n },\n version: `${SEMANTIC_CONCEPT_JUDGE_VERSION}-adapter-${ADAPTER_REV}`,\n async analyze(input, ctx) {\n const costLedger = new CostLedger(ctx.budgetUsd)\n let result: SemanticConceptJudgeResult\n try {\n result = await runSemanticConceptJudge(input, {\n ...opts.options,\n costLedger,\n signal: ctx.signal,\n })\n } finally {\n const usage = await settleUsageReceiptFromCostLedger(costLedger, {\n channel: 'judge',\n timeoutMs: settlementTimeoutMs,\n })\n if (!usage.settled) {\n ctx.log?.('semantic-concept judge provider settlement timed out', {\n pending_calls: usage.pendingCalls,\n timeout_ms: settlementTimeoutMs,\n })\n }\n ctx.recordUsage?.(usage.receipt)\n }\n if (!result.available) {\n return [\n makeFinding({\n analyst_id: id,\n area,\n claim: 'semantic-concept judge unavailable',\n rationale: result.error,\n severity: 'info',\n confidence: 1,\n evidence_refs: [],\n metadata: { reason: result.error },\n }),\n ]\n }\n const out: AnalystFinding[] = []\n for (const f of result.findings) {\n // Only surface gaps: missing concepts or low scores. Concepts at\n // 7+/10 with present=true are not findings — they're successes.\n if (f.present && f.score >= 7) continue\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: f.concept,\n claim: f.present\n ? `concept \"${f.concept}\" is weak (${f.score}/10)`\n : `concept \"${f.concept}\" is missing`,\n rationale: f.evidence,\n severity: liftSeverity(f.severity),\n confidence: 0.85,\n evidence_refs: [{ kind: 'artifact', uri: 'inline:evidence', excerpt: f.evidence }],\n metadata: {\n concept: f.concept,\n present: f.present,\n score_10: f.score,\n },\n }),\n )\n }\n return out\n },\n }\n}\n","// The realness-oracle firewall (docs/learning-flywheel.md, \"The steer is f(trace)\").\n//\n// A realness/authenticity signal has TWO legitimate roles that must stay\n// separated by a firewall:\n// (a) anchor judge J — write-only: scores the chosen output, gates promotion,\n// NEVER seen by the worker/optimizer mid-run (else the loop games it).\n// (b) steer f(trace) — an analyst observes the agent's OWN behavior in the\n// trace (\"imported a stub\", \"used a non-crypto PRNG where encryption was\n// required\") and steers the next attempt. Legitimate, because it is derived\n// from OBSERVABLE BEHAVIOR, not from J's held-out verdict.\n//\n// The correct discriminator is PROVENANCE, not evidence presence. A judge verdict\n// lifted into a finding (createJudgeAdapter → liftJudgeScore) is a verdict even\n// when it cites an artifact; an evidence-less trace-analyst bullet is an\n// observation even though it cites nothing. So the firewall keys on\n// `AnalystFinding.derived_from_judge` (set at the judge lift site), NOT on whether\n// evidence_refs is populated. The instant a verdict steers the next attempt it is\n// a back-channel for J and the loop Goodharts realness exactly as it would\n// Goodhart pass-rate.\n\nimport type { AnalystFinding, EvidenceRef } from './types'\n\n/** Evidence grounded in the agent's OWN execution: OTLP trace elements\n * (`span`/`event`) or the artifact it produced (`artifact`). */\nconst OBSERVABLE_KINDS: ReadonlySet<EvidenceRef['kind']> = new Set<EvidenceRef['kind']>([\n 'span',\n 'event',\n 'artifact',\n])\n\n/** DESCRIPTIVE predicate: does the finding cite at least one observable\n * (span/event/artifact) evidence ref. Useful for ranking evidence quality or\n * rendering — it is NOT the steer gate. Evidence presence is the WRONG\n * discriminator for steering: a legitimate trace-analyst observation may cite\n * nothing (it would be wrongly rejected), and a judge verdict may cite an\n * artifact (it would be wrongly admitted). Use `assertNoJudgeVerdict` to gate\n * steering; use this only where \"is this grounded in observable evidence\" is the\n * literal question. */\nexport function isTraceObservable(finding: AnalystFinding): boolean {\n return finding.evidence_refs.some((ref) => OBSERVABLE_KINDS.has(ref.kind))\n}\n\n/** True iff the finding is a JUDGE VERDICT (an acceptance score lifted into a\n * finding), identified by provenance set at the lift site — independent of\n * whatever evidence it cites. */\nexport function isJudgeVerdict(finding: AnalystFinding): boolean {\n return finding.derived_from_judge === true\n}\n\n/**\n * THE steer firewall. Fail-loud guard for any path that admits analyst findings\n * as STEERING input (the `f(trace)` role): rejects — naming the offenders — any\n * finding whose provenance is a judge verdict, rather than let `J` leak into the\n * loop. Returns the findings unchanged for chaining.\n *\n * Call this at the chokepoint where a detector that ALSO scores/gates has its\n * findings turned into a steer (the judge-and-steer dual-role case). It keys on\n * provenance, so it correctly admits evidence-less trace-analyst observations and\n * correctly rejects an artifact-citing judge verdict — the cases an evidence\n * check gets backwards.\n *\n * It is necessary, not sufficient: it stops PROVENANCE-tagged verdicts. A judge\n * whose output is laundered through a hand-built finding with no provenance flag\n * is out of its reach — provenance must be honestly set at every judge→finding\n * lift (today: createJudgeAdapter). That is why the integrity rule lives at the\n * lift site, and why ProposeContext.judgeScores?: never is the complementary\n * compile-time tripwire on the obvious direct channel.\n */\nexport function assertNoJudgeVerdict(\n findings: ReadonlyArray<AnalystFinding>,\n context = 'steer',\n): ReadonlyArray<AnalystFinding> {\n const leaks = findings.filter(isJudgeVerdict)\n if (leaks.length > 0) {\n throw new Error(\n `${context}: a judge verdict cannot be admitted as steering input — that is the ` +\n `held-out judge leaking into the loop. Offending judge-derived findings: [${leaks\n .map((f) => f.finding_id)\n .join(', ')}]. Steering consumes observations of behavior, never acceptance verdicts.`,\n )\n }\n return findings\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAwCA,IAAM,cAAc;AAIb,SAAS,aAAa,GAAmC;AAC9D,UAAQ,GAAG;AAAA,IACT,KAAK;AACH,aAAO;AAAA,IACT,KAAK;AACH,aAAO;AAAA,IACT,KAAK;AACH,aAAO;AAAA,IACT,KAAK;AACH,aAAO;AAAA,EACX;AACF;AAeO,SAAS,sBAA2B,MAA8C;AACvF,QAAM,KAAK,KAAK,MAAM;AACtB,QAAM,OAAO,KAAK,QAAQ;AAC1B,SAAO;AAAA,IACL;AAAA,IACA,aACE;AAAA,IACF,WAAW;AAAA,IACX,MAAM,EAAE,MAAM,gBAAgB;AAAA,IAC9B,SAAS,YAAY,WAAW;AAAA,IAChC,MAAM,QAAQ,KAAK,KAAK;AACtB,YAAM,SAAS,MAAM,KAAK,SAAS,IAAI,EAAE,KAAK,GAAG,KAAK,QAAQ,CAAC;AAC/D,YAAM,MAAwB,CAAC;AAC/B,iBAAW,SAAS,OAAO,QAAQ;AACjC,mBAAW,WAAW,MAAM,UAAU;AACpC,cAAI,KAAK,iBAAiB,IAAI,MAAM,MAAM,OAAO,OAAO,CAAC;AAAA,QAC3D;AAGA,YAAI,MAAM,WAAW,UAAU,MAAM,WAAW,WAAW,MAAM,WAAW,WAAW;AACrF,cAAI;AAAA,YACF,YAAY;AAAA,cACV,YAAY;AAAA,cACZ;AAAA,cACA,SAAS,MAAM;AAAA,cACf,OAAO,UAAU,MAAM,KAAK,KAAK,MAAM,MAAM,KAAK,MAAM,UAAU,iBAAiB;AAAA,cACnF,UACE,MAAM,WAAW,UAAU,SAAS,MAAM,WAAW,YAAY,WAAW;AAAA,cAC9E,YAAY;AAAA,cACZ,eAAe,CAAC;AAAA,cAChB,UAAU;AAAA,gBACR,cAAc,MAAM;AAAA,gBACpB,aAAa,MAAM;AAAA,gBACnB,OAAO,MAAM;AAAA,gBACb,aAAa,MAAM;AAAA,cACrB;AAAA,YACF,CAAC;AAAA,UACH;AAAA,QACF;AAAA,MACF;AACA,UAAI,MAAM,qBAAqB;AAAA,QAC7B,QAAQ,OAAO,OAAO;AAAA,QACtB,SAAS,OAAO;AAAA,QAChB,UAAU,OAAO;AAAA,MACnB,CAAC;AACD,aAAO;AAAA,IACT;AAAA,EACF;AACF;AAEA,SAAS,iBACP,YACA,MACA,OACA,GACgB;AAChB,SAAO,YAAY;AAAA,IACjB;AAAA,IACA;AAAA,IACA,SAAS,EAAE,SAAS;AAAA,IACpB,OAAO,EAAE;AAAA,IACT,UAAU,aAAa,EAAE,QAAQ;AAAA,IACjC,YAAY;AAAA,IACZ,eAAe,EAAE,WACb,CAAC,EAAE,MAAM,YAAY,KAAK,mBAAmB,SAAS,EAAE,SAAS,CAAC,IAClE,CAAC;AAAA,IACL,UAAU,EAAE;AAAA,EACd,CAAC;AACH;AAYO,SAAS,uBAAuB,OAA6B,CAAC,GAAsB;AACzF,QAAM,KAAK,KAAK,MAAM;AACtB,QAAM,OAAO,KAAK,QAAQ;AAC1B,QAAM,SAAS,KAAK,UAAU,IAAI,UAAU;AAC5C,QAAM,YAAY,KAAK,aAAa;AACpC,SAAO;AAAA,IACL;AAAA,IACA,aACE;AAAA,IACF,WAAW;AAAA,IACX,MAAM,EAAE,MAAM,gBAAgB;AAAA,IAC9B,SAAS,cAAc,WAAW;AAAA,IAClC,MAAM,QAAQ,OAAO;AACnB,YAAM,QAAQ,OAAO,WAAW,KAAK;AACrC,YAAM,MAAwB,CAAC;AAC/B,YAAM,OAA6D;AAAA,QACjE,CAAC,WAAW,YAAY,mCAAmC;AAAA,QAC3D,CAAC,gBAAgB,QAAQ,sBAAsB;AAAA,QAC/C,CAAC,oBAAoB,QAAQ,6CAA6C;AAAA,QAC1E,CAAC,kBAAkB,UAAU,yBAAyB;AAAA,QACtD,CAAC,gBAAgB,UAAU,6BAA6B;AAAA,QACxD,CAAC,eAAe,QAAQ,6BAA6B;AAAA,QACrD,CAAC,aAAa,YAAY,wBAAwB;AAAA,MACpD;AACA,iBAAW,CAAC,KAAK,KAAK,GAAG,KAAK,MAAM;AAClC,cAAM,QAAQ,MAAM,GAAG;AACvB,YAAI,OAAO,UAAU,YAAY,QAAQ,WAAW;AAClD,cAAI;AAAA,YACF,YAAY;AAAA,cACV,YAAY;AAAA,cACZ;AAAA,cACA,SAAS;AAAA,cACT,OAAO;AAAA,cACP,WAAW,GAAG,GAAG,IAAI,MAAM,QAAQ,CAAC,CAAC,oBAAoB,SAAS;AAAA,cAClE,UAAU;AAAA,cACV,YAAY;AAAA,cACZ,eAAe,CAAC;AAAA,cAChB,UAAU,EAAE,WAAW,KAAK,OAAO,WAAW,QAAQ,MAAM,IAAI,MAAM;AAAA,YACxE,CAAC;AAAA,UACH;AAAA,QACF;AAAA,MACF;AAEA,UAAI,MAAM,eAAe,IAAI,WAAW;AACtC,YAAI;AAAA,UACF,YAAY;AAAA,YACV,YAAY;AAAA,YACZ;AAAA,YACA,SAAS;AAAA,YACT,OAAO;AAAA,YACP,WAAW,gBAAgB,MAAM,aAAa,QAAQ,CAAC,CAAC;AAAA,YACxD,UAAU;AAAA,YACV,YAAY;AAAA,YACZ,eAAe,CAAC;AAAA,YAChB,UAAU,EAAE,eAAe,MAAM,cAAc,OAAO,MAAM,MAAM;AAAA,UACpE,CAAC;AAAA,QACH;AAAA,MACF;AACA,aAAO;AAAA,IACT;AAAA,EACF;AACF;AAgBO,SAAS,mBAAmB,MAA6C;AAC9E,QAAM,KAAK,KAAK,MAAM;AACtB,QAAM,OAAO,KAAK,QAAQ;AAC1B,QAAM,YAAY,KAAK,aAAa;AACpC,SAAO;AAAA,IACL;AAAA,IACA,aACE;AAAA,IACF,WAAW;AAAA,IACX,MAAM,KAAK,QAAQ,EAAE,MAAM,MAAM;AAAA,IACjC,SAAS,SAAS,WAAW;AAAA,IAC7B,MAAM,QAAQ,OAAO;AACnB,YAAM,SAAS,MAAM,KAAK,MAAM,KAAK,MAAM,KAAK;AAChD,aAAO,OACJ,OAAO,CAAC,MAAM,YAAY,EAAE,KAAK,IAAI,SAAS,EAC9C,IAAI,CAAC,MAAM,eAAe,IAAI,MAAM,CAAC,CAAC;AAAA,IAC3C;AAAA,EACF;AACF;AAEA,SAAS,YAAY,GAAmB;AAEtC,SAAO,KAAK,IAAI,IAAI,KAAK;AAC3B;AAEA,SAAS,eAAe,YAAoB,MAAc,GAA+B;AACvF,QAAM,UAAU,YAAY,EAAE,KAAK;AACnC,QAAM,WACJ,UAAU,IAAI,aAAa,UAAU,IAAI,SAAS,UAAU,IAAI,WAAW;AAC7E,SAAO,YAAY;AAAA,IACjB;AAAA,IACA;AAAA,IACA,SAAS,EAAE;AAAA,IACX,OAAO,GAAG,EAAE,SAAS,IAAI,EAAE,SAAS,WAAW,QAAQ,QAAQ,CAAC,CAAC;AAAA,IACjE,WAAW,EAAE;AAAA,IACb;AAAA,IACA,YAAY;AAAA,IACZ,eAAe,EAAE,WACb,CAAC,EAAE,MAAM,YAAY,KAAK,mBAAmB,SAAS,EAAE,SAAS,CAAC,IAClE,CAAC;AAAA;AAAA;AAAA;AAAA;AAAA,IAKL,oBAAoB;AAAA,IACpB,UAAU,EAAE,YAAY,EAAE,WAAW,WAAW,EAAE,WAAW,UAAU,QAAQ;AAAA,EACjF,CAAC;AACH;AAaO,SAAS,kCACd,OAAwC,CAAC,GACL;AACpC,QAAM,KAAK,KAAK,MAAM;AACtB,QAAM,OAAO,KAAK,QAAQ;AAC1B,QAAM,sBAAsB,+BAA+B,KAAK,mBAAmB;AACnF,SAAO;AAAA,IACL;AAAA,IACA,aACE;AAAA,IACF,WAAW;AAAA,IACX,MAAM;AAAA,MACJ,MAAM;AAAA,MACN,QAAQ,KAAK,SAAS,QAAQ,CAAC,KAAK,QAAQ,KAAK,IAAI;AAAA,MACrD,uBAAuB;AAAA,IACzB;AAAA,IACA,SAAS,GAAG,8BAA8B,YAAY,WAAW;AAAA,IACjE,MAAM,QAAQ,OAAO,KAAK;AACxB,YAAM,aAAa,IAAI,WAAW,IAAI,SAAS;AAC/C,UAAI;AACJ,UAAI;AACF,iBAAS,MAAM,wBAAwB,OAAO;AAAA,UAC5C,GAAG,KAAK;AAAA,UACR;AAAA,UACA,QAAQ,IAAI;AAAA,QACd,CAAC;AAAA,MACH,UAAE;AACA,cAAM,QAAQ,MAAM,iCAAiC,YAAY;AAAA,UAC/D,SAAS;AAAA,UACT,WAAW;AAAA,QACb,CAAC;AACD,YAAI,CAAC,MAAM,SAAS;AAClB,cAAI,MAAM,wDAAwD;AAAA,YAChE,eAAe,MAAM;AAAA,YACrB,YAAY;AAAA,UACd,CAAC;AAAA,QACH;AACA,YAAI,cAAc,MAAM,OAAO;AAAA,MACjC;AACA,UAAI,CAAC,OAAO,WAAW;AACrB,eAAO;AAAA,UACL,YAAY;AAAA,YACV,YAAY;AAAA,YACZ;AAAA,YACA,OAAO;AAAA,YACP,WAAW,OAAO;AAAA,YAClB,UAAU;AAAA,YACV,YAAY;AAAA,YACZ,eAAe,CAAC;AAAA,YAChB,UAAU,EAAE,QAAQ,OAAO,MAAM;AAAA,UACnC,CAAC;AAAA,QACH;AAAA,MACF;AACA,YAAM,MAAwB,CAAC;AAC/B,iBAAW,KAAK,OAAO,UAAU;AAG/B,YAAI,EAAE,WAAW,EAAE,SAAS,EAAG;AAC/B,YAAI;AAAA,UACF,YAAY;AAAA,YACV,YAAY;AAAA,YACZ;AAAA,YACA,SAAS,EAAE;AAAA,YACX,OAAO,EAAE,UACL,YAAY,EAAE,OAAO,cAAc,EAAE,KAAK,SAC1C,YAAY,EAAE,OAAO;AAAA,YACzB,WAAW,EAAE;AAAA,YACb,UAAU,aAAa,EAAE,QAAQ;AAAA,YACjC,YAAY;AAAA,YACZ,eAAe,CAAC,EAAE,MAAM,YAAY,KAAK,mBAAmB,SAAS,EAAE,SAAS,CAAC;AAAA,YACjF,UAAU;AAAA,cACR,SAAS,EAAE;AAAA,cACX,SAAS,EAAE;AAAA,cACX,UAAU,EAAE;AAAA,YACd;AAAA,UACF,CAAC;AAAA,QACH;AAAA,MACF;AACA,aAAO;AAAA,IACT;AAAA,EACF;AACF;;;ACxVA,IAAM,mBAAqD,oBAAI,IAAyB;AAAA,EACtF;AAAA,EACA;AAAA,EACA;AACF,CAAC;AAUM,SAAS,kBAAkB,SAAkC;AAClE,SAAO,QAAQ,cAAc,KAAK,CAAC,QAAQ,iBAAiB,IAAI,IAAI,IAAI,CAAC;AAC3E;AAKO,SAAS,eAAe,SAAkC;AAC/D,SAAO,QAAQ,uBAAuB;AACxC;AAqBO,SAAS,qBACd,UACA,UAAU,SACqB;AAC/B,QAAM,QAAQ,SAAS,OAAO,cAAc;AAC5C,MAAI,MAAM,SAAS,GAAG;AACpB,UAAM,IAAI;AAAA,MACR,GAAG,OAAO,sJACoE,MACzE,IAAI,CAAC,MAAM,EAAE,UAAU,EACvB,KAAK,IAAI,CAAC;AAAA,IACjB;AAAA,EACF;AACA,SAAO;AACT;","names":[]}
|
|
@@ -329,20 +329,12 @@ interface OffPolicyTrajectory {
|
|
|
329
329
|
* values must come from a model cross-fitted or trained outside this row.
|
|
330
330
|
*/
|
|
331
331
|
vHatTarget?: number | null;
|
|
332
|
-
/**
|
|
333
|
-
* @deprecated Use `qHatChosen` and `vHatTarget` together. When the new pair
|
|
334
|
-
* is absent, this scalar is used as both terms to preserve existing results.
|
|
335
|
-
* When the new pair is present, this field is ignored.
|
|
336
|
-
*/
|
|
337
|
-
qHat?: number | null;
|
|
338
332
|
}
|
|
339
333
|
interface OffPolicyContributionCounts {
|
|
340
334
|
/** Contributions using the contextual-bandit doubly-robust formula. */
|
|
341
335
|
dr: number;
|
|
342
336
|
/** Contributions using exact IPS because no reward-model estimate was supplied. */
|
|
343
337
|
ipsFallback: number;
|
|
344
|
-
/** Contributions using the deprecated single-scalar formula. */
|
|
345
|
-
legacyScalar: number;
|
|
346
338
|
}
|
|
347
339
|
interface OffPolicyEstimate {
|
|
348
340
|
/** Estimated value of the target policy. */
|
|
@@ -473,8 +465,6 @@ interface BeliefDecisionPoint {
|
|
|
473
465
|
targetProb?: number;
|
|
474
466
|
qHatChosen?: number | null;
|
|
475
467
|
vHatTarget?: number | null;
|
|
476
|
-
/** @deprecated Use `qHatChosen` and `vHatTarget` together. */
|
|
477
|
-
qHat?: number | null;
|
|
478
468
|
costUsd?: number;
|
|
479
469
|
evidence: BeliefEvidenceRef[];
|
|
480
470
|
outcome?: BeliefDecisionOutcome;
|
|
@@ -498,8 +488,6 @@ interface BeliefPolicyDecision {
|
|
|
498
488
|
targetProb?: number;
|
|
499
489
|
qHatChosen?: number | null;
|
|
500
490
|
vHatTarget?: number | null;
|
|
501
|
-
/** @deprecated Use `qHatChosen` and `vHatTarget` together. */
|
|
502
|
-
qHat?: number | null;
|
|
503
491
|
reason?: string;
|
|
504
492
|
reasons?: BeliefDecisionReason[];
|
|
505
493
|
}
|
|
@@ -512,8 +500,6 @@ interface BeliefOpeTargetPolicy {
|
|
|
512
500
|
targetProbOf(point: BeliefDecisionPoint): number | null | undefined;
|
|
513
501
|
qHatChosenOf?(point: BeliefDecisionPoint): number | null | undefined;
|
|
514
502
|
vHatTargetOf?(point: BeliefDecisionPoint): number | null | undefined;
|
|
515
|
-
/** @deprecated Use `qHatChosenOf` and `vHatTargetOf` together. */
|
|
516
|
-
qHatOf?(point: BeliefDecisionPoint): number | null | undefined;
|
|
517
503
|
}
|
|
518
504
|
interface BeliefUtilityOptions {
|
|
519
505
|
successUtility?: number;
|
|
@@ -1110,7 +1096,8 @@ interface BeliefShadowProbeResponse {
|
|
|
1110
1096
|
evidenceRefs?: string[];
|
|
1111
1097
|
wouldChangeMindIf?: string[];
|
|
1112
1098
|
targetProb?: number;
|
|
1113
|
-
|
|
1099
|
+
qHatChosen?: number | null;
|
|
1100
|
+
vHatTarget?: number | null;
|
|
1114
1101
|
metadata?: Record<string, unknown>;
|
|
1115
1102
|
}
|
|
1116
1103
|
interface BeliefShadowProbeRecord extends BeliefShadowProbeResponse {
|
|
@@ -1220,8 +1207,6 @@ interface RuntimeBeliefDecisionPointOptions {
|
|
|
1220
1207
|
targetProb?: number;
|
|
1221
1208
|
qHatChosen?: number | null;
|
|
1222
1209
|
vHatTarget?: number | null;
|
|
1223
|
-
/** @deprecated Use `qHatChosen` and `vHatTarget` together. */
|
|
1224
|
-
qHat?: number | null;
|
|
1225
1210
|
costUsd?: number;
|
|
1226
1211
|
outcome?: BeliefDecisionOutcome;
|
|
1227
1212
|
metadata?: Record<string, unknown>;
|
|
@@ -1264,8 +1249,6 @@ interface RuntimeBeliefDecisionLabel {
|
|
|
1264
1249
|
targetProb?: number;
|
|
1265
1250
|
qHatChosen?: number | null;
|
|
1266
1251
|
vHatTarget?: number | null;
|
|
1267
|
-
/** @deprecated Use `qHatChosen` and `vHatTarget` together. */
|
|
1268
|
-
qHat?: number | null;
|
|
1269
1252
|
costUsd?: number;
|
|
1270
1253
|
splitTag?: RunSplitTag;
|
|
1271
1254
|
metadata?: Record<string, unknown>;
|