@tangle-network/agent-eval 0.128.2 → 0.129.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +265 -0
- package/README.md +18 -0
- package/dist/analyst/index.d.ts +107 -165
- package/dist/analyst/index.js +5 -9
- package/dist/analyst/index.js.map +1 -1
- package/dist/belief-state/index.d.ts +2 -19
- package/dist/belief-state/index.js +30 -31
- package/dist/belief-state/index.js.map +1 -1
- package/dist/benchmarks/index.d.ts +5 -8
- package/dist/benchmarks/index.js +12 -11
- package/dist/builder-eval/index.js +1 -1
- package/dist/campaign/index.d.ts +30 -39
- package/dist/campaign/index.js +11 -10
- package/dist/{chunk-NKAGIDE2.js → chunk-2QU3YOPR.js} +15 -274
- package/dist/chunk-2QU3YOPR.js.map +1 -0
- package/dist/{chunk-EJGRPCO3.js → chunk-3OCR4R5I.js} +245 -134
- package/dist/chunk-3OCR4R5I.js.map +1 -0
- package/dist/{chunk-2JX3CFMB.js → chunk-56TAVBOK.js} +5 -2
- package/dist/chunk-56TAVBOK.js.map +1 -0
- package/dist/{chunk-DPUHNQLN.js → chunk-7FO3TNPI.js} +2 -2
- package/dist/{chunk-DJKY2TSY.js → chunk-BSO5JDQH.js} +27 -120
- package/dist/chunk-BSO5JDQH.js.map +1 -0
- package/dist/{chunk-EZJEIH2R.js → chunk-C6LXANRU.js} +11 -20
- package/dist/chunk-C6LXANRU.js.map +1 -0
- package/dist/{chunk-ZUUWPZCV.js → chunk-DODXQREJ.js} +4 -4
- package/dist/{chunk-2MKQIFS4.js → chunk-E7QXT7SX.js} +2 -2
- package/dist/{chunk-NYLOYM6N.js → chunk-EG66UGL4.js} +37 -28
- package/dist/chunk-EG66UGL4.js.map +1 -0
- package/dist/{chunk-P5W7RQKK.js → chunk-FXTVJPYD.js} +2 -2
- package/dist/{chunk-VZSRQ272.js → chunk-G7MGMCZD.js} +6 -2
- package/dist/chunk-G7MGMCZD.js.map +1 -0
- package/dist/{chunk-IHQDPH7D.js → chunk-H23X7XKK.js} +85 -75
- package/dist/chunk-H23X7XKK.js.map +1 -0
- package/dist/{chunk-VBQ3CRKH.js → chunk-HPWUNB47.js} +4 -6
- package/dist/chunk-HPWUNB47.js.map +1 -0
- package/dist/{chunk-XPRT64IE.js → chunk-IYCLP2N2.js} +3 -3
- package/dist/chunk-IYCLP2N2.js.map +1 -0
- package/dist/{chunk-XDWDC2MP.js → chunk-JQSF5DQT.js} +11 -5
- package/dist/chunk-JQSF5DQT.js.map +1 -0
- package/dist/{chunk-NACAGYSY.js → chunk-M4YBQKIJ.js} +11 -11
- package/dist/chunk-M4YBQKIJ.js.map +1 -0
- package/dist/{chunk-YJBNWCAA.js → chunk-NY44NC4A.js} +3 -3
- package/dist/chunk-OIUOT4QD.js +44 -0
- package/dist/chunk-OIUOT4QD.js.map +1 -0
- package/dist/chunk-OWN5NPMC.js +152 -0
- package/dist/chunk-OWN5NPMC.js.map +1 -0
- package/dist/chunk-PC5DOSM7.js +579 -0
- package/dist/chunk-PC5DOSM7.js.map +1 -0
- package/dist/{chunk-UB2LOJ6Q.js → chunk-QB6BDBP2.js} +23 -20
- package/dist/chunk-QB6BDBP2.js.map +1 -0
- package/dist/chunk-RXHCETDZ.js +536 -0
- package/dist/chunk-RXHCETDZ.js.map +1 -0
- package/dist/{chunk-PBE2LOSS.js → chunk-SFLLL76A.js} +7 -7
- package/dist/chunk-SFLLL76A.js.map +1 -0
- package/dist/{chunk-VGRCHJON.js → chunk-T6RLYGAD.js} +3 -8
- package/dist/chunk-T6RLYGAD.js.map +1 -0
- package/dist/{chunk-VLOATJQ2.js → chunk-TJVT4QFF.js} +21 -18
- package/dist/chunk-TJVT4QFF.js.map +1 -0
- package/dist/{chunk-S5YLIBFX.js → chunk-TQ7LNKZ3.js} +2 -2
- package/dist/{chunk-EOSZT7PL.js → chunk-U4L7JRPZ.js} +2 -297
- package/dist/chunk-U4L7JRPZ.js.map +1 -0
- package/dist/chunk-U4PHLT2N.js +419 -0
- package/dist/chunk-U4PHLT2N.js.map +1 -0
- package/dist/{chunk-WS3NZZQQ.js → chunk-VCZ5FQYW.js} +3 -4
- package/dist/chunk-VCZ5FQYW.js.map +1 -0
- package/dist/{chunk-BYT7ELPS.js → chunk-WVATSFCP.js} +2 -2
- package/dist/{chunk-TSN7JT6D.js → chunk-X4YIBDER.js} +21 -5
- package/dist/{chunk-TSN7JT6D.js.map → chunk-X4YIBDER.js.map} +1 -1
- package/dist/{chunk-TBL77AUT.js → chunk-YQN4ICPP.js} +5 -5
- package/dist/{chunk-MHELPNRP.js → chunk-ZHTZ4EYI.js} +1 -1
- package/dist/chunk-ZHTZ4EYI.js.map +1 -0
- package/dist/cli.js +6 -5
- package/dist/cli.js.map +1 -1
- package/dist/contract/index.d.ts +47 -87
- package/dist/contract/index.js +14 -13
- package/dist/contract/index.js.map +1 -1
- package/dist/control.js +3 -2
- package/dist/fuzz.js +3 -2
- package/dist/fuzz.js.map +1 -1
- package/dist/index.d.ts +659 -203
- package/dist/index.js +145 -117
- package/dist/index.js.map +1 -1
- package/dist/meta-eval/index.js +2 -2
- package/dist/multishot/index.d.ts +3 -4
- package/dist/multishot/index.js.map +1 -1
- package/dist/openapi.json +1 -1
- package/dist/pipelines/index.js +5 -5
- package/dist/reporting.d.ts +14 -0
- package/dist/reporting.js +7 -6
- package/dist/rl.d.ts +652 -82
- package/dist/rl.js +415 -171
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +1071 -32
- package/dist/rollout/index.js +68 -10
- package/dist/run-campaign-OJJ7CZF4.js +18 -0
- package/dist/supervisor-run/index.d.ts +114 -4
- package/dist/supervisor-run/index.js +4 -3
- package/dist/traces.d.ts +1 -1
- package/dist/traces.js +6 -5
- package/dist/wire/index.d.ts +10 -11
- package/dist/wire/index.js +3 -3
- package/docs/feature-guide.md +1 -1
- package/docs/rollout.md +116 -2
- package/package.json +4 -4
- package/dist/chunk-2JX3CFMB.js.map +0 -1
- package/dist/chunk-DJKY2TSY.js.map +0 -1
- package/dist/chunk-EJGRPCO3.js.map +0 -1
- package/dist/chunk-EOSZT7PL.js.map +0 -1
- package/dist/chunk-EZJEIH2R.js.map +0 -1
- package/dist/chunk-IHQDPH7D.js.map +0 -1
- package/dist/chunk-MHELPNRP.js.map +0 -1
- package/dist/chunk-NACAGYSY.js.map +0 -1
- package/dist/chunk-NKAGIDE2.js.map +0 -1
- package/dist/chunk-NYLOYM6N.js.map +0 -1
- package/dist/chunk-PBE2LOSS.js.map +0 -1
- package/dist/chunk-TT4KNT67.js +0 -124
- package/dist/chunk-TT4KNT67.js.map +0 -1
- package/dist/chunk-UB2LOJ6Q.js.map +0 -1
- package/dist/chunk-UWZZKKU7.js +0 -237
- package/dist/chunk-UWZZKKU7.js.map +0 -1
- package/dist/chunk-VBQ3CRKH.js.map +0 -1
- package/dist/chunk-VGRCHJON.js.map +0 -1
- package/dist/chunk-VLOATJQ2.js.map +0 -1
- package/dist/chunk-VZSRQ272.js.map +0 -1
- package/dist/chunk-WS3NZZQQ.js.map +0 -1
- package/dist/chunk-XDWDC2MP.js.map +0 -1
- package/dist/chunk-XPRT64IE.js.map +0 -1
- package/dist/run-campaign-ISHFZ7FJ.js +0 -17
- /package/dist/{chunk-DPUHNQLN.js.map → chunk-7FO3TNPI.js.map} +0 -0
- /package/dist/{chunk-ZUUWPZCV.js.map → chunk-DODXQREJ.js.map} +0 -0
- /package/dist/{chunk-2MKQIFS4.js.map → chunk-E7QXT7SX.js.map} +0 -0
- /package/dist/{chunk-P5W7RQKK.js.map → chunk-FXTVJPYD.js.map} +0 -0
- /package/dist/{chunk-YJBNWCAA.js.map → chunk-NY44NC4A.js.map} +0 -0
- /package/dist/{chunk-S5YLIBFX.js.map → chunk-TQ7LNKZ3.js.map} +0 -0
- /package/dist/{chunk-BYT7ELPS.js.map → chunk-WVATSFCP.js.map} +0 -0
- /package/dist/{chunk-TBL77AUT.js.map → chunk-YQN4ICPP.js.map} +0 -0
- /package/dist/{run-campaign-ISHFZ7FJ.js.map → run-campaign-OJJ7CZF4.js.map} +0 -0
|
@@ -137,9 +137,8 @@ interface CostLedgerSummary {
|
|
|
137
137
|
* { baseUrl: 'https://router.tangle.tools/v1', apiKey: process.env.KEY },
|
|
138
138
|
* )
|
|
139
139
|
*
|
|
140
|
-
*
|
|
141
|
-
*
|
|
142
|
-
* that need free-form text use `callLlm` and parse output themselves.
|
|
140
|
+
* `createChatClient` wraps this implementation for provider-neutral package
|
|
141
|
+
* entry points. Direct callers can use `callLlm` or `callLlmJson`.
|
|
143
142
|
*/
|
|
144
143
|
|
|
145
144
|
interface LlmUsage {
|
|
@@ -290,7 +289,7 @@ interface DispatchContext {
|
|
|
290
289
|
/** The canonical judge verdict shape — one declaration, shared by campaign
|
|
291
290
|
* judges and the multishot judge runner (which re-exports this type).
|
|
292
291
|
*
|
|
293
|
-
* Scale is PRODUCER-DEFINED: campaign convention is [0,1]; the
|
|
292
|
+
* Scale is PRODUCER-DEFINED: campaign convention is [0,1]; the
|
|
294
293
|
* multishot runner emits 0-10. Cross-scale comparison must go through
|
|
295
294
|
* `detectScale` (src/campaign/gates/statistical-heldout.ts, used by
|
|
296
295
|
* promotion-policy) — never renormalize a producer's values in place, as
|
|
@@ -470,8 +469,6 @@ interface CampaignAggregates {
|
|
|
470
469
|
byScenario: Record<string, ScenarioAggregate>;
|
|
471
470
|
/** Canonical campaign accounting, including worker and judge calls. */
|
|
472
471
|
cost: CostLedgerSummary;
|
|
473
|
-
/** Compatibility alias of `cost.totalCostUsd`. */
|
|
474
|
-
totalCostUsd: number;
|
|
475
472
|
/** Cells whose dispatch completed, including cells whose later judge failed. */
|
|
476
473
|
cellsExecuted: number;
|
|
477
474
|
cellsSkipped: number;
|
|
@@ -482,7 +479,7 @@ interface CampaignAggregates {
|
|
|
482
479
|
cellsDispatchFailed?: number;
|
|
483
480
|
/** Present on results that record failure stages. */
|
|
484
481
|
cellsJudgeFailed?: number;
|
|
485
|
-
/**
|
|
482
|
+
/** Failures whose stage could not be classified. */
|
|
486
483
|
cellsUnclassifiedFailed?: number;
|
|
487
484
|
}
|
|
488
485
|
interface CampaignResult<TArtifact = unknown, TScenario extends Scenario = Scenario> {
|
|
@@ -734,7 +731,7 @@ interface CampaignStorage {
|
|
|
734
731
|
write(path: string, content: string | Uint8Array): void;
|
|
735
732
|
/** Append only when the current UTF-8 byte length matches `expectedBytes`.
|
|
736
733
|
* Returns the new length, or undefined when another writer won. */
|
|
737
|
-
append
|
|
734
|
+
append(path: string, content: string, expectedBytes: number): number | undefined;
|
|
738
735
|
}
|
|
739
736
|
|
|
740
737
|
interface BenchmarkRunOptions<TPayload = unknown, TArtifact = string> {
|
package/dist/benchmarks/index.js
CHANGED
|
@@ -16,24 +16,25 @@ import {
|
|
|
16
16
|
routing_exports,
|
|
17
17
|
runBenchmarkAdapter,
|
|
18
18
|
summarizeBenchmarkCampaign
|
|
19
|
-
} from "../chunk-
|
|
20
|
-
import "../chunk-
|
|
21
|
-
import "../chunk-
|
|
22
|
-
import "../chunk-
|
|
19
|
+
} from "../chunk-IYCLP2N2.js";
|
|
20
|
+
import "../chunk-QB6BDBP2.js";
|
|
21
|
+
import "../chunk-2QU3YOPR.js";
|
|
22
|
+
import "../chunk-C6LXANRU.js";
|
|
23
23
|
import "../chunk-WGXIEX7P.js";
|
|
24
|
-
import "../chunk-
|
|
25
|
-
import "../chunk-
|
|
26
|
-
import "../chunk-
|
|
27
|
-
import "../chunk-
|
|
28
|
-
import "../chunk-
|
|
29
|
-
import "../chunk-
|
|
24
|
+
import "../chunk-EG66UGL4.js";
|
|
25
|
+
import "../chunk-E7QXT7SX.js";
|
|
26
|
+
import "../chunk-SFLLL76A.js";
|
|
27
|
+
import "../chunk-7FO3TNPI.js";
|
|
28
|
+
import "../chunk-ZHTZ4EYI.js";
|
|
29
|
+
import "../chunk-VCZ5FQYW.js";
|
|
30
30
|
import "../chunk-VI2UW6B6.js";
|
|
31
31
|
import "../chunk-5DTSBUL2.js";
|
|
32
32
|
import "../chunk-GGE4NNQT.js";
|
|
33
33
|
import "../chunk-P6FYH6K4.js";
|
|
34
34
|
import "../chunk-PC4UYEBM.js";
|
|
35
|
-
import "../chunk-
|
|
35
|
+
import "../chunk-56TAVBOK.js";
|
|
36
36
|
import "../chunk-MA6HLL3S.js";
|
|
37
|
+
import "../chunk-OIUOT4QD.js";
|
|
37
38
|
import "../chunk-ONWEPEDO.js";
|
|
38
39
|
import "../chunk-K4DBDHLK.js";
|
|
39
40
|
import "../chunk-PZ5AY32C.js";
|
package/dist/campaign/index.d.ts
CHANGED
|
@@ -249,9 +249,8 @@ type CostLedgerHandle = Pick<CostLedger, Exclude<keyof CostLedger, 'listPending'
|
|
|
249
249
|
* { baseUrl: 'https://router.tangle.tools/v1', apiKey: process.env.KEY },
|
|
250
250
|
* )
|
|
251
251
|
*
|
|
252
|
-
*
|
|
253
|
-
*
|
|
254
|
-
* that need free-form text use `callLlm` and parse output themselves.
|
|
252
|
+
* `createChatClient` wraps this implementation for provider-neutral package
|
|
253
|
+
* entry points. Direct callers can use `callLlm` or `callLlmJson`.
|
|
255
254
|
*/
|
|
256
255
|
|
|
257
256
|
interface LlmMessage {
|
|
@@ -335,42 +334,27 @@ interface LlmCallResult {
|
|
|
335
334
|
type LlmCallMetadata = Pick<LlmCallResult, 'usage' | 'costUsd' | 'model' | 'durationMs'>;
|
|
336
335
|
|
|
337
336
|
/**
|
|
338
|
-
*
|
|
339
|
-
*
|
|
340
|
-
*
|
|
341
|
-
*
|
|
342
|
-
*
|
|
343
|
-
* couples analyst code to runtime concerns (cli-bridge vs router vs
|
|
344
|
-
* sandbox-sdk) it shouldn't know about.
|
|
345
|
-
*
|
|
346
|
-
* `ChatClient` is one interface every analyst takes via `AnalystContext.chat`.
|
|
347
|
-
* The operator decides at the registry boundary which transport binds
|
|
348
|
-
* to it. Analyst code stays transport-agnostic; swapping production
|
|
349
|
-
* (sandbox-sdk) for local dev (cli-bridge) or tests (mock) is a one-
|
|
350
|
-
* line factory call.
|
|
351
|
-
*
|
|
352
|
-
* Designed to coexist: existing `LlmClient` callers and existing
|
|
353
|
-
* `TCloud`-based judges keep working untouched. New analyst code uses
|
|
354
|
-
* `ChatClient`. When old call sites migrate, they pick up budgeting,
|
|
355
|
-
* cancellation, and unified telemetry for free.
|
|
337
|
+
* Provider-neutral chat contract for every model call made by agent-eval.
|
|
338
|
+
*
|
|
339
|
+
* Callers choose the transport at the package boundary with `createChatClient`.
|
|
340
|
+
* Evaluation code receives canonical requests and results without importing a
|
|
341
|
+
* provider SDK.
|
|
356
342
|
*/
|
|
357
343
|
|
|
358
344
|
/**
|
|
359
|
-
* Unified chat interface
|
|
360
|
-
* compatible mental model stays. Two methods: a one-shot `chat()` and
|
|
361
|
-
* an `streamChat()` for future agentic loops (not yet exposed).
|
|
345
|
+
* Unified chat interface using the package's canonical LLM request and result.
|
|
362
346
|
*/
|
|
363
347
|
interface ChatClient {
|
|
364
|
-
/** Display name of the bound transport
|
|
348
|
+
/** Display name of the bound transport, included in telemetry. */
|
|
365
349
|
readonly transport: ChatTransport;
|
|
366
|
-
/** Default model when caller omits
|
|
350
|
+
/** Default model when the caller omits one. */
|
|
367
351
|
readonly defaultModel?: string;
|
|
368
352
|
/** Total provider attempts this transport can make for one chat call. */
|
|
369
353
|
readonly maximumAttempts?: number;
|
|
370
354
|
/** Implementations must enforce `req.maxTokens` when it is present. */
|
|
371
355
|
chat(req: ChatRequest, opts?: ChatCallOpts): Promise<ChatResponse>;
|
|
372
356
|
}
|
|
373
|
-
type ChatTransport = 'router' | 'sandbox-sdk' | 'cli-bridge' | 'direct-provider' | 'mock';
|
|
357
|
+
type ChatTransport = 'router' | 'sandbox-sdk' | 'cli-bridge' | 'direct-provider' | 'custom' | 'mock';
|
|
374
358
|
interface ChatRequest extends Omit<LlmCallRequest, 'model'> {
|
|
375
359
|
/** Optional — falls back to ChatClient.defaultModel. */
|
|
376
360
|
model?: string;
|
|
@@ -738,7 +722,7 @@ interface JudgeConfig<TArtifact, TScenario extends Scenario = Scenario> {
|
|
|
738
722
|
/** The canonical judge verdict shape — one declaration, shared by campaign
|
|
739
723
|
* judges and the multishot judge runner (which re-exports this type).
|
|
740
724
|
*
|
|
741
|
-
* Scale is PRODUCER-DEFINED: campaign convention is [0,1]; the
|
|
725
|
+
* Scale is PRODUCER-DEFINED: campaign convention is [0,1]; the
|
|
742
726
|
* multishot runner emits 0-10. Cross-scale comparison must go through
|
|
743
727
|
* `detectScale` (src/campaign/gates/statistical-heldout.ts, used by
|
|
744
728
|
* promotion-policy) — never renormalize a producer's values in place, as
|
|
@@ -961,9 +945,6 @@ interface SurfaceProposer<TFindings = unknown> {
|
|
|
961
945
|
reason?: string;
|
|
962
946
|
};
|
|
963
947
|
}
|
|
964
|
-
/** Optional vocabulary alias. The loop is the optimizer; this object is the
|
|
965
|
-
* proposer inside that loop. */
|
|
966
|
-
type OptimizationProposer<TFindings = unknown> = SurfaceProposer<TFindings>;
|
|
967
948
|
interface OptimizerConfigBase {
|
|
968
949
|
populationSize: number;
|
|
969
950
|
maxGenerations: number;
|
|
@@ -1243,8 +1224,6 @@ interface CampaignAggregates {
|
|
|
1243
1224
|
byScenario: Record<string, ScenarioAggregate>;
|
|
1244
1225
|
/** Canonical campaign accounting, including worker and judge calls. */
|
|
1245
1226
|
cost: CostLedgerSummary;
|
|
1246
|
-
/** Compatibility alias of `cost.totalCostUsd`. */
|
|
1247
|
-
totalCostUsd: number;
|
|
1248
1227
|
/** Cells whose dispatch completed, including cells whose later judge failed. */
|
|
1249
1228
|
cellsExecuted: number;
|
|
1250
1229
|
cellsSkipped: number;
|
|
@@ -1255,7 +1234,7 @@ interface CampaignAggregates {
|
|
|
1255
1234
|
cellsDispatchFailed?: number;
|
|
1256
1235
|
/** Present on results that record failure stages. */
|
|
1257
1236
|
cellsJudgeFailed?: number;
|
|
1258
|
-
/**
|
|
1237
|
+
/** Failures whose stage could not be classified. */
|
|
1259
1238
|
cellsUnclassifiedFailed?: number;
|
|
1260
1239
|
}
|
|
1261
1240
|
interface CampaignResult<TArtifact = unknown, TScenario extends Scenario = Scenario> {
|
|
@@ -2313,7 +2292,7 @@ interface CampaignStorage {
|
|
|
2313
2292
|
write(path: string, content: string | Uint8Array): void;
|
|
2314
2293
|
/** Append only when the current UTF-8 byte length matches `expectedBytes`.
|
|
2315
2294
|
* Returns the new length, or undefined when another writer won. */
|
|
2316
|
-
append
|
|
2295
|
+
append(path: string, content: string, expectedBytes: number): number | undefined;
|
|
2317
2296
|
}
|
|
2318
2297
|
/** Node-filesystem storage — the default. Lazily requires `node:fs` so the
|
|
2319
2298
|
* module imports cleanly in non-Node runtimes (where the caller passes
|
|
@@ -2455,9 +2434,9 @@ interface RunCampaignOptions<TScenario extends Scenario, TArtifact> {
|
|
|
2455
2434
|
}) => string | undefined;
|
|
2456
2435
|
}
|
|
2457
2436
|
/** Durable `<cell>/failure-receipt.json` written before a failed cell can
|
|
2458
|
-
* trigger campaign-wide cancellation. The cell
|
|
2459
|
-
*
|
|
2460
|
-
*
|
|
2437
|
+
* trigger campaign-wide cancellation. The cell records dispatch measurements;
|
|
2438
|
+
* `cost` covers every settled agent and judge call attributed to this exact run
|
|
2439
|
+
* attempt. */
|
|
2461
2440
|
interface CampaignCellFailureReceipt<TArtifact = unknown> {
|
|
2462
2441
|
schemaVersion: 1;
|
|
2463
2442
|
runAttemptId: string;
|
|
@@ -3110,6 +3089,18 @@ interface VerifiableRewardExtractionOptions {
|
|
|
3110
3089
|
* doesn't report one. Default `0.7`.
|
|
3111
3090
|
*/
|
|
3112
3091
|
judgeConfidenceFloor?: number;
|
|
3092
|
+
/**
|
|
3093
|
+
* Whether the anti-Goodhart realness gate applies. Default `true`, and the
|
|
3094
|
+
* default is the one every training path must keep.
|
|
3095
|
+
*
|
|
3096
|
+
* Set `false` ONLY for detection and analysis. `rl/reward-hacking.ts` does,
|
|
3097
|
+
* for the same reason it reads `observedScore` for its proxy: it measures the
|
|
3098
|
+
* DIVERGENCE between the judge signal and the deterministic one, and a
|
|
3099
|
+
* deterministic reward that another gate already forced to 0 manufactures
|
|
3100
|
+
* exactly that divergence on exactly the gamed population. The detector would
|
|
3101
|
+
* then be re-reporting a verdict it was supposed to reach independently.
|
|
3102
|
+
*/
|
|
3103
|
+
applyRealnessGate?: boolean;
|
|
3113
3104
|
}
|
|
3114
3105
|
|
|
3115
3106
|
/**
|
|
@@ -6387,4 +6378,4 @@ declare function verifyCodeSurface(surface: CodeSurface, worktreeDir?: string):
|
|
|
6387
6378
|
* identity against the checkout at `worktreeRef`. */
|
|
6388
6379
|
declare function resolveWorktreePath(surface: CodeSurface, worktreeDir?: string): string;
|
|
6389
6380
|
|
|
6390
|
-
export { type AnalystArtifact, type AnalystScenario, type AnalyzeCrossSurfaceInteractionsInput, type AxisEvidence, type AxisVerdict, type BuildAnalystSurfaceDispatchOptions, type BuildEvidenceVectorOptions, type BuildLoopProvenanceArgs, type CampaignAggregates, type CampaignArtifactWriter, type CampaignBreakdown, type CampaignCellFailureReceipt, type CampaignCellResult, type CampaignCostMeter, type CampaignResult, type CampaignRunPlan, type CampaignRunPlanCell, type CampaignScenarioIdentity, type CampaignStorage, type CampaignTokenUsage, type CampaignTraceWriter, type CodeSurface, type CodeSurfaceVerification, type CompareOptimizationMethodsOptions, type ComparisonCost, type ComponentSurface, type CostLedgerHandle, type CrossSurfaceAdditionDecision, type CrossSurfaceAdditionRejectionReason, type CrossSurfaceAttemptCompleteness, type CrossSurfaceBestSingleSelection, type CrossSurfaceBootstrapPolicy, type CrossSurfaceCandidate, type CrossSurfaceCandidateComparison, type CrossSurfaceCandidateEvidence, type CrossSurfaceCandidateOutcome, type CrossSurfaceCandidateSummary, type CrossSurfaceComponent, type CrossSurfaceComponentEvidence, type CrossSurfaceCompositionStep, type CrossSurfaceDistribution, type CrossSurfaceEligibility, type CrossSurfaceEvidenceBreakdown, type CrossSurfaceIneligibilityReason, type CrossSurfaceInteractionAwareSelection, type CrossSurfaceInteractionEffect, type CrossSurfaceInteractionPath, type CrossSurfaceInteractionReport, type CrossSurfaceInteractionTask, type CrossSurfaceNaiveStackSelection, type CrossSurfacePairCompatibility, type CrossSurfacePairEvidence, type CrossSurfacePairIncompatibilityReason, type CrossSurfacePairwiseEntry, type CrossSurfaceRankedSingle, type CrossSurfaceRelativeCost, type CrossSurfaceSelectionPolicy, type CrossSurfaceSelections, type CrossSurfaceTaskRow, type DefaultProductionGateCheck, type DefaultProductionGateOptions, type DefaultProductionRewardHackingOptions, type DimensionRegression, type DiscriminationScore, type DispatchContext, type DispatchFn, type EmitLoopProvenanceArgs, type EmitLoopProvenanceResult, type EvalFixture, type EvalFixtureFile, type EvalFixtureLoadOptions, type EvalFixtureRunPlan, type EvalFixtureScenario, type EvalFixtureValidationMode, type EvidenceVector, type ExternalOptimizationExample, type ExternalTextEvaluationResponse, type ExternalTextOptimizationMethodConfig, type ExternalTextOptimizerContext, type ExternalTextOptimizerResult, type FailureModeRecallJudgeOptions, FileSearchLedger, FsLabeledScenarioStore, type FsLabeledScenarioStoreOptions, type Gate, type GateCheckStatus, type GateContext, type GateContribution, type GateDecision, type GateResult, type GenerationCandidate, type GenerationRecord, type GepaAdaptiveEngineRun, type GepaEngineOptions, type GepaEngineRun, type GepaOptimizationMethodConfig, type GepaOptimizationRecipe, type GepaRunnerCommand, type GitWorktreeAdapterOptions, type HeldOutGateOptions, type HeldoutSignificance, type HeldoutSignificanceOptions, type JudgeAggregate, type JudgeConfig, type JudgeDimension, type JudgeScore, type LabelTrust, type LabeledScenarioRecord, type LabeledScenarioSampleArgs, type LabeledScenarioSource, type LabeledScenarioStore, LabeledScenarioStoreError, type LabeledScenarioWrite, type LlmJudgeDimension, type LlmJudgeOptions, type LoadEvalFixtureScenariosOptions, type LoopProvenanceArgsFromResult, type LoopProvenanceBackend, type LoopProvenanceCandidate, type LoopProvenanceEvidence, type LoopProvenanceOptimizationMethod, type LoopProvenanceRecord, type MutableSurface, type NeutralizationGateOptions, type ObjectiveSource, type OpenAICompatibleOptimizerModel, type OpenAutoPrOptions, type OpenAutoPrResult, type OpenSearchLedgerOptions, type OptimizationMethod, type OptimizationMethodComparison, type OptimizationMethodInput, type OptimizationMethodPairwise, type OptimizationMethodProvenance, type OptimizationMethodResult, type OptimizationMethodRunOptions, type OptimizationMethodScore, type OptimizationPackageSource, type
|
|
6381
|
+
export { type AnalystArtifact, type AnalystScenario, type AnalyzeCrossSurfaceInteractionsInput, type AxisEvidence, type AxisVerdict, type BuildAnalystSurfaceDispatchOptions, type BuildEvidenceVectorOptions, type BuildLoopProvenanceArgs, type CampaignAggregates, type CampaignArtifactWriter, type CampaignBreakdown, type CampaignCellFailureReceipt, type CampaignCellResult, type CampaignCostMeter, type CampaignResult, type CampaignRunPlan, type CampaignRunPlanCell, type CampaignScenarioIdentity, type CampaignStorage, type CampaignTokenUsage, type CampaignTraceWriter, type CodeSurface, type CodeSurfaceVerification, type CompareOptimizationMethodsOptions, type ComparisonCost, type ComponentSurface, type CostLedgerHandle, type CrossSurfaceAdditionDecision, type CrossSurfaceAdditionRejectionReason, type CrossSurfaceAttemptCompleteness, type CrossSurfaceBestSingleSelection, type CrossSurfaceBootstrapPolicy, type CrossSurfaceCandidate, type CrossSurfaceCandidateComparison, type CrossSurfaceCandidateEvidence, type CrossSurfaceCandidateOutcome, type CrossSurfaceCandidateSummary, type CrossSurfaceComponent, type CrossSurfaceComponentEvidence, type CrossSurfaceCompositionStep, type CrossSurfaceDistribution, type CrossSurfaceEligibility, type CrossSurfaceEvidenceBreakdown, type CrossSurfaceIneligibilityReason, type CrossSurfaceInteractionAwareSelection, type CrossSurfaceInteractionEffect, type CrossSurfaceInteractionPath, type CrossSurfaceInteractionReport, type CrossSurfaceInteractionTask, type CrossSurfaceNaiveStackSelection, type CrossSurfacePairCompatibility, type CrossSurfacePairEvidence, type CrossSurfacePairIncompatibilityReason, type CrossSurfacePairwiseEntry, type CrossSurfaceRankedSingle, type CrossSurfaceRelativeCost, type CrossSurfaceSelectionPolicy, type CrossSurfaceSelections, type CrossSurfaceTaskRow, type DefaultProductionGateCheck, type DefaultProductionGateOptions, type DefaultProductionRewardHackingOptions, type DimensionRegression, type DiscriminationScore, type DispatchContext, type DispatchFn, type EmitLoopProvenanceArgs, type EmitLoopProvenanceResult, type EvalFixture, type EvalFixtureFile, type EvalFixtureLoadOptions, type EvalFixtureRunPlan, type EvalFixtureScenario, type EvalFixtureValidationMode, type EvidenceVector, type ExternalOptimizationExample, type ExternalTextEvaluationResponse, type ExternalTextOptimizationMethodConfig, type ExternalTextOptimizerContext, type ExternalTextOptimizerResult, type FailureModeRecallJudgeOptions, FileSearchLedger, FsLabeledScenarioStore, type FsLabeledScenarioStoreOptions, type Gate, type GateCheckStatus, type GateContext, type GateContribution, type GateDecision, type GateResult, type GenerationCandidate, type GenerationRecord, type GepaAdaptiveEngineRun, type GepaEngineOptions, type GepaEngineRun, type GepaOptimizationMethodConfig, type GepaOptimizationRecipe, type GepaRunnerCommand, type GitWorktreeAdapterOptions, type HeldOutGateOptions, type HeldoutSignificance, type HeldoutSignificanceOptions, type JudgeAggregate, type JudgeConfig, type JudgeDimension, type JudgeScore, type LabelTrust, type LabeledScenarioRecord, type LabeledScenarioSampleArgs, type LabeledScenarioSource, type LabeledScenarioStore, LabeledScenarioStoreError, type LabeledScenarioWrite, type LlmJudgeDimension, type LlmJudgeOptions, type LoadEvalFixtureScenariosOptions, type LoopProvenanceArgsFromResult, type LoopProvenanceBackend, type LoopProvenanceCandidate, type LoopProvenanceEvidence, type LoopProvenanceOptimizationMethod, type LoopProvenanceRecord, type MutableSurface, type NeutralizationGateOptions, type ObjectiveSource, type OpenAICompatibleOptimizerModel, type OpenAutoPrOptions, type OpenAutoPrResult, type OpenSearchLedgerOptions, type OptimizationMethod, type OptimizationMethodComparison, type OptimizationMethodInput, type OptimizationMethodPairwise, type OptimizationMethodProvenance, type OptimizationMethodResult, type OptimizationMethodRunOptions, type OptimizationMethodScore, type OptimizationPackageSource, type OptimizationTokenUsage, type OptimizerConfig, type OptimizerModelBudget, type PairedHoldout, type ParetoParent, type ParetoSignificanceGateOptions, type PendingCostCallView, type PlanCampaignRunOptions, type PlanEvalFixtureRunOptions, type PlaybackContext, type PlaybackDriver, type PlaybackStep, type PowerPreflight, type PowerPreflightOptions, type PremeasuredOptimizationBaseline, type ProfileDispatchFn, ProfileMatrixError, type ProfileSummary, type PromotionObjective, type PromotionPolicy, type ProposalTrackContext, type ProposeContext, type ProposedCandidate, type RedactionStatus, type ReferenceEquivalenceJudgeOptions, type ReferenceEquivalenceScenario, type RolloutArgumentDiff, type RolloutArgumentDiffOptions, type RolloutCall, type RunCampaignOptions, type RunEvalOptions, type RunImprovementLoopOptions, type RunImprovementLoopResult, type RunOptimizationOptions, type RunOptimizationResult, type RunProfileMatrixOptions, type RunProfileMatrixResult, SEARCH_LEDGER_SCHEMA, type Scenario, type ScenarioAggregate, type ScenarioRollup, type ScenarioSignal, type ScoreboardRenderOptions, type ScoreboardRow, type ScoreboardSummary, type ScoredRollout, type ScoredSurfaceOutcome, type SearchAccountingAudit, type SearchArtifactRef, type SearchAttemptAccounting, type SearchCandidateDecidedEvent, type SearchCandidateLineage, type SearchCandidateRegisteredEvent, type SearchCandidateSlot, type SearchCandidateSlotClosedEvent, type SearchCandidateSurface, type SearchCompletedEvent, type SearchCostAccounting, type SearchFailureReason, type SearchLedger, type SearchLedgerAppendResult, SearchLedgerConflictError, type SearchLedgerEntry, SearchLedgerError, type SearchLedgerEvent, type SearchLedgerHash, SearchLedgerIntegrityError, type SearchLedgerReplay, type SearchModelIdentity, type SearchOperationKind, type SearchOperationRecordedEvent, type SearchPlan, type SearchPlannedEvent, type SearchPlannedOperation, type SearchPlannedTask, type SearchSourceRef, type SearchSurfaceEffect, type SearchSurfaceEvidence, type SearchSurfaceKind, type SearchTaskAttemptedEvent, type SearchTaskOutcome, type SearchTokenAccounting, type SequentialDecideFn, type SequentialDecideOptions, type SequentialDecision, type SequentialObservation, type SequentialPairedGate, type SequentialPairedGateOptions, type SessionScript, type SingleRunLock, type SingleRunLockOptions, type SkillOptOptimizationMethodConfig, type SkillOptRunnerCommand, type SkillOptTrainerConfig, type SurfaceProposer, type TraceSpan, type TransientFailureOptions, type UngroundedLiteralReport, type UserStory, type UserStoryVerdict, type Worktree, type WorktreeAdapter, WorktreeAdapterError, acquireSingleRunLock, analyzeCrossSurfaceInteractions, assertCampaignDesign, assertCampaignSplitIdentity, assertCodeSurfaceIdentity, assertComponentSurface, buildAnalystSurfaceDispatch, buildEvidenceVector, buildLoopProvenanceRecord, campaignBreakdown, campaignMeanComposite, campaignMeasurementDigest, campaignScenarioIdentity, campaignSplitDigest, campaignSplitDigestFromIdentities, canonicalDigest, classifyUngroundedLiterals, codeSurfaceIdentityMaterial, compareOptimizationMethods, compareRankKeys, componentSurfaceIdentityMaterial, composeGate, costFromLedgerSummary, createReferenceEquivalenceJudge, createRunCostLedger, defaultProductionGate, detectScale, dimensionRegressions, discoverEvalFixtures, emitLoopProvenance, externalTextOptimizationMethod, failureModeRecallJudge, fsCampaignStorage, gepaOptimizationMethod, gitWorktreeAdapter, heldOutGate, heldoutSignificance, inMemoryCampaignStorage, isProposedCandidate, isTransientTransportFailure, labelTrustRank, llmJudge, loadEvalFixture, loadEvalFixtureScenarios, loopProvenanceArgsFromResult, loopProvenanceSpans, makePlaybackDispatch, neutralizationGate, neutralizeText, openAutoPr, openSearchLedger, optimizationTokenUsageFromSummary, pairHoldout, paretoPolicy, paretoSignificanceGate, planCampaignRun, planEvalFixtureRun, powerPreflight, provenanceRecordPath, provenanceSpansPath, renderScoreboardMarkdown, renderSurfaceDiff, resolveRunDir, resolveWorktreePath, rolloutArgumentDiff, runCampaign, runEval, runImprovementLoop, runOptimization, runProfileMatrix, scoreDiscrimination, scoreUserStory, scoreboardSummary, selectDiscriminative, sequentialDecide, sequentialPairedGate, skillOptOptimizationMethod, surfaceContentHash, surfaceHash, tangleTracesRoot, userStoryScoreboard, validateSearchLedgerEvent, verifyCodeSurface, verifyLoopProvenanceRecord };
|
package/dist/campaign/index.js
CHANGED
|
@@ -32,7 +32,7 @@ import {
|
|
|
32
32
|
userStoryScoreboard,
|
|
33
33
|
validateSearchLedgerEvent,
|
|
34
34
|
verifyCodeSurface
|
|
35
|
-
} from "../chunk-
|
|
35
|
+
} from "../chunk-QB6BDBP2.js";
|
|
36
36
|
import {
|
|
37
37
|
acquireSingleRunLock,
|
|
38
38
|
assertCodeSurfaceIdentity,
|
|
@@ -79,7 +79,7 @@ import {
|
|
|
79
79
|
surfaceContentHash,
|
|
80
80
|
surfaceHash,
|
|
81
81
|
verifyLoopProvenanceRecord
|
|
82
|
-
} from "../chunk-
|
|
82
|
+
} from "../chunk-2QU3YOPR.js";
|
|
83
83
|
import {
|
|
84
84
|
SearchLedgerConflictError,
|
|
85
85
|
SearchLedgerError,
|
|
@@ -96,21 +96,22 @@ import {
|
|
|
96
96
|
resolveRunDir,
|
|
97
97
|
runCampaign,
|
|
98
98
|
tangleTracesRoot
|
|
99
|
-
} from "../chunk-
|
|
99
|
+
} from "../chunk-C6LXANRU.js";
|
|
100
100
|
import "../chunk-WGXIEX7P.js";
|
|
101
|
-
import "../chunk-
|
|
102
|
-
import "../chunk-
|
|
103
|
-
import "../chunk-
|
|
104
|
-
import "../chunk-
|
|
105
|
-
import "../chunk-
|
|
106
|
-
import "../chunk-
|
|
101
|
+
import "../chunk-EG66UGL4.js";
|
|
102
|
+
import "../chunk-E7QXT7SX.js";
|
|
103
|
+
import "../chunk-SFLLL76A.js";
|
|
104
|
+
import "../chunk-7FO3TNPI.js";
|
|
105
|
+
import "../chunk-ZHTZ4EYI.js";
|
|
106
|
+
import "../chunk-VCZ5FQYW.js";
|
|
107
107
|
import "../chunk-VI2UW6B6.js";
|
|
108
108
|
import "../chunk-5DTSBUL2.js";
|
|
109
109
|
import "../chunk-GGE4NNQT.js";
|
|
110
110
|
import "../chunk-P6FYH6K4.js";
|
|
111
111
|
import "../chunk-PC4UYEBM.js";
|
|
112
|
-
import "../chunk-
|
|
112
|
+
import "../chunk-56TAVBOK.js";
|
|
113
113
|
import "../chunk-MA6HLL3S.js";
|
|
114
|
+
import "../chunk-OIUOT4QD.js";
|
|
114
115
|
import "../chunk-ONWEPEDO.js";
|
|
115
116
|
import "../chunk-K4DBDHLK.js";
|
|
116
117
|
import "../chunk-PZ5AY32C.js";
|