@tangle-network/agent-eval 0.136.0 → 0.137.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +38 -1
- package/README.md +4 -2
- package/dist/{agent-profile-cell-OhuTee9n.js → agent-profile-cell-CbfBm2g6.js} +2 -2
- package/dist/{agent-profile-cell-OhuTee9n.js.map → agent-profile-cell-CbfBm2g6.js.map} +1 -1
- package/dist/{agent-profile-cell-CCm3l2v2.d.ts → agent-profile-cell-Cw0PVwDr.d.ts} +2 -2
- package/dist/{agent-profile-cell-CCm3l2v2.d.ts.map → agent-profile-cell-Cw0PVwDr.d.ts.map} +1 -1
- package/dist/analyst/index.d.ts +139 -17
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +606 -4
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyze-runs-jjCmF8pU.js → analyze-runs-BScZvqMV.js} +6 -6
- package/dist/{analyze-runs-jjCmF8pU.js.map → analyze-runs-BScZvqMV.js.map} +1 -1
- package/dist/{analyze-runs-Cda5Xkj1.d.ts → analyze-runs-PVtnfjvA.d.ts} +6 -6
- package/dist/{analyze-runs-Cda5Xkj1.d.ts.map → analyze-runs-PVtnfjvA.d.ts.map} +1 -1
- package/dist/{baseline-BUeFcgrn.js → baseline-C-GocmIW.js} +2 -2
- package/dist/{baseline-BUeFcgrn.js.map → baseline-C-GocmIW.js.map} +1 -1
- package/dist/benchmark-CHX4orG7.d.ts +184 -0
- package/dist/benchmark-CHX4orG7.d.ts.map +1 -0
- package/dist/benchmark-YDrpumqB.js +414 -0
- package/dist/benchmark-YDrpumqB.js.map +1 -0
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +1 -1
- package/dist/{benchmarks-Dfgm9ts5.js → benchmarks-DCLkQOmc.js} +3 -3
- package/dist/{benchmarks-Dfgm9ts5.js.map → benchmarks-DCLkQOmc.js.map} +1 -1
- package/dist/builder-eval/index.js +2 -2
- package/dist/campaign/index.d.ts +6 -6
- package/dist/campaign/index.js +3 -3
- package/dist/{campaign-Dz8uQnhC.js → campaign-lgObcHFC.js} +212 -78
- package/dist/campaign-lgObcHFC.js.map +1 -0
- package/dist/cli.js +1 -1
- package/dist/client-C8L6h6Wf.d.ts +202 -0
- package/dist/client-C8L6h6Wf.d.ts.map +1 -0
- package/dist/completion-verifier-DSyRNVzU.d.ts +240 -0
- package/dist/completion-verifier-DSyRNVzU.d.ts.map +1 -0
- package/dist/contract/index.d.ts +10 -9
- package/dist/contract/index.d.ts.map +1 -1
- package/dist/contract/index.js +10 -10
- package/dist/control.d.ts +2 -2
- package/dist/control.js +1 -1
- package/dist/{cost-ledger-DHAjwNj7.js → cost-ledger-D-5_-dhi.js} +2 -2
- package/dist/{cost-ledger-DHAjwNj7.js.map → cost-ledger-D-5_-dhi.js.map} +1 -1
- package/dist/{cost-ledger-fGS_u_O1.d.ts → cost-ledger-D2o6JOrL.d.ts} +2 -2
- package/dist/{cost-ledger-fGS_u_O1.d.ts.map → cost-ledger-D2o6JOrL.d.ts.map} +1 -1
- package/dist/{dataset-BvtnC8Dc.d.ts → dataset-v_Y5902-.d.ts} +2 -2
- package/dist/{dataset-BvtnC8Dc.d.ts.map → dataset-v_Y5902-.d.ts.map} +1 -1
- package/dist/{default-registry-CHmdy2An.js → default-registry-CLXbRt0f.js} +119 -38
- package/dist/default-registry-CLXbRt0f.js.map +1 -0
- package/dist/{default-registry-Brxr728w.d.ts → default-registry-Dc5D_Loc.d.ts} +63 -137
- package/dist/default-registry-Dc5D_Loc.d.ts.map +1 -0
- package/dist/{errors-8YnH8WlF.js → errors-D-LKuDhb.js} +8 -2
- package/dist/errors-D-LKuDhb.js.map +1 -0
- package/dist/{errors-CEk209JS.d.ts → errors-DkfjIDvD.d.ts} +9 -3
- package/dist/errors-DkfjIDvD.d.ts.map +1 -0
- package/dist/{eval-campaign-Cc8WZJ6b.js → eval-campaign-CHqfLnff.js} +6 -6
- package/dist/{eval-campaign-Cc8WZJ6b.js.map → eval-campaign-CHqfLnff.js.map} +1 -1
- package/dist/{extract-usage-DIQpN-ww.js → extract-usage-p-56bh8q.js} +3 -3
- package/dist/{extract-usage-DIQpN-ww.js.map → extract-usage-p-56bh8q.js.map} +1 -1
- package/dist/{feedback-trajectory-CVaeREXV.d.ts → feedback-trajectory-N_F0PwHz.d.ts} +90 -3
- package/dist/feedback-trajectory-N_F0PwHz.d.ts.map +1 -0
- package/dist/fuzz.d.ts +1 -1
- package/dist/fuzz.js +2 -2
- package/dist/hosted/index.d.ts +3 -2
- package/dist/hosted/index.d.ts.map +1 -1
- package/dist/index-BipJlj-C.d.ts +316 -0
- package/dist/index-BipJlj-C.d.ts.map +1 -0
- package/dist/{index-CQsJcqch.d.ts → index-BnP1QJUv.d.ts} +5 -5
- package/dist/{index-CQsJcqch.d.ts.map → index-BnP1QJUv.d.ts.map} +1 -1
- package/dist/{index-B4Fjfo5U.d.ts → index-C-Pr4OWg.d.ts} +88 -317
- package/dist/index-C-Pr4OWg.d.ts.map +1 -0
- package/dist/{index-DuhJaaiH.d.ts → index-DEb46kc6.d.ts} +2 -2
- package/dist/{index-DuhJaaiH.d.ts.map → index-DEb46kc6.d.ts.map} +1 -1
- package/dist/{index-C2fkZhv_.d.ts → index-DRNl6g_N.d.ts} +3 -3
- package/dist/{index-C2fkZhv_.d.ts.map → index-DRNl6g_N.d.ts.map} +1 -1
- package/dist/{index-AbhwHp0V.d.ts → index-U3RHOShi.d.ts} +2 -2
- package/dist/{index-AbhwHp0V.d.ts.map → index-U3RHOShi.d.ts.map} +1 -1
- package/dist/index.d.ts +29 -70
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +507 -33
- package/dist/index.js.map +1 -1
- package/dist/{client-DcvgkaZi.d.ts → insight-report-B9ooYH_g.d.ts} +5 -203
- package/dist/insight-report-B9ooYH_g.d.ts.map +1 -0
- package/dist/integrity-CCXTftiL.js +1360 -0
- package/dist/integrity-CCXTftiL.js.map +1 -0
- package/dist/{integrity-rmVhXWA7.d.ts → integrity-CKxosZ5Z.d.ts} +3 -3
- package/dist/{integrity-rmVhXWA7.d.ts.map → integrity-CKxosZ5Z.d.ts.map} +1 -1
- package/dist/{integrity-BzRbCHzi.js → integrity-fdt8XPAv.js} +2 -2
- package/dist/{integrity-BzRbCHzi.js.map → integrity-fdt8XPAv.js.map} +1 -1
- package/dist/ledger-core/index.d.ts +1 -1
- package/dist/ledger-core/index.js +1 -1
- package/dist/{ledger-core-DAKFKRzi.js → ledger-core-t6sItivm.js} +85 -85
- package/dist/{ledger-core-DAKFKRzi.js.map → ledger-core-t6sItivm.js.map} +1 -1
- package/dist/{llm-client-DHx8pzyJ.js → llm-client-DKB25jV8.js} +3 -3
- package/dist/{llm-client-DHx8pzyJ.js.map → llm-client-DKB25jV8.js.map} +1 -1
- package/dist/meta-eval/index.d.ts +2 -2
- package/dist/meta-eval/index.js +3 -3
- package/dist/{mint-DyRUc9k6.js → mint-Ctwk079K.js} +4 -4
- package/dist/{mint-DyRUc9k6.js.map → mint-Ctwk079K.js.map} +1 -1
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/{paired-arms-BbFKrAU-.js → paired-arms-iZ08VFMN.js} +3 -3
- package/dist/{paired-arms-BbFKrAU-.js.map → paired-arms-iZ08VFMN.js.map} +1 -1
- package/dist/pipelines/index.js +2 -2
- package/dist/profile-cell.d.ts +1 -1
- package/dist/profile-cell.js +1 -1
- package/dist/{propose-review-control-SQ-n9-We.js → propose-review-control-DLXz4FCX.js} +2 -2
- package/dist/{propose-review-control-SQ-n9-We.js.map → propose-review-control-DLXz4FCX.js.map} +1 -1
- package/dist/registry-BdM7SuTr.d.ts +124 -0
- package/dist/registry-BdM7SuTr.d.ts.map +1 -0
- package/dist/{release-report-DooPguBc.js → release-report-B5XPBvAU.js} +4 -4
- package/dist/{release-report-DooPguBc.js.map → release-report-B5XPBvAU.js.map} +1 -1
- package/dist/{release-report-DpBxGGI1.d.ts → release-report-CofgVNZt.d.ts} +4 -4
- package/dist/{release-report-DpBxGGI1.d.ts.map → release-report-CofgVNZt.d.ts.map} +1 -1
- package/dist/{replay-C6wRg47C.js → replay-Bju0T8Ls.js} +248 -8
- package/dist/replay-Bju0T8Ls.js.map +1 -0
- package/dist/{replay-BRfMIs81.d.ts → replay-K8FaC0CB.d.ts} +227 -52
- package/dist/replay-K8FaC0CB.d.ts.map +1 -0
- package/dist/reporting.d.ts +4 -4
- package/dist/reporting.js +4 -4
- package/dist/{researcher-Doo95b50.d.ts → researcher-Da0Wj-bt.d.ts} +6 -7
- package/dist/researcher-Da0Wj-bt.d.ts.map +1 -0
- package/dist/{reward-hacking-D-QqXvg-.d.ts → reward-hacking-CQ3hTCO3.d.ts} +2 -2
- package/dist/{reward-hacking-D-QqXvg-.d.ts.map → reward-hacking-CQ3hTCO3.d.ts.map} +1 -1
- package/dist/{reward-hacking-a-kYs0-i.js → reward-hacking-GyN0kMd8.js} +3 -3
- package/dist/{reward-hacking-a-kYs0-i.js.map → reward-hacking-GyN0kMd8.js.map} +1 -1
- package/dist/rl.d.ts +6 -6
- package/dist/rl.js +9 -9
- package/dist/rollout/index.d.ts +1 -1
- package/dist/rollout/index.js +3 -3
- package/dist/{rollout-DLSUIWLu.js → rollout-DQFl0UXA.js} +2 -2
- package/dist/{rollout-DLSUIWLu.js.map → rollout-DQFl0UXA.js.map} +1 -1
- package/dist/{rubric-predictive-validity-BJf-8ejY.js → rubric-predictive-validity-BRR632r1.js} +2 -2
- package/dist/{rubric-predictive-validity-BJf-8ejY.js.map → rubric-predictive-validity-BRR632r1.js.map} +1 -1
- package/dist/{rubric-predictive-validity-C1dCLcvb.d.ts → rubric-predictive-validity-C4sztLR3.d.ts} +2 -2
- package/dist/{rubric-predictive-validity-C1dCLcvb.d.ts.map → rubric-predictive-validity-C4sztLR3.d.ts.map} +1 -1
- package/dist/{run-evidence-DokQtX0-.d.ts → run-evidence-BDIircdA.d.ts} +3 -3
- package/dist/{run-evidence-DokQtX0-.d.ts.map → run-evidence-BDIircdA.d.ts.map} +1 -1
- package/dist/{run-record-DcObtIGh.d.ts → run-record-BPCa2rQ8.d.ts} +4 -4
- package/dist/{run-record-DcObtIGh.d.ts.map → run-record-BPCa2rQ8.d.ts.map} +1 -1
- package/dist/{run-record-BIwU2wdV.js → run-record-vRgqWmJw.js} +3 -3
- package/dist/{run-record-BIwU2wdV.js.map → run-record-vRgqWmJw.js.map} +1 -1
- package/dist/{semantic-concept-judge-Btozx3Vc.js → semantic-concept-judge-Bz64IckK.js} +4 -4
- package/dist/{semantic-concept-judge-Btozx3Vc.js.map → semantic-concept-judge-Bz64IckK.js.map} +1 -1
- package/dist/{server-Bz3WQJs6.js → server-KjXZZUDX.js} +3 -3
- package/dist/{server-Bz3WQJs6.js.map → server-KjXZZUDX.js.map} +1 -1
- package/dist/{skill-usage-BDQVPIG1.d.ts → skill-usage-CFDLLlhF.d.ts} +25 -47
- package/dist/skill-usage-CFDLLlhF.d.ts.map +1 -0
- package/dist/{skillopt-optimization-method-CwSYkv35.d.ts → skillopt-optimization-method-BpbnlvAZ.d.ts} +11 -12
- package/dist/skillopt-optimization-method-BpbnlvAZ.d.ts.map +1 -0
- package/dist/{skillopt-optimization-method-0UmPD6aP.js → skillopt-optimization-method-f4o9sUT4.js} +7 -7
- package/dist/{skillopt-optimization-method-0UmPD6aP.js.map → skillopt-optimization-method-f4o9sUT4.js.map} +1 -1
- package/dist/{statistics-CnGCLLqc.js → statistics-ByxzSiOM.js} +2 -2
- package/dist/{statistics-CnGCLLqc.js.map → statistics-ByxzSiOM.js.map} +1 -1
- package/dist/{statistics-CKOqre5S.d.ts → statistics-_7P642CN.d.ts} +2 -2
- package/dist/{statistics-CKOqre5S.d.ts.map → statistics-_7P642CN.d.ts.map} +1 -1
- package/dist/{summary-report-BEk8OFLs.js → summary-report-9A5y7EsK.js} +4 -4
- package/dist/{summary-report-BEk8OFLs.js.map → summary-report-9A5y7EsK.js.map} +1 -1
- package/dist/{summary-report-CPMINBqs.d.ts → summary-report-DHipz9Kx.d.ts} +3 -3
- package/dist/{summary-report-CPMINBqs.d.ts.map → summary-report-DHipz9Kx.d.ts.map} +1 -1
- package/dist/supervisor-run/index.d.ts +3 -2
- package/dist/supervisor-run/index.js +3 -2
- package/dist/{supervisor-run-Dr5HnTup.js → supervisor-run-B2EWUmQY.js} +28 -454
- package/dist/supervisor-run-B2EWUmQY.js.map +1 -0
- package/dist/{test-graded-scenario-BsqWLmPt.js → test-graded-scenario-JHcKQNpq.js} +2 -2
- package/dist/{test-graded-scenario-BsqWLmPt.js.map → test-graded-scenario-JHcKQNpq.js.map} +1 -1
- package/dist/tools-DZk2Jn64.js +1876 -0
- package/dist/tools-DZk2Jn64.js.map +1 -0
- package/dist/traces.d.ts +6 -7
- package/dist/traces.js +5 -6
- package/dist/{types-DVjczBM9.d.ts → types-CKswbJGO.d.ts} +260 -6
- package/dist/types-CKswbJGO.d.ts.map +1 -0
- package/dist/{types-DiWLru6Z.d.ts → types-CTGbIm57.d.ts} +5 -5
- package/dist/{types-DiWLru6Z.d.ts.map → types-CTGbIm57.d.ts.map} +1 -1
- package/dist/types-CTvKfr5F.d.ts +804 -0
- package/dist/types-CTvKfr5F.d.ts.map +1 -0
- package/dist/{index-CyC1BTmn.d.ts → types-Dea6tiVI.d.ts} +16 -238
- package/dist/types-Dea6tiVI.d.ts.map +1 -0
- package/dist/wire/index.d.ts +3 -3
- package/dist/wire/index.js +1 -1
- package/docs/feedback-trajectories.md +100 -1
- package/docs/trace-analysis.md +374 -58
- package/package.json +5 -1
- package/dist/analyst-BkTS3C58.d.ts +0 -89
- package/dist/analyst-BkTS3C58.d.ts.map +0 -1
- package/dist/analyst-j5je5J7c.js +0 -152
- package/dist/analyst-j5je5J7c.js.map +0 -1
- package/dist/campaign-Dz8uQnhC.js.map +0 -1
- package/dist/client-DcvgkaZi.d.ts.map +0 -1
- package/dist/default-registry-Brxr728w.d.ts.map +0 -1
- package/dist/default-registry-CHmdy2An.js.map +0 -1
- package/dist/errors-8YnH8WlF.js.map +0 -1
- package/dist/errors-CEk209JS.d.ts.map +0 -1
- package/dist/feedback-trajectory-CVaeREXV.d.ts.map +0 -1
- package/dist/index-B4Fjfo5U.d.ts.map +0 -1
- package/dist/index-CyC1BTmn.d.ts.map +0 -1
- package/dist/llm-client-BiK4HW0u.d.ts +0 -290
- package/dist/llm-client-BiK4HW0u.d.ts.map +0 -1
- package/dist/raw-provider-sink-BU29Sh8h.d.ts +0 -134
- package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +0 -1
- package/dist/replay-BRfMIs81.d.ts.map +0 -1
- package/dist/replay-C6wRg47C.js.map +0 -1
- package/dist/researcher-Doo95b50.d.ts.map +0 -1
- package/dist/skill-usage-BDQVPIG1.d.ts.map +0 -1
- package/dist/skillopt-optimization-method-CwSYkv35.d.ts.map +0 -1
- package/dist/store-CxJry_cs.d.ts +0 -229
- package/dist/store-CxJry_cs.d.ts.map +0 -1
- package/dist/supervisor-run-Dr5HnTup.js.map +0 -1
- package/dist/tools-D8yTtNSN.js +0 -1190
- package/dist/tools-D8yTtNSN.js.map +0 -1
- package/dist/types-Cc3qbqzj.d.ts +0 -387
- package/dist/types-Cc3qbqzj.d.ts.map +0 -1
- package/dist/types-DVjczBM9.d.ts.map +0 -1
package/dist/cli.js
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
2
|
import { o as runRolloutReleaseCli } from "./hf-dataset-DBJXXoY1.js";
|
|
3
|
-
import { a as runRpcBatch, o as runRpcOnce, p as handleVersion, r as startServerAsync, s as buildOpenApi } from "./server-
|
|
3
|
+
import { a as runRpcBatch, o as runRpcOnce, p as handleVersion, r as startServerAsync, s as buildOpenApi } from "./server-KjXZZUDX.js";
|
|
4
4
|
import { writeFileSync } from "node:fs";
|
|
5
5
|
//#region src/cli-config.ts
|
|
6
6
|
function resolveCliLlmConfig(env = process.env) {
|
|
@@ -0,0 +1,202 @@
|
|
|
1
|
+
import { l as RunTerminalOutcome } from "./run-record-BPCa2rQ8.js";
|
|
2
|
+
import { _ as GateDecision, j as MutableSurface } from "./types-CTGbIm57.js";
|
|
3
|
+
import { o as InsightReport } from "./insight-report-B9ooYH_g.js";
|
|
4
|
+
//#region src/hosted/types.d.ts
|
|
5
|
+
declare const HOSTED_WIRE_VERSION: '2026-07-24.v1';
|
|
6
|
+
type HostedWireVersion = typeof HOSTED_WIRE_VERSION;
|
|
7
|
+
/** Every ingest request carries these. */
|
|
8
|
+
interface HostedIngestHeaders {
|
|
9
|
+
/** Bearer token. The orchestrator validates against the tenant key. */
|
|
10
|
+
authorization: `Bearer ${string}`;
|
|
11
|
+
/** Stable tenant id (the orchestrator-side primary key for the tenant). */
|
|
12
|
+
'x-tangle-tenant-id': string;
|
|
13
|
+
/** Wire-version pin so the server can reject incompatible payloads. */
|
|
14
|
+
'x-tangle-wire-version': HostedWireVersion;
|
|
15
|
+
/** Stable request key generated once and reused across retries. */
|
|
16
|
+
'idempotency-key': string;
|
|
17
|
+
}
|
|
18
|
+
/** Lifecycle stages of an eval-run as the substrate reports them. */
|
|
19
|
+
type EvalRunStatus = 'started' | 'baseline-complete' | 'generation-complete' | 'gate-decided' | 'finished' | 'errored';
|
|
20
|
+
interface EvalRunCellScore {
|
|
21
|
+
/** Stable scenario id from the consumer's scenario set. */
|
|
22
|
+
scenarioId: string;
|
|
23
|
+
/** Repetition index when reps > 1; 0 for the default. */
|
|
24
|
+
rep: number;
|
|
25
|
+
/** Composite score across successful judges, or null when unscored. */
|
|
26
|
+
compositeMean: number | null;
|
|
27
|
+
/** Per-judge and per-dimension scores; failed or missing judges are absent. */
|
|
28
|
+
dimensions: Record<string, Record<string, number>>;
|
|
29
|
+
/** Root execution result, kept separate from task quality. */
|
|
30
|
+
terminalOutcome: RunTerminalOutcome;
|
|
31
|
+
/** Canonical execution-error count, or null when the producer did not measure it. */
|
|
32
|
+
executionErrorCount: number | null;
|
|
33
|
+
/** Per-cell dispatch or judge error. Missing on success. */
|
|
34
|
+
errorMessage?: string;
|
|
35
|
+
}
|
|
36
|
+
interface EvalRunGenerationSnapshot {
|
|
37
|
+
/** Generation index. 0 is baseline. */
|
|
38
|
+
index: number;
|
|
39
|
+
/** Candidate surface fingerprint (stable hash) — pivot key into the
|
|
40
|
+
* trace stream to fetch the underlying execution. */
|
|
41
|
+
surfaceHash: string;
|
|
42
|
+
/** The candidate surface itself. May be omitted to avoid PII when the
|
|
43
|
+
* consumer prefers not to ship verbatim prompts. */
|
|
44
|
+
surface?: MutableSurface;
|
|
45
|
+
/** Per-cell scores for this generation. */
|
|
46
|
+
cells: EvalRunCellScore[];
|
|
47
|
+
/** Mean across scored cells, or null when no cell has a task-quality label. */
|
|
48
|
+
compositeMean: number | null;
|
|
49
|
+
/** Total $ spent across this generation. */
|
|
50
|
+
costUsd: number;
|
|
51
|
+
/** Wall-clock duration of this generation. */
|
|
52
|
+
durationMs: number;
|
|
53
|
+
}
|
|
54
|
+
/**
|
|
55
|
+
* The top-level eval-run event. One ingest call per logical eval-run;
|
|
56
|
+
* generations stream in incrementally via repeated calls with the same
|
|
57
|
+
* `runId`. The orchestrator deduplicates by `(runId, generation.index)`.
|
|
58
|
+
*/
|
|
59
|
+
interface EvalRunEvent {
|
|
60
|
+
/** Stable run id (the substrate's `runId`). UUID or substrate-generated. */
|
|
61
|
+
runId: string;
|
|
62
|
+
/** Where this run was happening — derived from `RunCampaignOptions.runDir`. */
|
|
63
|
+
runDir: string;
|
|
64
|
+
/** ISO-8601 timestamp the substrate recorded the event. */
|
|
65
|
+
timestamp: string;
|
|
66
|
+
/** Lifecycle stage this event represents. */
|
|
67
|
+
status: EvalRunStatus;
|
|
68
|
+
/** Free-form consumer tags (env, branch, model id, etc.). Searchable. */
|
|
69
|
+
labels: Record<string, string>;
|
|
70
|
+
/** Baseline campaign snapshot. Present when status >= baseline-complete. */
|
|
71
|
+
baseline?: EvalRunGenerationSnapshot;
|
|
72
|
+
/** Per-generation snapshots. Streams in; orchestrator appends. */
|
|
73
|
+
generations: EvalRunGenerationSnapshot[];
|
|
74
|
+
/** Final gate decision. Present when status >= gate-decided. */
|
|
75
|
+
gateDecision?: GateDecision;
|
|
76
|
+
/** Held-out lift = winner-on-holdout - baseline-on-holdout. */
|
|
77
|
+
holdoutLift?: number;
|
|
78
|
+
/** Total $ spent across baseline + every generation. */
|
|
79
|
+
totalCostUsd: number;
|
|
80
|
+
/** Total wall-clock duration. */
|
|
81
|
+
totalDurationMs: number;
|
|
82
|
+
/** Error message if status === 'errored'. */
|
|
83
|
+
errorMessage?: string;
|
|
84
|
+
/** Rigor packet emitted alongside the run — distributional summary,
|
|
85
|
+
* paired-bootstrap lift CI, judge stats, inter-rater agreement,
|
|
86
|
+
* contamination check, failure clusters (when an analyst is wired),
|
|
87
|
+
* outcome correlation (when downstream signal is supplied), and the
|
|
88
|
+
* recommendations the dashboard surfaces verbatim. */
|
|
89
|
+
insightReport?: InsightReport;
|
|
90
|
+
}
|
|
91
|
+
/**
|
|
92
|
+
* Canonical unsigned 64-bit integer encoded as a base-10 string.
|
|
93
|
+
* JSON numbers cannot represent OTLP nanosecond timestamps exactly.
|
|
94
|
+
*/
|
|
95
|
+
type UnixNanoTimestamp = string;
|
|
96
|
+
/**
|
|
97
|
+
* OTel-shape span with a few additional attributes for eval-run pivoting.
|
|
98
|
+
* Compatible with any OTLP collector — `name`, `traceId`, `spanId`,
|
|
99
|
+
* `startTimeUnixNano`, `endTimeUnixNano`, `attributes` are stock OTel.
|
|
100
|
+
*/
|
|
101
|
+
interface TraceSpanEvent {
|
|
102
|
+
traceId: string;
|
|
103
|
+
spanId: string;
|
|
104
|
+
parentSpanId?: string;
|
|
105
|
+
name: string;
|
|
106
|
+
startTimeUnixNano: UnixNanoTimestamp;
|
|
107
|
+
endTimeUnixNano: UnixNanoTimestamp;
|
|
108
|
+
attributes: Record<string, string | number | boolean>;
|
|
109
|
+
events?: Array<{
|
|
110
|
+
timeUnixNano: UnixNanoTimestamp;
|
|
111
|
+
name: string;
|
|
112
|
+
attributes?: Record<string, string | number | boolean>;
|
|
113
|
+
}>;
|
|
114
|
+
status?: {
|
|
115
|
+
code: 'OK' | 'ERROR' | 'UNSET';
|
|
116
|
+
message?: string;
|
|
117
|
+
};
|
|
118
|
+
/** Pivot back into the eval-run stream. */
|
|
119
|
+
'tangle.runId'?: string;
|
|
120
|
+
/** Pivot to the specific generation. */
|
|
121
|
+
'tangle.generation'?: number;
|
|
122
|
+
/** Pivot to the specific cell. */
|
|
123
|
+
'tangle.cellId'?: string;
|
|
124
|
+
/** Pivot to the specific scenario. */
|
|
125
|
+
'tangle.scenarioId'?: string;
|
|
126
|
+
}
|
|
127
|
+
interface IngestEvalRunsRequest {
|
|
128
|
+
wireVersion: HostedWireVersion;
|
|
129
|
+
events: EvalRunEvent[];
|
|
130
|
+
}
|
|
131
|
+
interface IngestTracesRequest {
|
|
132
|
+
wireVersion: HostedWireVersion;
|
|
133
|
+
spans: TraceSpanEvent[];
|
|
134
|
+
}
|
|
135
|
+
interface IngestResponse {
|
|
136
|
+
/** Accepted events / spans count. */
|
|
137
|
+
accepted: number;
|
|
138
|
+
/** Rejected events with reasons (validation failures, dup idempotency key, etc.). */
|
|
139
|
+
rejected: Array<{
|
|
140
|
+
index: number;
|
|
141
|
+
reason: string;
|
|
142
|
+
}>;
|
|
143
|
+
}
|
|
144
|
+
//#endregion
|
|
145
|
+
//#region src/hosted/client.d.ts
|
|
146
|
+
interface HostedTenant {
|
|
147
|
+
/** Orchestrator endpoint base URL (no trailing slash). Required. */
|
|
148
|
+
endpoint: string;
|
|
149
|
+
/** Bearer token issued by the orchestrator. Required. */
|
|
150
|
+
apiKey: string;
|
|
151
|
+
/** Tenant id — the orchestrator's primary key for this consumer. Required. */
|
|
152
|
+
tenantId: string;
|
|
153
|
+
/** Optional `fetch` override (auth wrappers, custom agent, test mocks). */
|
|
154
|
+
fetchImpl?: typeof fetch;
|
|
155
|
+
/** Per-call timeout in ms. Default 30s. */
|
|
156
|
+
timeoutMs?: number;
|
|
157
|
+
/** Retries on 5xx / network errors. Default 2. */
|
|
158
|
+
retries?: number;
|
|
159
|
+
}
|
|
160
|
+
interface HostedClient {
|
|
161
|
+
ingestEvalRun(event: EvalRunEvent, idempotencyKey?: string): Promise<IngestResponse>;
|
|
162
|
+
ingestEvalRuns(events: EvalRunEvent[], idempotencyKey?: string): Promise<IngestResponse>;
|
|
163
|
+
ingestTraces(spans: TraceSpanEvent[], idempotencyKey?: string): Promise<IngestResponse>;
|
|
164
|
+
readonly tenant: HostedTenant;
|
|
165
|
+
readonly wireVersion: HostedWireVersion;
|
|
166
|
+
}
|
|
167
|
+
declare function createHostedClient(tenant: HostedTenant): HostedClient;
|
|
168
|
+
/**
|
|
169
|
+
* Build a `HostedClient` from environment, or `undefined` when ingest is not
|
|
170
|
+
* configured — the canonical, fail-soft wiring every product uses so eval-run +
|
|
171
|
+
* trace provenance lands in the Intelligence dashboard with ONE call:
|
|
172
|
+
*
|
|
173
|
+
* const hosted = hostedClientFromEnv()
|
|
174
|
+
* // ...run the loop...
|
|
175
|
+
* await emitLoopProvenance({ ..., hostedClient: hosted }) // no-op if undefined
|
|
176
|
+
*
|
|
177
|
+
* Returns `undefined` (NOT an error) when any of endpoint / apiKey / tenantId is
|
|
178
|
+
* missing — so a product wires the ship call unconditionally and it stays a
|
|
179
|
+
* no-op until the env is set. Env precedence:
|
|
180
|
+
* - endpoint: `TANGLE_INGEST_URL` → `TANGLE_ORCHESTRATOR_URL`
|
|
181
|
+
* - apiKey: `TANGLE_INGEST_API_KEY` → `TANGLE_API_KEY`
|
|
182
|
+
* - tenantId: `TANGLE_TENANT_ID`
|
|
183
|
+
* A trailing slash on the endpoint is stripped. Pass `overrides` to supply any
|
|
184
|
+
* field directly (e.g. a fixed `tenantId` per product) — overrides win over env.
|
|
185
|
+
*/
|
|
186
|
+
/**
|
|
187
|
+
* Build a {@link HostedTenant} config from env — the input `selfImprove`'s
|
|
188
|
+
* `hostedTenant` and `emitLoopProvenance` take. Same env precedence + overrides
|
|
189
|
+
* as {@link hostedClientFromEnv}; returns `undefined` (not an error) when any of
|
|
190
|
+
* endpoint / apiKey / tenantId is missing, so a product wires
|
|
191
|
+
* `hostedTenant: hostedTenantFromEnv({ tenantId: 'my-agent' })` unconditionally
|
|
192
|
+
* and it stays off until the env is set.
|
|
193
|
+
*/
|
|
194
|
+
declare function hostedTenantFromEnv(overrides?: Partial<HostedTenant> & {
|
|
195
|
+
env?: Record<string, string | undefined>;
|
|
196
|
+
}): HostedTenant | undefined;
|
|
197
|
+
declare function hostedClientFromEnv(overrides?: Partial<HostedTenant> & {
|
|
198
|
+
env?: Record<string, string | undefined>;
|
|
199
|
+
}): HostedClient | undefined;
|
|
200
|
+
//#endregion
|
|
201
|
+
export { UnixNanoTimestamp as _, hostedTenantFromEnv as a, EvalRunGenerationSnapshot as c, HostedIngestHeaders as d, HostedWireVersion as f, TraceSpanEvent as g, IngestTracesRequest as h, hostedClientFromEnv as i, EvalRunStatus as l, IngestResponse as m, HostedTenant as n, EvalRunCellScore as o, IngestEvalRunsRequest as p, createHostedClient as r, EvalRunEvent as s, HostedClient as t, HOSTED_WIRE_VERSION as u };
|
|
202
|
+
//# sourceMappingURL=client-C8L6h6Wf.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"client-C8L6h6Wf.d.ts","names":[],"sources":["../src/hosted/types.ts","../src/hosted/client.ts"],"mappings":";;;;cAgCa;KACD,2BAA2B;;UAKtB;;EAEf;;EAEA;;EAEA,yBAAyB;;EAEzB;;;KAMU;UAQK;;EAEf;;EAEA;;EAEA;;EAEA,YAAY,eAAe;;EAE3B,iBAAiB;;EAEjB;;EAEA;;UAGe;;EAEf;;;EAGA;;;EAGA,UAAU;;EAEV,OAAO;;EAEP;;EAEA;;EAEA;;;;;;;UAQe;;EAEf;;EAEA;;EAEA;;EAEA,QAAQ;;EAER,QAAQ;;EAER,WAAW;;EAEX,aAAa;;EAEb,eAAe;;EAEf;;EAEA;;EAEA;;EAEA;;;;;;EAMA,gBAAgB;;;;;;KASN;;;;;;UAOK;EACf;EACA;EACA;EACA;EACA,mBAAmB;EACnB,iBAAiB;EACjB,YAAY;EACZ,SAAS;IACP,cAAc;IACd;IACA,aAAa;;EAEf;IAAW;IAAgC;;;EAE3C;;EAEA;;EAEA;;EAEA;;UAKe;EACf,aAAa;EACb,QAAQ;;UAGO;EACf,aAAa;EACb,OAAO;;UAGQ;;EAEf;;EAEA,UAAU;IAAQ;IAAe;;;;;UC1JlB;;EAEf;;EAEA;;EAEA;;EAEA,mBAAmB;;EAEnB;;EAEA;;UAGe;EACf,cAAc,OAAO,cAAc,0BAA0B,QAAQ;EACrE,eAAe,QAAQ,gBAAgB,0BAA0B,QAAQ;EACzE,aAAa,OAAO,kBAAkB,0BAA0B,QAAQ;WAC/D,QAAQ;WACR,aAAa;;iBAyGR,mBAAmB,QAAQ,eAAe;;;;;;;;;;;;;;;;;;;;;;;;;;;iBAsE1C,oBACd,YAAW,QAAQ;EAAkB,MAAM;IAC1C;iBAiBa,oBACd,YAAW,QAAQ;EAAkB,MAAM;IAC1C"}
|
|
@@ -0,0 +1,240 @@
|
|
|
1
|
+
import { t as DefaultVerdict } from "./verdict-Dps8_okt.js";
|
|
2
|
+
import { c as CostLedgerHandle } from "./cost-ledger-D2o6JOrL.js";
|
|
3
|
+
import { A as ChatClient, _t as RawProviderSink } from "./types-CTvKfr5F.js";
|
|
4
|
+
//#region src/artifact-validator.d.ts
|
|
5
|
+
/**
|
|
6
|
+
* Artifact validators.
|
|
7
|
+
*
|
|
8
|
+
* Generic "score a produced artifact" primitive. Tax uses it for PDF form
|
|
9
|
+
* correctness, research for sourced briefs, browser for task assertions, coding
|
|
10
|
+
* for social posts. One interface, many validators; all plug into
|
|
11
|
+
* `BenchmarkRunner` the same way.
|
|
12
|
+
*
|
|
13
|
+
* A validator receives an `Artifact` (file on disk, JSON blob, text, binary)
|
|
14
|
+
* plus a `ValidationContext` (scenario id, the turns that produced it) and
|
|
15
|
+
* returns a `ValidationResult` with pass/fail + 0..1 score + structured
|
|
16
|
+
* issues.
|
|
17
|
+
*/
|
|
18
|
+
interface Artifact {
|
|
19
|
+
/** Logical kind — validators type-guard on this */
|
|
20
|
+
kind: 'file' | 'json' | 'text' | 'binary' | string;
|
|
21
|
+
/** Filesystem-style path, optional */
|
|
22
|
+
path?: string;
|
|
23
|
+
/** String content for text/json/file kinds */
|
|
24
|
+
content?: string;
|
|
25
|
+
/** Binary content (if kind === 'binary') */
|
|
26
|
+
bytes?: Uint8Array;
|
|
27
|
+
/** Caller-supplied metadata (mimeType, sha256, size, etc.) */
|
|
28
|
+
metadata?: Record<string, unknown>;
|
|
29
|
+
}
|
|
30
|
+
interface ValidationContext {
|
|
31
|
+
scenarioId: string;
|
|
32
|
+
turnIndex?: number;
|
|
33
|
+
/** Prior artifacts for multi-artifact scenarios */
|
|
34
|
+
priorArtifacts?: Artifact[];
|
|
35
|
+
/** Free-form hints the validator uses for domain-specific checks */
|
|
36
|
+
hints?: Record<string, unknown>;
|
|
37
|
+
}
|
|
38
|
+
interface ValidationIssue {
|
|
39
|
+
severity: 'error' | 'warning' | 'info';
|
|
40
|
+
message: string;
|
|
41
|
+
/** Optional path into the artifact (e.g. JSON path or byte offset) */
|
|
42
|
+
locus?: string;
|
|
43
|
+
}
|
|
44
|
+
interface ValidationResult {
|
|
45
|
+
pass: boolean;
|
|
46
|
+
/** 0–1 normalized score. Validators should be monotonic in pass-ness. */
|
|
47
|
+
score: number;
|
|
48
|
+
issues: ValidationIssue[];
|
|
49
|
+
/** Diagnostic payload for reporters */
|
|
50
|
+
evidence?: Record<string, unknown>;
|
|
51
|
+
}
|
|
52
|
+
interface ArtifactValidator {
|
|
53
|
+
/** Stable identifier for the validator; appears in reports. */
|
|
54
|
+
name: string;
|
|
55
|
+
/** Optional description for human-facing reports. */
|
|
56
|
+
description?: string;
|
|
57
|
+
/** Called once per artifact; validators are expected to be pure + idempotent. */
|
|
58
|
+
validate(artifact: Artifact, context: ValidationContext): Promise<ValidationResult>;
|
|
59
|
+
}
|
|
60
|
+
/**
|
|
61
|
+
* Run every validator on the same artifact; aggregate pass as AND, score as
|
|
62
|
+
* (weighted) mean, issues concatenated. Weights default to 1 each.
|
|
63
|
+
*/
|
|
64
|
+
declare function composeValidators(validators: ArtifactValidator[], options?: {
|
|
65
|
+
name?: string;
|
|
66
|
+
weights?: number[];
|
|
67
|
+
}): ArtifactValidator;
|
|
68
|
+
/** Pass if the artifact body matches a provided regex. */
|
|
69
|
+
declare function regexMatch(name: string, pattern: RegExp): ArtifactValidator;
|
|
70
|
+
/** Pass if JSON parses and every required key is present. */
|
|
71
|
+
declare function jsonHasKeys(name: string, requiredPaths: string[]): ArtifactValidator;
|
|
72
|
+
/** Pass if min ≤ byte length ≤ max. */
|
|
73
|
+
declare function byteLengthRange(name: string, min: number, max: number): ArtifactValidator;
|
|
74
|
+
/** Pass if the artifact contains every required substring (case-insensitive by default). */
|
|
75
|
+
declare function containsAll(name: string, required: string[], options?: {
|
|
76
|
+
caseSensitive?: boolean;
|
|
77
|
+
}): ArtifactValidator;
|
|
78
|
+
//#endregion
|
|
79
|
+
//#region src/completion-verifier.d.ts
|
|
80
|
+
/** What kind of produced state can satisfy a requirement structurally. */
|
|
81
|
+
type SatisfiedBy = 'artifact' | 'proposal' | 'tool-call' | 'any';
|
|
82
|
+
interface CompletionRequirement {
|
|
83
|
+
/** Stable id from the task gold (e.g. a persona's `expected_requirements[].req_id`). */
|
|
84
|
+
reqId: string;
|
|
85
|
+
/** Human-readable description of the required deliverable. */
|
|
86
|
+
title: string;
|
|
87
|
+
/** Optional kind/category hint, matched against a produced item's kind. */
|
|
88
|
+
category?: string;
|
|
89
|
+
/** What produced state satisfies this requirement. Defaults to 'any'. */
|
|
90
|
+
satisfiedBy?: SatisfiedBy;
|
|
91
|
+
}
|
|
92
|
+
interface TaskGold {
|
|
93
|
+
taskId: string;
|
|
94
|
+
requirements: CompletionRequirement[];
|
|
95
|
+
}
|
|
96
|
+
interface ProducedProposal {
|
|
97
|
+
id: string;
|
|
98
|
+
title: string;
|
|
99
|
+
status: 'pending' | 'approved' | 'rejected';
|
|
100
|
+
/** Optional persisted body — when present, enables a correctness check. */
|
|
101
|
+
content?: string;
|
|
102
|
+
}
|
|
103
|
+
/** Everything observable about what a run actually produced. */
|
|
104
|
+
interface ProducedState {
|
|
105
|
+
/** Persisted vault artifacts. Reuses the shared `Artifact` shape. */
|
|
106
|
+
artifacts: Artifact[];
|
|
107
|
+
/** Proposals / filings the agent created. */
|
|
108
|
+
proposals: ProducedProposal[];
|
|
109
|
+
/** Names of tools the agent invoked. */
|
|
110
|
+
toolCalls: string[];
|
|
111
|
+
}
|
|
112
|
+
interface RequirementCheck {
|
|
113
|
+
reqId: string;
|
|
114
|
+
title: string;
|
|
115
|
+
/** A produced item of the right kind matched the requirement, non-empty. */
|
|
116
|
+
structurallyPresent: boolean;
|
|
117
|
+
/**
|
|
118
|
+
* Whether the matched item actually fulfils the requirement. `null` when
|
|
119
|
+
* not structurally present, when the matched item carries no content
|
|
120
|
+
* to assess, or when the correctness check itself failed (`unmeasured`).
|
|
121
|
+
*/
|
|
122
|
+
correct: boolean | null;
|
|
123
|
+
/** structurallyPresent && !unmeasured && correct !== false. */
|
|
124
|
+
satisfied: boolean;
|
|
125
|
+
/**
|
|
126
|
+
* Set when the correctness check itself errored (LLM call failure or an
|
|
127
|
+
* unparseable response after retry). The requirement's fulfilment is
|
|
128
|
+
* UNKNOWN — `correct` stays null, `satisfied` is false, and
|
|
129
|
+
* `completionVerdict` excludes the row from `completionRate`'s
|
|
130
|
+
* denominator. Never folded into a zero: a synthetic zero is
|
|
131
|
+
* indistinguishable from a real failure (see `JudgeParseError`).
|
|
132
|
+
*/
|
|
133
|
+
unmeasured?: true;
|
|
134
|
+
/** Why the correctness check could not be measured (present iff `unmeasured`). */
|
|
135
|
+
unmeasuredReason?: string;
|
|
136
|
+
/** Human-readable evidence for the verdict. */
|
|
137
|
+
evidence: string[];
|
|
138
|
+
}
|
|
139
|
+
/** Extends the substrate verdict spine: `valid` = `fullyComplete` and
|
|
140
|
+
* `score` = `completionRate` — derived in `completionVerdict()`, the one
|
|
141
|
+
* place those equalities hold by construction. */
|
|
142
|
+
interface CompletionVerdict extends DefaultVerdict {
|
|
143
|
+
taskId: string;
|
|
144
|
+
requirements: RequirementCheck[];
|
|
145
|
+
/** satisfied / MEASURABLE requirements (unmeasured rows leave the denominator). */
|
|
146
|
+
completionRate: number;
|
|
147
|
+
/** Every measurable requirement satisfied (false when anything is unmeasured). */
|
|
148
|
+
fullyComplete: boolean;
|
|
149
|
+
/** Requirements whose correctness check errored — reported, never scored as zero. */
|
|
150
|
+
unmeasuredCount: number;
|
|
151
|
+
}
|
|
152
|
+
/**
|
|
153
|
+
* Construct a `CompletionVerdict` from the per-requirement checks, deriving
|
|
154
|
+
* `completionRate` / `fullyComplete` and the spine fields (`valid` =
|
|
155
|
+
* `fullyComplete`, `score` = `completionRate`) in one place. Throws on zero
|
|
156
|
+
* requirements — a verdict over nothing is a misconfiguration, mirroring
|
|
157
|
+
* `verifyCompletion`'s gold-spec guard.
|
|
158
|
+
*/
|
|
159
|
+
declare function completionVerdict(input: {
|
|
160
|
+
taskId: string;
|
|
161
|
+
requirements: RequirementCheck[];
|
|
162
|
+
}): CompletionVerdict;
|
|
163
|
+
/**
|
|
164
|
+
* Decides whether a produced item's content actually fulfils a requirement.
|
|
165
|
+
* Injected so the structural verifier stays pure and unit-testable; the
|
|
166
|
+
* production implementation is `createLlmCorrectnessChecker`.
|
|
167
|
+
*/
|
|
168
|
+
type CorrectnessChecker = (requirement: CompletionRequirement, content: string) => Promise<{
|
|
169
|
+
correct: boolean;
|
|
170
|
+
reason: string;
|
|
171
|
+
}>;
|
|
172
|
+
/**
|
|
173
|
+
* Verify whether a run completed the task. `checkCorrectness` is injected —
|
|
174
|
+
* `createLlmCorrectnessChecker` for production, a deterministic stub in tests.
|
|
175
|
+
*
|
|
176
|
+
* Throws on a gold spec with no requirements: an eval task that requires
|
|
177
|
+
* nothing is a misconfiguration, not a vacuously-complete task.
|
|
178
|
+
*/
|
|
179
|
+
declare function verifyCompletion(gold: TaskGold, state: ProducedState, checkCorrectness: CorrectnessChecker): Promise<CompletionVerdict>;
|
|
180
|
+
interface LlmCorrectnessCheckerOpts {
|
|
181
|
+
model?: string;
|
|
182
|
+
/** Optional ledger for direct use. */
|
|
183
|
+
costLedger?: CostLedgerHandle;
|
|
184
|
+
costPhase?: string;
|
|
185
|
+
costTags?: Record<string, string>;
|
|
186
|
+
signal?: AbortSignal;
|
|
187
|
+
/** Max chars of artifact content sent to the checker. */
|
|
188
|
+
maxContentChars?: number;
|
|
189
|
+
/**
|
|
190
|
+
* Checker LLM calls per requirement before giving up (parse failures and
|
|
191
|
+
* call errors both consume attempts). The failure then surfaces as an
|
|
192
|
+
* `unmeasured` requirement, never a zero.
|
|
193
|
+
*/
|
|
194
|
+
maxAttempts?: number;
|
|
195
|
+
/**
|
|
196
|
+
* Forensic capture of every checker request/response/error — without it a
|
|
197
|
+
* checker failure is unauditable (the agent-turn raws never contain the
|
|
198
|
+
* checker's own calls). Same sink contract as `LlmClient`.
|
|
199
|
+
*/
|
|
200
|
+
rawSink?: RawProviderSink;
|
|
201
|
+
}
|
|
202
|
+
/**
|
|
203
|
+
* Parse the correctness checker's model response. Tolerates a response
|
|
204
|
+
* truncated mid-JSON (max_tokens cap) by auto-closing the prefix — the
|
|
205
|
+
* verdict boolean usually lands in the first few tokens, so a recovered
|
|
206
|
+
* prefix with a boolean `correct` is a real measurement, not a guess.
|
|
207
|
+
* Fails loud (JudgeParseError) when no boolean verdict is recoverable.
|
|
208
|
+
*/
|
|
209
|
+
declare function parseCorrectnessResponse(raw: string): {
|
|
210
|
+
correct: boolean;
|
|
211
|
+
reason: string;
|
|
212
|
+
};
|
|
213
|
+
/**
|
|
214
|
+
* Production `CorrectnessChecker` — one LLM call per matched artifact,
|
|
215
|
+
* deterministic (temperature 0), structured JSON out. Judges fulfilment
|
|
216
|
+
* only: a plan, a gesture, or a description of what should be done does not
|
|
217
|
+
* fulfil a requirement — the artifact must BE the deliverable.
|
|
218
|
+
*/
|
|
219
|
+
declare function createLlmCorrectnessChecker(chat: ChatClient, opts?: LlmCorrectnessCheckerOpts): CorrectnessChecker;
|
|
220
|
+
/**
|
|
221
|
+
* Deterministic `CorrectnessChecker` — the no-LLM counterpart to
|
|
222
|
+
* `createLlmCorrectnessChecker`. A produced item fulfils a requirement when its
|
|
223
|
+
* content is substantive (≥ `minContentLength` chars) AND recalls ≥ `minRecall`
|
|
224
|
+
* of the requirement title's significant tokens. No network.
|
|
225
|
+
*
|
|
226
|
+
* Polarity-blind: token recall credits a negation that contains the
|
|
227
|
+
* requirement's tokens ("I will NOT produce the comparison" recalls every token
|
|
228
|
+
* of "produce the comparison"). The structural match stage is ALSO lexical, so
|
|
229
|
+
* pairing the two collapses to a single gameable gate. Use this only as an
|
|
230
|
+
* opt-in structural pre-filter or for tasks whose requirements have no polarity
|
|
231
|
+
* to invert; for produced-state grading the correctness checker MUST be semantic
|
|
232
|
+
* (`createLlmCorrectnessChecker`). See the anti-game fixtures in the test suite.
|
|
233
|
+
*/
|
|
234
|
+
declare function createTokenRecallChecker(opts?: {
|
|
235
|
+
minRecall?: number;
|
|
236
|
+
minContentLength?: number;
|
|
237
|
+
}): CorrectnessChecker;
|
|
238
|
+
//#endregion
|
|
239
|
+
export { jsonHasKeys as C, containsAll as S, ValidationContext as _, ProducedProposal as a, byteLengthRange as b, SatisfiedBy as c, createLlmCorrectnessChecker as d, createTokenRecallChecker as f, ArtifactValidator as g, Artifact as h, LlmCorrectnessCheckerOpts as i, TaskGold as l, verifyCompletion as m, CompletionVerdict as n, ProducedState as o, parseCorrectnessResponse as p, CorrectnessChecker as r, RequirementCheck as s, CompletionRequirement as t, completionVerdict as u, ValidationIssue as v, regexMatch as w, composeValidators as x, ValidationResult as y };
|
|
240
|
+
//# sourceMappingURL=completion-verifier-DSyRNVzU.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"completion-verifier-DSyRNVzU.d.ts","names":[],"sources":["../src/artifact-validator.ts","../src/completion-verifier.ts"],"mappings":";;;;;;;;;;;;;;;;;UAciB;;EAEf;;EAEA;;EAEA;;EAEA,QAAQ;;EAER,WAAW;;UAGI;EACf;EACA;;EAEA,iBAAiB;;EAEjB,QAAQ;;UAGO;EACf;EACA;;EAEA;;UAGe;EACf;;EAEA;EACA,QAAQ;;EAER,WAAW;;UAGI;;EAEf;;EAEA;;EAEA,SAAS,UAAU,UAAU,SAAS,oBAAoB,QAAQ;;;;;;iBAWpD,kBACd,YAAY,qBACZ;EAAY;EAAe;IAC1B;;iBAgCa,WAAW,cAAc,SAAS,SAAS;;iBAkB3C,YAAY,cAAc,0BAA0B;;iBAuCpD,gBAAgB,cAAc,aAAa,cAAc;;iBAoBzD,YACd,cACA,oBACA;EAAY;IACX;;;;KCnJS;UAEK;;EAEf;;EAEA;;EAEA;;EAEA,cAAc;;UAGC;EACf;EACA,cAAc;;UAGC;EACf;EACA;EACA;;EAEA;;;UAIe;;EAEf,WAAW;;EAEX,WAAW;;EAEX;;UAGe;EACf;EACA;;EAEA;;;;;;EAMA;;EAEA;;;;;;;;;EASA;;EAEA;;EAEA;;;;;UAMe,0BAA0B;EACzC;EACA,cAAc;;EAEd;;EAEA;;EAEA;;;;;;;;;iBAUc,kBAAkB;EAChC;EACA,cAAc;IACZ;;;;;;KAoCQ,sBACV,aAAa,uBACb,oBACG;EAAU;EAAkB;;;;;;;;;iBAqLX,iBACpB,MAAM,UACN,OAAO,eACP,kBAAkB,qBACjB,QAAQ;UAoFM;EACf;;EAEA,aAAa;EACb;EACA,WAAW;EACX,SAAS;;EAET;;;;;;EAMA;;;;;;EAMA,UAAU;;;;;;;;;iBAUI,yBAAyB;EAAgB;EAAkB;;;;;;;;iBAiC3D,4BACd,MAAM,YACN,OAAM,4BACL;;;;;;;;;;;;;;;iBA6Ia,yBACd;EAAQ;EAAoB;IAC3B"}
|
package/dist/contract/index.d.ts
CHANGED
|
@@ -1,13 +1,14 @@
|
|
|
1
|
-
import { c as CostLedgerHandle, f as CostLedgerSummary, m as CostReceipt } from "../cost-ledger-
|
|
2
|
-
import { a as RunRecord, n as RunCostProvenance, s as RunSplitTag } from "../run-record-
|
|
3
|
-
import { A as ChatClient, F as CreateChatClientOpts, V as createChatClient } from "../types-
|
|
4
|
-
import { h as ProposalFindingOrigin, i as AnalystFinding, m as ProposalFinding, v as makeProposalFinding } from "../types-
|
|
5
|
-
import { n as buildDefaultAnalystRegistry, t as DefaultAnalystRegistryOptions } from "../default-registry-
|
|
6
|
-
import { C as JudgeDimension, H as SurfaceProposer, M as OptimizerConfig, R as Scenario, S as JudgeConfig, V as SessionScript, _ as GateDecision, a as CampaignResult, b as GenerationRecord, c as CampaignTraceWriter, d as DispatchContext, f as DispatchFn, g as GateContribution, h as GateContext, i as CampaignCostMeter, j as MutableSurface, k as LabeledScenarioStore, l as CodeSurface, m as GateCheckStatus, n as CampaignArtifactWriter, p as Gate, r as CampaignCellResult, t as CampaignAggregates, v as GateResult, w as JudgeScore, y as GenerationCandidate } from "../types-
|
|
7
|
-
import { A as RunEvalOptions, B as OptimizerModelBudget, Bt as ComparisonCost, C as RunImprovementLoopOptions, Cn as ReferenceEquivalenceJudgeResult, D as RunOptimizationOptions, Dn as LlmJudgeDimension, E as PremeasuredOptimizationBaseline, En as runReferenceEquivalenceJudge, F as GepaOptimizationMethodConfig, Ft as ExternalTextOptimizerContext, G as ObjectiveSource, Gt as OptimizationMethodProvenance, H as AxisVerdict, Ht as OptimizationMethodComparison, I as GepaOptimizationRecipe, It as ExternalTextOptimizerResult, J as PromotionPolicy, K as ParetoSignificanceGateOptions, Kt as OptimizationMethodResult, L as GepaRunnerCommand, Lt as ExternalOptimizationExample, M as GepaAdaptiveEngineRun, Mt as composeGate, N as GepaEngineOptions, Nt as externalTextOptimizationMethod, On as LlmJudgeOptions, P as GepaEngineRun, Pt as ExternalTextOptimizationMethodConfig, R as gepaOptimizationMethod, Rt as ExternalTextEvaluationResponse, Sn as ReferenceEquivalenceJudgeOptions, T as runImprovementLoop, Tn as createReferenceEquivalenceJudge, U as BuildEvidenceVectorOptions, Ut as OptimizationMethodInput, V as AxisEvidence, Vt as OptimizationMethod, W as EvidenceVector, X as paretoPolicy, Xt as OptimizationTokenUsage, Y as buildEvidenceVector, Yt as OptimizationPackageSource, Z as paretoSignificanceGate, Zt as compareOptimizationMethods, bn as REFERENCE_EQUIVALENCE_JUDGE_VERSION, dt as DefaultProductionGateCheck, en as CampaignCellFailureReceipt, ft as DefaultProductionGateOptions, i as skillOptOptimizationMethod, in as RunCampaignOptions, j as runEval, kn as llmJudge, ln as fsCampaignStorage, lt as HeldOutGateOptions, mn as campaignSplitDigest, mt as defaultProductionGate, n as SkillOptRunnerCommand, on as runCampaign, ot as PowerPreflight, p as LoopProvenanceRecord, pt as DefaultProductionRewardHackingOptions, q as PromotionObjective, r as SkillOptTrainerConfig, sn as CampaignStorage, t as SkillOptOptimizationMethodConfig, un as inMemoryCampaignStorage, ut as heldOutGate, w as RunImprovementLoopResult, wn as ReferenceEquivalenceScenario, xn as ReferenceEquivalenceJudgeInput, yn as REFERENCE_EQUIVALENCE_INPUT_LIMITS, z as OpenAICompatibleOptimizerModel, zt as CompareOptimizationMethodsOptions } from "../skillopt-optimization-method-
|
|
8
|
-
import {
|
|
1
|
+
import { c as CostLedgerHandle, f as CostLedgerSummary, m as CostReceipt } from "../cost-ledger-D2o6JOrL.js";
|
|
2
|
+
import { a as RunRecord, n as RunCostProvenance, s as RunSplitTag } from "../run-record-BPCa2rQ8.js";
|
|
3
|
+
import { A as ChatClient, F as CreateChatClientOpts, V as createChatClient } from "../types-CTvKfr5F.js";
|
|
4
|
+
import { h as ProposalFindingOrigin, i as AnalystFinding, m as ProposalFinding, v as makeProposalFinding } from "../types-CKswbJGO.js";
|
|
5
|
+
import { n as buildDefaultAnalystRegistry, t as DefaultAnalystRegistryOptions } from "../default-registry-Dc5D_Loc.js";
|
|
6
|
+
import { C as JudgeDimension, H as SurfaceProposer, M as OptimizerConfig, R as Scenario, S as JudgeConfig, V as SessionScript, _ as GateDecision, a as CampaignResult, b as GenerationRecord, c as CampaignTraceWriter, d as DispatchContext, f as DispatchFn, g as GateContribution, h as GateContext, i as CampaignCostMeter, j as MutableSurface, k as LabeledScenarioStore, l as CodeSurface, m as GateCheckStatus, n as CampaignArtifactWriter, p as Gate, r as CampaignCellResult, t as CampaignAggregates, v as GateResult, w as JudgeScore, y as GenerationCandidate } from "../types-CTGbIm57.js";
|
|
7
|
+
import { A as RunEvalOptions, B as OptimizerModelBudget, Bt as ComparisonCost, C as RunImprovementLoopOptions, Cn as ReferenceEquivalenceJudgeResult, D as RunOptimizationOptions, Dn as LlmJudgeDimension, E as PremeasuredOptimizationBaseline, En as runReferenceEquivalenceJudge, F as GepaOptimizationMethodConfig, Ft as ExternalTextOptimizerContext, G as ObjectiveSource, Gt as OptimizationMethodProvenance, H as AxisVerdict, Ht as OptimizationMethodComparison, I as GepaOptimizationRecipe, It as ExternalTextOptimizerResult, J as PromotionPolicy, K as ParetoSignificanceGateOptions, Kt as OptimizationMethodResult, L as GepaRunnerCommand, Lt as ExternalOptimizationExample, M as GepaAdaptiveEngineRun, Mt as composeGate, N as GepaEngineOptions, Nt as externalTextOptimizationMethod, On as LlmJudgeOptions, P as GepaEngineRun, Pt as ExternalTextOptimizationMethodConfig, R as gepaOptimizationMethod, Rt as ExternalTextEvaluationResponse, Sn as ReferenceEquivalenceJudgeOptions, T as runImprovementLoop, Tn as createReferenceEquivalenceJudge, U as BuildEvidenceVectorOptions, Ut as OptimizationMethodInput, V as AxisEvidence, Vt as OptimizationMethod, W as EvidenceVector, X as paretoPolicy, Xt as OptimizationTokenUsage, Y as buildEvidenceVector, Yt as OptimizationPackageSource, Z as paretoSignificanceGate, Zt as compareOptimizationMethods, bn as REFERENCE_EQUIVALENCE_JUDGE_VERSION, dt as DefaultProductionGateCheck, en as CampaignCellFailureReceipt, ft as DefaultProductionGateOptions, i as skillOptOptimizationMethod, in as RunCampaignOptions, j as runEval, kn as llmJudge, ln as fsCampaignStorage, lt as HeldOutGateOptions, mn as campaignSplitDigest, mt as defaultProductionGate, n as SkillOptRunnerCommand, on as runCampaign, ot as PowerPreflight, p as LoopProvenanceRecord, pt as DefaultProductionRewardHackingOptions, q as PromotionObjective, r as SkillOptTrainerConfig, sn as CampaignStorage, t as SkillOptOptimizationMethodConfig, un as inMemoryCampaignStorage, ut as heldOutGate, w as RunImprovementLoopResult, wn as ReferenceEquivalenceScenario, xn as ReferenceEquivalenceJudgeInput, yn as REFERENCE_EQUIVALENCE_INPUT_LIMITS, z as OpenAICompatibleOptimizerModel, zt as CompareOptimizationMethodsOptions } from "../skillopt-optimization-method-BpbnlvAZ.js";
|
|
8
|
+
import { a as FailureClusterInsight, c as JudgeInsight, d as Recommendation, f as ReleaseSummary, i as FailureClassTally, l as LiftInsight, m as TokenUsageInsight, n as ExecutionErrorOutcomeCell, o as InsightReport, p as ScalarDistribution, r as ExecutionInsight, s as InterRaterInsight, t as CostProvenanceSummary, u as OutcomeCorrelationInsight } from "../insight-report-B9ooYH_g.js";
|
|
9
|
+
import { c as EvalRunGenerationSnapshot, g as TraceSpanEvent, n as HostedTenant, o as EvalRunCellScore, s as EvalRunEvent } from "../client-C8L6h6Wf.js";
|
|
9
10
|
import { i as InMemoryOutcomeStore, n as FileSystemOutcomeStore, o as OutcomeStore, r as FileSystemOutcomeStoreOptions, t as DeploymentOutcome } from "../outcome-store-BYHIuO0e.js";
|
|
10
|
-
import { a as summarizeExecution, i as analyzeRuns, n as ExecutionReport, r as SummarizeExecutionOptions, t as AnalyzeRunsOptions } from "../analyze-runs-
|
|
11
|
+
import { a as summarizeExecution, i as analyzeRuns, n as ExecutionReport, r as SummarizeExecutionOptions, t as AnalyzeRunsOptions } from "../analyze-runs-PVtnfjvA.js";
|
|
11
12
|
import { AgentCandidateBenchmarkCellRef, AgentCandidateBenchmarkSuiteInputs, AgentCandidateBenchmarkTask, AgentCandidateBenchmarkTaskMaterial, AgentCandidateBundle, AgentCandidateEvaluationPolicy, AgentCandidateExperiment, AgentCandidateExperimentMaterial, AgentCandidateExperimentMeasurement, AgentImprovementCost, AgentImprovementMeasuredComparison, AgentProfileImprovementExperiment, AgentProfileImprovementExperimentMaterial, AgentProfileImprovementMeasuredComparison, AgentProfileImprovementMeasurement, AgentProfileImprovementRunCell, AgentProfileImprovementRunReceipt, AgentProfileImprovementSuiteInputs, AgentProfileImprovementTask, AgentProfileImprovementTaskMaterial, CandidateExecutionEvidence, Sha256Digest } from "@tangle-network/agent-interface";
|
|
12
13
|
//#region src/contract/self-improve.d.ts
|
|
13
14
|
interface SelfImproveBudget {
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"index.d.ts","names":[],"sources":["../../src/contract/self-improve.ts","../../src/contract/define-agent-eval.ts","../../src/contract/measured-comparison.ts","../../src/contract/profile-measured-comparison.ts","../../src/contract/intake/run-record-dir.ts","../../src/contract/eval-reporting-suite.ts","../../src/contract/diff.ts","../../src/contract/intake/agent-trace.ts","../../src/contract/intake/code-agent-observation.ts","../../src/contract/intake/code-agent-session.ts","../../src/contract/intake/feedback-table.ts","../../src/contract/intake/otel-spans.ts"],"mappings":"
|
|
1
|
+
{"version":3,"file":"index.d.ts","names":[],"sources":["../../src/contract/self-improve.ts","../../src/contract/define-agent-eval.ts","../../src/contract/measured-comparison.ts","../../src/contract/profile-measured-comparison.ts","../../src/contract/intake/run-record-dir.ts","../../src/contract/eval-reporting-suite.ts","../../src/contract/diff.ts","../../src/contract/intake/agent-trace.ts","../../src/contract/intake/code-agent-observation.ts","../../src/contract/intake/code-agent-session.ts","../../src/contract/intake/feedback-table.ts","../../src/contract/intake/otel-spans.ts"],"mappings":";;;;;;;;;;;;;UA+DiB;;;EAGf;;;;EAIA;;EAEA;;EAEA;;;EAGA;;;EAGA;;;;EAIA;;EAEA,mBAAmB;;;;;;;;;EASnB;;EAEA;;;;;EAKA;;KAGU;EACN;EAA0B;;EAC1B;EAA4B;EAAuB;;EACnD;EAA4B;EAAe;;EAC3C;EAA8B;EAAe;EAAuB;;EAGpE;EAAsB;EAAkB;;EACxC;EAAyB;EAAW;EAAY;EAAa;;UAElD,mBAAmB,kBAAkB,UAAU;;;;;;;;;;;;;;EAc9D,QAAQ,SAAS,gBAAgB,UAAU,WAAW,KAAK,oBAAoB,QAAQ;;;;;;;EAQvF;;;EAIA,WAAW;;;EAIX,OAAO,YAAY,WAAW;;;EAI9B,iBAAiB;;EAGjB,SAAS;;;;;;;;;;;;EAaT,sBAAsB,gCAAgC,WAAW;;;;;EAMjE,WAAW,gBAAgB;;;;;;EAO3B,SAAS,mBAAmB,WAAW;;;EAIvC,qBAAqB;;;EAIrB,OAAO,KAAK,WAAW;;;;;;;;EASvB,cAAc,eAAe,gBAAgB,iBAAiB,mBAAmB;;;;EAKjF,UAAU;;;;EAKV;;;EAIA,gBAAgB,QAAQ;;;;;EAMxB,iBAAiB;IACf,UAAU;IACV;IACA;;;;EAKF;;;EAIA,cAAc,OAAO;;;;EAKrB;EACA;EACA;;;;;;;;;;;;;;EAeA,eAAe;;;EAIf,eAAe;;;;EAKf,eAAe;;EAGf;;;;;;;;EASA;;;;;;;;;EAUA,oBAAoB,uBAAuB,WAAW;;;EAItD,WAAW;;;;;;;EAQX,mBAAmB,uBAAuB,WAAW;;UAGtC,kBAAkB,kBAAkB,UAAU;;;;EAI7D;IACE;IACA,aAAa;;;;;EAKf;IACE;IACA,aAAa;IACb,SAAS;;;IAGT;;;;IAIA;;;;;;EAMF;;;EAGA;;;;EAIA,YAAY;;EAEZ;;;EAGA;;EAEA;;EAEA;;EAEA,MAAM;;;EAGN,UAAU;;EAEV;IACE;IACA,MAAM;IACN;IACA,aAAa;;;;;;;;EAQf,SAAS;;;;;EAKT,QAAQ;;;;;;EAMR,KAAK,yBAAyB,WAAW;;;cAI9B,4BAA4B;WAC9B,MAAM;WACN,UAAU;EAEnB,YAAY,gBAAgB,QAAQ;;;;;;;;;;;;;;;;;;;;;;;;;;;;iBA+LhB,YAAY,kBAAkB,UAAU,WAC5D,MAAM,mBAAmB,WAAW,aACnC,QAAQ,kBAAkB,WAAW;;;KCviB5B,eAAe,kBAAkB,UAAU,cACrD,SAAS,gBACT,UAAU,WACV,KAAK,oBACF,QAAQ;KAED,uBAAuB,kBAAkB,UAAU,aAAa,mBAC1E,WACA;UAGe,yBAAyB,kBAAkB,UAAU,mBAC5D,KACN,eAAe,WAAW;;EAI5B,YAAY;;EAEZ,UAAU;;EAEV,QAAQ,eAAe,WAAW;;EAElC,QAAQ,YAAY,WAAW;;EAE/B,SAAS,YAAY,WAAW;;EAEhC;;KAGU,wBAAwB,kBAAkB,UAAU,aAAa,KAC3E,QAAQ,mBAAmB,WAAW;EAGtC,SAAS,QAAQ;EACjB,eAAe,QAAQ;;UAGR,iBAAiB,kBAAkB,UAAU;;WAEnD,oBAAoB;;WAEpB,iBAAiB;;;;;EAK1B,SACE,OAAO,yBAAyB,WAAW,aAC1C,QAAQ,eAAe,WAAW;;;;;;;EAOrC,QACE,OAAO,wBAAwB,WAAW,aACzC,QAAQ,kBAAkB,WAAW;;;;;;;;;iBAU1B,gBAAgB,kBAAkB,UAAU,WAC1D,UAAU,uBAAuB,WAAW,aAC3C,iBAAiB,WAAW;;;UCvDd;EACf,QAAQ,gCAAgC;EACxC;EACA;;UAGe;EACf,YAAY;EACZ;EACA,QAAQ;EACR,MAAM;EACN,eAAe;EACf;EACA,SAAS;;UAGM;EACf,YAAY;EACZ,QAAQ,OAAO,oCAAoC,QAAQ;;EAE3D;;EAEA,aAAa;EACb,SAAS;;UAGM;EACf,cAAc;EACd;IACE;IACA,MAAM;;;UAIO;EACf,YAAY;EACZ,cAAc;EACd;IACE;IACA,MAAM;;EAER,aAAa;EACb;EACA,YAAY;EACZ;EACA,WAAW;;;UAII,kBAAkB;EACjC;EACA,UAAU;EACV,WAAW;;;UAII,yBAAyB;EACxC,MAAM,KAAK;EACX,WAAW,KAAK;IAAkB;IAAc;;EAChD,QAAQ,KAAK;EACb,eAAe,KAAK,OAAO;EAC3B,UAAU,KAAK;EACf,UAAU,KAAK;EACf,OAAO,KAAK;;UAGG,kCAAkC;EACjD,uBAAuB,kBAAkB;EACzC,QAAQ;EACR,SAAS,yBAAyB;;EAElC;;EAEA,kBAAkB;;EAElB,kBAAkB;;;KAIR,8BAA8B,KACxC;EAGA,iBAAiB;EACjB,WAAW;EACX;;;iBAIc,2BACd,UAAU,sCACT;;iBAQa,4BACd,SAAS,qCACR;;iBAiBa,wBACd,UAAU,mCACT;iBAQa,0BAA0B,iBAAiB;;iBAarC,uBACpB,SAAS,gCACR,QAAQ;;;;;;;;iBAkEK,2BAA2B,MACzC,SAAS,kCAAkC,QAC1C;;iBAwRa,0CACd,SAAS,oCACR;;iBAuEa,oCACd,iBACC;iBAkHa,6BAA6B,iBAAiB;iBAM9C,oCACd,iBACC;iBAkBa,8BAA8B;;;;;;;;;;UCvsB7B;EACf,aAAa;EACb,QAAQ,gCAAgC;EACxC;EACA;;UAGe;EACf,YAAY;EACZ;EACA,aAAa;EACb,MAAM;EACN,SAAS;EACT;EACA,SAAS;;UAGM;EACf,YAAY;EACZ,QACE,OAAO,kDACN,QAAQ;;EAEX;;EAEA,aAAa;EACb,SAAS;;UAGM;EACf,cAAc;EACd;IACE;IACA,MAAM;;;UAIO;EACf,YAAY;EACZ,cAAc;EACd;IACE;IACA,MAAM;;EAER,aAAa;EACb;EACA,YAAY;EACZ;EACA,WAAW;;;iBAIG,gCACd,UAAU,sCACT;;iBAQa,iCACd,SAAS,0CACR;;iBAqBa,sCACd,UAAU,4CACT;iBAOa,kCAAkC,iBAAiB;iBAInD,yCACd,iBACC;iBAIa,wCACd,iBACC;;;;;;iBASmB,qCACpB,SAAS,8CACR,QAAQ;;iBA0CK,wDACd,SAAS,kDACR;;iBAuEa,kDACd,iBACC;;;;UCnPc;;EAEf;;EAEA;;EAEA;;UAGe;;;;;;EAMf;;;;;;;EAOA,WAAW;;;;;;EAMX;;UAGe;;EAEf,MAAM;;EAEN,UAAU;;EAEV;;;;;;;;;;iBAkBoB,iBACpB,cACA,UAAS,0BACR,QAAQ;;;;;KC7CC,0BAA0B;UAErB;;;;;EAKf,UAAU,KAAK;;EAEf,OAAO;;;;;;;;;;EAUP;;;;UAKe;;;EAGf,QAAQ;;EAER;;IAEE;;IAEA;;;IAGA;;IAEA;;;IAGA,UAAU;;;EAGZ;;;;;;;iBAUoB,mBACpB,OAAO,yBACP,UAAS,4BACR,QAAQ;;;;;;UC5DM;EACf;EACA;EACA;;;UAIe;EACf;EACA;EACA;EACA;EACA;;;EAGA,YAAY,eAAe,eAAe;;;;UAK3B;EACf;EACA;EACA;EACA;EACA;;EAEA,SAAS;;EAET,SAAS;;EAET,OAAO;;EAEP;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;;;;UAMe;EACf;EACA;EACA;EACA;EACA,oBAAoB;EACpB,mBAAmB;EACnB;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;EAEA,cAAc;;;;EAId,aAAa;;;;;;;;;iBAoDC,gBACd,QAAQ,2BACR,OAAO,4BACN;;;;;;iBAqEa,SAAS,QAAQ,cAAc,OAAO,eAAe;;;;;;;iBAsCrD,wBAAwB,KAAK,eAAe;;;KC1OhD;UAEK;EACf,MAAM;;EAEN;;UAGe;EACf;EACA;EACA;;;EAGA,cAAc;;UAGC;EACf;EACA,cAAc;EACd,QAAQ;;UAGO;EACf;EACA,eAAe;;UAGA;EACf;EACA;EACA;EACA;IAAQ;IAAc;;EACtB;IAAS;IAAe;;EACxB,OAAO;;;;UAOQ;EACf;;EAEA;;EAEA;EACA;EACA;;EAEA;;EAEA;;KAGU,kBAAkB,YAAY;;;;;;iBAW1B,gBAAgB,SAAS,qBAAqB;UAmE7C;;;;EAIf,SAAS,YAAY;;;EAGrB,cAAc;;;;;;;;iBASA,8BACd,MAAM,aACN,OAAO,kBACN;;;KCjLS;KAEA;KAEA;KAEA;KASA;UAEK;EACf;EACA;EACA;;UAGe;EACf;EACA;EACA,MAAM;EACN,SAAS;EACT;EACA,QAAQ;EACR;EACA;EACA,UAAU;;UAGK;EACf,QAAQ;EACR;EACA;EACA;IACE,QAAQ;IACR;;EAEF,SAAS;;UAGM;EACf,QAAQ;EACR;EACA;EACA,YAAY;;;;;;;iBAeE,wBACd,SAAS,iCACR;;;UCrCc;EACf;EACA;;UAGe;EACf;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf,QAAQ;EACR;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA,WAAW;EACX;;UAGe;EACf,MAAM;EACN,aAAa;EACb,SAAS;EACT,cAAc;;UAGC;EACf;EACA;EACA;EACA;EACA;EACA;EACA,WAAW;EACX;EACA;EACA;EACA;EACA;EACA;;;EAGA,iBAAiB;;;EAGjB,YAAY;;iBAGE,oBAAoB,gBAAgB;iBAepC,iBACd,SAAS,gCACR;iBAIa,sBACd,SAAS,gCACR;iBAIa,oBACd,SAAS,gCACR;iBAIa,oBACd,SAAS,gCACR;iBAIa,cACd,SAAS,gCACR;cAIU,2BAAkB;;;UCpJd;;;EAGf;;EAEA;;;EAGA;;;EAGA,WAAW;;UAGI;EACf;;;EAGA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;EAGA,WAAW;;;EAGX,SAAS;;UAGM;;EAEf,SAAS;;;EAGT,OAAO;;;;EAIP;IAAU;IAAa;;;;;EAIvB;;UAGe;EACf,MAAM;;;EAGN,aAAa;IAAQ;IAAe;IAAe;;;iBAGrC,kBAAkB,MAAM,2BAA2B;;;UCvBlD;EACf,OAAO;;EAEP,eAAe;;EAEf;;;;;;EAMA,eAAe,eAAe,gBAAgB;;iBAGhC,cAAc,MAAM,uBAAuB"}
|
package/dist/contract/index.js
CHANGED
|
@@ -1,17 +1,17 @@
|
|
|
1
|
-
import {
|
|
2
|
-
import {
|
|
1
|
+
import { c as ValidationError } from "../errors-D-LKuDhb.js";
|
|
2
|
+
import { R as createChatClient, t as buildDefaultAnalystRegistry } from "../default-registry-CLXbRt0f.js";
|
|
3
3
|
import { i as estimateCost, o as isModelPriced } from "../metrics-C9YY1OcL.js";
|
|
4
|
-
import { i as CostLedger } from "../cost-ledger-
|
|
4
|
+
import { i as CostLedger } from "../cost-ledger-D-5_-dhi.js";
|
|
5
5
|
import { LLM_MODEL_ATTR_KEYS, SPAN_KIND_ATTR_KEYS } from "../trace-attributes.js";
|
|
6
|
-
import {
|
|
6
|
+
import { M as isOtlpModelCall, j as classifyOtlpSpanRole } from "../tools-DZk2Jn64.js";
|
|
7
7
|
import { l as makeProposalFinding } from "../proposal-findings-DCawte-y.js";
|
|
8
8
|
import { r as mapConcurrentRange } from "../concurrency-MUjT7VjM.js";
|
|
9
|
-
import { B as surfaceContentHash, Ct as llmJudge, Ft as decidePairedPromotion, M as compareOptimizationMethods, O as composeGate, Q as inMemoryCampaignStorage, S as defaultProductionGate, T as heldoutSignificance, V as surfaceHash, X as createRunCostLedger, Y as runCampaign, Z as fsCampaignStorage, _ as buildEvidenceVector, a as emitLoopProvenance, b as powerPreflight, ct as campaignSplitDigest, d as runImprovementLoop, dt as REFERENCE_EQUIVALENCE_INPUT_LIMITS, ft as REFERENCE_EQUIVALENCE_JUDGE_VERSION, g as gepaOptimizationMethod, j as assertOptimizationResult, k as externalTextOptimizationMethod, mt as runReferenceEquivalenceJudge, o as loopProvenanceArgsFromResult, p as runEval, pt as createReferenceEquivalenceJudge, rt as resolveRunDir, t as skillOptOptimizationMethod, v as paretoPolicy, x as heldOutGate, y as paretoSignificanceGate } from "../skillopt-optimization-method-
|
|
10
|
-
import { E as pairedBootstrap } from "../statistics-
|
|
11
|
-
import { i as parseRunRecordSafe, r as modelHasSnapshot } from "../run-record-
|
|
12
|
-
import { a as recordAggregateMeasurements, i as readTaskFailureLabels, o as summarizeExecutionMeasurements, s as summarizeTraceErrors, t as extractUsage } from "../extract-usage-
|
|
13
|
-
import { n as summarizeExecution, t as analyzeRuns } from "../analyze-runs-
|
|
14
|
-
import { a as campaignCellExecutionEvidence, c as campaignCellToRunRecord, o as campaignCellJudgeDimensions, s as campaignCellTaskScore } from "../reward-hacking-
|
|
9
|
+
import { B as surfaceContentHash, Ct as llmJudge, Ft as decidePairedPromotion, M as compareOptimizationMethods, O as composeGate, Q as inMemoryCampaignStorage, S as defaultProductionGate, T as heldoutSignificance, V as surfaceHash, X as createRunCostLedger, Y as runCampaign, Z as fsCampaignStorage, _ as buildEvidenceVector, a as emitLoopProvenance, b as powerPreflight, ct as campaignSplitDigest, d as runImprovementLoop, dt as REFERENCE_EQUIVALENCE_INPUT_LIMITS, ft as REFERENCE_EQUIVALENCE_JUDGE_VERSION, g as gepaOptimizationMethod, j as assertOptimizationResult, k as externalTextOptimizationMethod, mt as runReferenceEquivalenceJudge, o as loopProvenanceArgsFromResult, p as runEval, pt as createReferenceEquivalenceJudge, rt as resolveRunDir, t as skillOptOptimizationMethod, v as paretoPolicy, x as heldOutGate, y as paretoSignificanceGate } from "../skillopt-optimization-method-f4o9sUT4.js";
|
|
10
|
+
import { E as pairedBootstrap } from "../statistics-ByxzSiOM.js";
|
|
11
|
+
import { i as parseRunRecordSafe, r as modelHasSnapshot } from "../run-record-vRgqWmJw.js";
|
|
12
|
+
import { a as recordAggregateMeasurements, i as readTaskFailureLabels, o as summarizeExecutionMeasurements, s as summarizeTraceErrors, t as extractUsage } from "../extract-usage-p-56bh8q.js";
|
|
13
|
+
import { n as summarizeExecution, t as analyzeRuns } from "../analyze-runs-BScZvqMV.js";
|
|
14
|
+
import { a as campaignCellExecutionEvidence, c as campaignCellToRunRecord, o as campaignCellJudgeDimensions, s as campaignCellTaskScore } from "../reward-hacking-GyN0kMd8.js";
|
|
15
15
|
import { n as InMemoryOutcomeStore, t as FileSystemOutcomeStore } from "../outcome-store-ChBKlTd_.js";
|
|
16
16
|
import { t as createHostedClient } from "../client-LIuo-KPv.js";
|
|
17
17
|
import { dirname, join } from "node:path";
|
package/dist/control.d.ts
CHANGED
|
@@ -1,3 +1,3 @@
|
|
|
1
|
-
import {
|
|
2
|
-
import { A as ActionExecutionPolicy, M as evaluateActionPolicy, _ as ProposeReviewReport, a as ProposeReviewControlAction, c as ProposeReviewControlState, g as ProposeReviewConfig, i as scoreFromEvals, j as ActionPolicyDecision, k as runProposeReview, n as RunEvidenceMetadata, o as ProposeReviewControlConfig, r as controlRunToRunRecord, s as ProposeReviewControlResult, t as ControlRunToRunRecordOptions, u as runProposeReviewAsControlLoop } from "./run-evidence-
|
|
1
|
+
import { $ as ControlRunResult, J as ControlActionOutcome, Q as ControlEvalResult, X as ControlContext, Y as ControlBudget, Z as ControlDecision, at as StopDecision, ct as runAgentControlLoop, dt as subjectiveEval, et as ControlRuntimeConfig, it as ControlStopPolicies, lt as stopOnNoProgress, nt as ControlSeverity, ot as allCriticalPassed, q as ControlActionFailureMode, rt as ControlStep, st as objectiveEval, tt as ControlRuntimeError, ut as stopOnRepeatedAction } from "./feedback-trajectory-N_F0PwHz.js";
|
|
2
|
+
import { A as ActionExecutionPolicy, M as evaluateActionPolicy, _ as ProposeReviewReport, a as ProposeReviewControlAction, c as ProposeReviewControlState, g as ProposeReviewConfig, i as scoreFromEvals, j as ActionPolicyDecision, k as runProposeReview, n as RunEvidenceMetadata, o as ProposeReviewControlConfig, r as controlRunToRunRecord, s as ProposeReviewControlResult, t as ControlRunToRunRecordOptions, u as runProposeReviewAsControlLoop } from "./run-evidence-BDIircdA.js";
|
|
3
3
|
export { type ActionExecutionPolicy, type ActionPolicyDecision, type ControlActionFailureMode, type ControlActionOutcome, type ControlBudget, type ControlContext, type ControlDecision, type ControlEvalResult, type ControlRunResult, type ControlRunToRunRecordOptions, type ControlRuntimeConfig, type ControlRuntimeError, type ControlSeverity, type ControlStep, type ControlStopPolicies, type ProposeReviewConfig, type ProposeReviewControlAction, type ProposeReviewControlConfig, type ProposeReviewControlResult, type ProposeReviewControlState, type ProposeReviewReport, type RunEvidenceMetadata, type StopDecision, allCriticalPassed, controlRunToRunRecord, evaluateActionPolicy, objectiveEval, runAgentControlLoop, runProposeReview, runProposeReviewAsControlLoop, scoreFromEvals, stopOnNoProgress, stopOnRepeatedAction, subjectiveEval };
|
package/dist/control.js
CHANGED
|
@@ -1,2 +1,2 @@
|
|
|
1
|
-
import { c as scoreFromEvals, d as runAgentControlLoop, f as stopOnNoProgress, l as allCriticalPassed, m as subjectiveEval, n as runProposeReviewAsControlLoop, o as runProposeReview, p as stopOnRepeatedAction, s as controlRunToRunRecord, u as objectiveEval, y as evaluateActionPolicy } from "./propose-review-control-
|
|
1
|
+
import { c as scoreFromEvals, d as runAgentControlLoop, f as stopOnNoProgress, l as allCriticalPassed, m as subjectiveEval, n as runProposeReviewAsControlLoop, o as runProposeReview, p as stopOnRepeatedAction, s as controlRunToRunRecord, u as objectiveEval, y as evaluateActionPolicy } from "./propose-review-control-DLXz4FCX.js";
|
|
2
2
|
export { allCriticalPassed, controlRunToRunRecord, evaluateActionPolicy, objectiveEval, runAgentControlLoop, runProposeReview, runProposeReviewAsControlLoop, scoreFromEvals, stopOnNoProgress, stopOnRepeatedAction, subjectiveEval };
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import {
|
|
1
|
+
import { c as ValidationError } from "./errors-D-LKuDhb.js";
|
|
2
2
|
import { i as estimateCost, o as isModelPriced, s as resolveModelPricing } from "./metrics-C9YY1OcL.js";
|
|
3
3
|
import { z } from "zod";
|
|
4
4
|
//#region src/cost-ledger.ts
|
|
@@ -836,4 +836,4 @@ function assertString(value, name) {
|
|
|
836
836
|
//#endregion
|
|
837
837
|
export { CostLedgerPersistenceError as a, costForTokenPricing as c, CostLedger as i, costForUsage as l, CostCallConflictError as n, CostReceiptCaptureError as o, CostCeilingReachedError as r, CostReservationExceededError as s, CostAccountingIncompleteError as t, modelPriceKey as u };
|
|
838
838
|
|
|
839
|
-
//# sourceMappingURL=cost-ledger-
|
|
839
|
+
//# sourceMappingURL=cost-ledger-D-5_-dhi.js.map
|