@tangle-network/agent-eval 0.108.1 → 0.110.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/analyst/index.d.ts +10 -12
- package/dist/analyst/index.js +8 -11
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyze-runs-DJYpep3L.d.ts → analyze-runs-Dmz6LA9e.d.ts} +4 -4
- package/dist/{baseline-Bbid3WoO.d.ts → baseline-DsNteOgR.d.ts} +32 -2
- package/dist/belief-state/index.d.ts +6 -6
- package/dist/benchmarks/index.d.ts +4 -4
- package/dist/benchmarks/index.js +7 -8
- package/dist/builder-eval/index.d.ts +4 -4
- package/dist/builder-eval/index.js +1 -2
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/{calibration-BPmzuVPk.d.ts → calibration-Dz8TQV4y.d.ts} +2 -2
- package/dist/campaign/index.d.ts +161 -20
- package/dist/campaign/index.js +15 -8
- package/dist/{chunk-LOJ2QVCE.js → chunk-2IY4ILP4.js} +2 -2
- package/dist/{chunk-LIEJUH2I.js → chunk-6PL5MGDL.js} +9 -9
- package/dist/{chunk-2OGPXHOB.js → chunk-7NX6ZSBG.js} +36 -7
- package/dist/chunk-7NX6ZSBG.js.map +1 -0
- package/dist/{chunk-OVPVM4JC.js → chunk-GTERJI6Q.js} +4 -4
- package/dist/{chunk-YEHAEDUD.js → chunk-IMWDSFUM.js} +604 -2
- package/dist/chunk-IMWDSFUM.js.map +1 -0
- package/dist/{chunk-JZXGWLK5.js → chunk-MHNQWM4I.js} +62 -6
- package/dist/chunk-MHNQWM4I.js.map +1 -0
- package/dist/{chunk-QRVS7MX4.js → chunk-OW47B5WA.js} +3 -5
- package/dist/{chunk-QRVS7MX4.js.map → chunk-OW47B5WA.js.map} +1 -1
- package/dist/{chunk-DBDRR6GF.js → chunk-PLOMR3HP.js} +48 -2
- package/dist/chunk-PLOMR3HP.js.map +1 -0
- package/dist/{chunk-GDZAWO2I.js → chunk-QFGTU7MT.js} +2 -2
- package/dist/{chunk-6SKVFBTR.js → chunk-RNB2NICW.js} +115 -13
- package/dist/chunk-RNB2NICW.js.map +1 -0
- package/dist/{chunk-V7HNA47Z.js → chunk-RSVSSZKF.js} +5 -5
- package/dist/{chunk-5PK3626Q.js → chunk-XRGOKCMO.js} +88 -17
- package/dist/chunk-XRGOKCMO.js.map +1 -0
- package/dist/{code-agent-session-rnJKlqmT.d.ts → code-agent-session-yitf9I-F.d.ts} +1 -1
- package/dist/contract/index.d.ts +20 -23
- package/dist/contract/index.js +11 -13
- package/dist/contract/index.js.map +1 -1
- package/dist/{control-B8UthSBL.d.ts → control-U8LBKUES.d.ts} +5 -6
- package/dist/control.d.ts +8 -9
- package/dist/control.js +6 -8
- package/dist/{dataset-DS7ytHZU.d.ts → dataset-NENEzRgk.d.ts} +1 -1
- package/dist/{default-registry-BswHCXnU.d.ts → default-registry-Bcf1uKVI.d.ts} +1 -2
- package/dist/{emitter-C2rqGH_l.d.ts → emitter-BRchAAAx.d.ts} +2 -2
- package/dist/{failure-cluster-DH9Flgcf.d.ts → failure-cluster-C48PiReX.d.ts} +2 -2
- package/dist/feedback-trajectory-pDcz1lQ1.d.ts +348 -0
- package/dist/{gepa-B3x5Ulcv.d.ts → gepa-BUNP3606.d.ts} +143 -2
- package/dist/hosted/index.d.ts +7 -7
- package/dist/{index-pPtfoIJO.d.ts → index-Dc3VLGhp.d.ts} +2 -2
- package/dist/index.d.ts +645 -61
- package/dist/index.js +1282 -190
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-B4xrdwEK.d.ts → insight-report-D4cXFsLt.d.ts} +1 -1
- package/dist/{integrity-DqGZg3st.d.ts → integrity-qemeBAyx.d.ts} +1 -1
- package/dist/{types-D1ytG0Yg.d.ts → kind-factory-20hcaYpf.d.ts} +169 -2
- package/dist/meta-eval/index.d.ts +5 -5
- package/dist/meta-eval/index.js +1 -2
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/{multi-layer-verifier-CI4jdX-q.d.ts → multi-layer-verifier-BsqKuLyN.d.ts} +1 -1
- package/dist/multishot/index.d.ts +3 -3
- package/dist/openapi.json +1 -1
- package/dist/pipelines/index.d.ts +6 -7
- package/dist/pipelines/index.js +3 -6
- package/dist/pipelines/index.js.map +1 -1
- package/dist/{policy-edit-DQUXYMDm.d.ts → policy-edit-D2bBDZDf.d.ts} +2 -2
- package/dist/{pre-registration-BUhVPzE7.d.ts → pre-registration-BepVVa6P.d.ts} +3 -3
- package/dist/{provenance-DdDhf6cg.d.ts → provenance-DMvsfknv.d.ts} +3 -5
- package/dist/{query-0aTmbmQe.d.ts → query-Ck190MOd.d.ts} +2 -2
- package/dist/{release-report-DeJpsBiA.d.ts → release-report-oBfOz8ku.d.ts} +3 -3
- package/dist/reporting.d.ts +8 -8
- package/dist/{researcher-Wc7dx6GM.d.ts → researcher-CaH0CwFC.d.ts} +6 -6
- package/dist/rl.d.ts +568 -15
- package/dist/rl.js +4 -4
- package/dist/{rubric-predictive-validity-DPnyG-CE.d.ts → rubric-predictive-validity-C-fMteAW.d.ts} +1 -1
- package/dist/{run-record-I-Z3JNvO.d.ts → run-record-DksGsfgv.d.ts} +1 -1
- package/dist/{runtime-trajectory-iW9IhV3e.d.ts → runtime-trajectory-h5i0SZUj.d.ts} +1 -1
- package/dist/{schema-m0gsnbt3.d.ts → schema-SGWcK9wa.d.ts} +1 -1
- package/dist/{semantic-concept-judge-BmNZPB_j.d.ts → semantic-concept-judge-D7z6JCLZ.d.ts} +57 -4
- package/dist/{store-BcFXE6LG.d.ts → store-BsVi7ncX.d.ts} +1 -1
- package/dist/storyboard/index.d.ts +1 -1
- package/dist/{summary-report-QMZVe3P-.d.ts → summary-report-Bz-0-t8v.d.ts} +2 -2
- package/dist/{test-graded-scenario-DeODGLra.d.ts → test-graded-scenario-mzYBKspu.d.ts} +3 -3
- package/dist/traces.d.ts +54 -11
- package/dist/traces.js +25 -27
- package/dist/{types-BdIv5dvA.d.ts → types-v--ctu-b.d.ts} +2 -2
- package/dist/wire/index.d.ts +5 -6
- package/package.json +1 -71
- package/dist/adapters/http.d.ts +0 -142
- package/dist/adapters/http.js +0 -203
- package/dist/adapters/http.js.map +0 -1
- package/dist/adapters/langchain.d.ts +0 -95
- package/dist/adapters/langchain.js +0 -34
- package/dist/adapters/langchain.js.map +0 -1
- package/dist/adapters/otel.d.ts +0 -112
- package/dist/adapters/otel.js +0 -110
- package/dist/adapters/otel.js.map +0 -1
- package/dist/chunk-2OGPXHOB.js.map +0 -1
- package/dist/chunk-45EEMHTC.js +0 -35
- package/dist/chunk-45EEMHTC.js.map +0 -1
- package/dist/chunk-5BKGXME7.js +0 -65
- package/dist/chunk-5BKGXME7.js.map +0 -1
- package/dist/chunk-5PK3626Q.js.map +0 -1
- package/dist/chunk-6SK5VFYK.js +0 -100
- package/dist/chunk-6SK5VFYK.js.map +0 -1
- package/dist/chunk-6SKVFBTR.js.map +0 -1
- package/dist/chunk-DBDRR6GF.js.map +0 -1
- package/dist/chunk-DJWX3GVS.js +0 -81
- package/dist/chunk-DJWX3GVS.js.map +0 -1
- package/dist/chunk-FOUG2VVS.js +0 -855
- package/dist/chunk-FOUG2VVS.js.map +0 -1
- package/dist/chunk-JZXGWLK5.js.map +0 -1
- package/dist/chunk-K7QEIHHJ.js +0 -613
- package/dist/chunk-K7QEIHHJ.js.map +0 -1
- package/dist/chunk-KKHDIONI.js +0 -414
- package/dist/chunk-KKHDIONI.js.map +0 -1
- package/dist/chunk-KMPRBJK4.js +0 -74
- package/dist/chunk-KMPRBJK4.js.map +0 -1
- package/dist/chunk-Q2JRAWRI.js +0 -196
- package/dist/chunk-Q2JRAWRI.js.map +0 -1
- package/dist/chunk-RZTMDUO7.js +0 -49
- package/dist/chunk-RZTMDUO7.js.map +0 -1
- package/dist/chunk-STGVSCDH.js +0 -202
- package/dist/chunk-STGVSCDH.js.map +0 -1
- package/dist/chunk-YEHAEDUD.js.map +0 -1
- package/dist/control-runtime-Acf9CGhw.d.ts +0 -182
- package/dist/corpus-eBVwhCp1.d.ts +0 -560
- package/dist/counterfactual-DlOz8PBx.d.ts +0 -85
- package/dist/diagnose.d.ts +0 -252
- package/dist/diagnose.js +0 -382
- package/dist/diagnose.js.map +0 -1
- package/dist/feedback-trajectory-C9KCo8ag.d.ts +0 -169
- package/dist/governance/index.d.ts +0 -135
- package/dist/governance/index.js +0 -18
- package/dist/governance/index.js.map +0 -1
- package/dist/groundedness/index.d.ts +0 -112
- package/dist/groundedness/index.js +0 -77
- package/dist/groundedness/index.js.map +0 -1
- package/dist/harness-optimizer-mOl9XX_O.d.ts +0 -106
- package/dist/kind-factory-DvIGo_cP.d.ts +0 -171
- package/dist/knowledge/index.d.ts +0 -103
- package/dist/knowledge/index.js +0 -18
- package/dist/knowledge/index.js.map +0 -1
- package/dist/pareto-E-pembql.d.ts +0 -81
- package/dist/perf/index.d.ts +0 -123
- package/dist/perf/index.js +0 -18
- package/dist/perf/index.js.map +0 -1
- package/dist/prm/index.d.ts +0 -104
- package/dist/prm/index.js +0 -265
- package/dist/prm/index.js.map +0 -1
- package/dist/product-benchmark/index.d.ts +0 -247
- package/dist/product-benchmark/index.js +0 -37
- package/dist/product-benchmark/index.js.map +0 -1
- package/dist/red-team-KmmiqBlY.d.ts +0 -63
- package/dist/redact-B40YG2M_.d.ts +0 -45
- package/dist/rubric-Cc6UHvUb.d.ts +0 -73
- package/dist/run-critic-CmMf05uV.d.ts +0 -56
- package/dist/sink-fetch-B1Yg4Til.d.ts +0 -101
- package/dist/telemetry/file.d.ts +0 -19
- package/dist/telemetry/file.js +0 -45
- package/dist/telemetry/file.js.map +0 -1
- package/dist/telemetry/index.d.ts +0 -38
- package/dist/telemetry/index.js +0 -130
- package/dist/telemetry/index.js.map +0 -1
- package/dist/testing-C21CHsq2.d.ts +0 -20
- package/dist/testing.d.ts +0 -1
- package/dist/testing.js +0 -8
- package/dist/testing.js.map +0 -1
- package/dist/trajectory-2TkpSEVh.d.ts +0 -33
- package/dist/workflow/index.d.ts +0 -496
- package/dist/workflow/index.js +0 -2178
- package/dist/workflow/index.js.map +0 -1
- /package/dist/{chunk-LOJ2QVCE.js.map → chunk-2IY4ILP4.js.map} +0 -0
- /package/dist/{chunk-LIEJUH2I.js.map → chunk-6PL5MGDL.js.map} +0 -0
- /package/dist/{chunk-OVPVM4JC.js.map → chunk-GTERJI6Q.js.map} +0 -0
- /package/dist/{chunk-GDZAWO2I.js.map → chunk-QFGTU7MT.js.map} +0 -0
- /package/dist/{chunk-V7HNA47Z.js.map → chunk-RSVSSZKF.js.map} +0 -0
package/dist/index.js
CHANGED
|
@@ -9,52 +9,34 @@ import {
|
|
|
9
9
|
checkBehavioralCanary,
|
|
10
10
|
checkCanaries,
|
|
11
11
|
runBehavioralCanaries
|
|
12
|
-
} from "./chunk-
|
|
13
|
-
import {
|
|
14
|
-
classifyEuAiRisk,
|
|
15
|
-
euAiActReport,
|
|
16
|
-
nistAiRmfReport,
|
|
17
|
-
renderMarkdown,
|
|
18
|
-
soc2Report,
|
|
19
|
-
summarize
|
|
20
|
-
} from "./chunk-KKHDIONI.js";
|
|
21
|
-
import {
|
|
22
|
-
acquisitionPlansForKnowledgeGaps,
|
|
23
|
-
blockingKnowledgeEval,
|
|
24
|
-
knowledgeReadinessTracePayload,
|
|
25
|
-
scoreKnowledgeReadiness,
|
|
26
|
-
userQuestionsForKnowledgeGaps
|
|
27
|
-
} from "./chunk-Q2JRAWRI.js";
|
|
28
|
-
import {
|
|
29
|
-
assertRecordIntegrity,
|
|
30
|
-
checkRecordIntegrity,
|
|
31
|
-
expandMatrix,
|
|
32
|
-
gatePerf,
|
|
33
|
-
scenarioKey,
|
|
34
|
-
summarizeRecords
|
|
35
|
-
} from "./chunk-STGVSCDH.js";
|
|
36
|
-
import {
|
|
37
|
-
assertProductBenchmarkRun,
|
|
38
|
-
buildProductBenchmarkManifest,
|
|
39
|
-
exportProductBenchmark,
|
|
40
|
-
exportProductBenchmarkRuns,
|
|
41
|
-
findProductBenchmarkArtifacts,
|
|
42
|
-
productBenchmarkIntegrityFailures,
|
|
43
|
-
productBenchmarkMutableSurfaces,
|
|
44
|
-
productBenchmarkRepoIdentity,
|
|
45
|
-
productBenchmarkSplits,
|
|
46
|
-
readProductBenchmarkManifest,
|
|
47
|
-
readProductBenchmarkRecords,
|
|
48
|
-
runRecordToProductBenchmarkRecord,
|
|
49
|
-
validateProductBenchmarkManifest,
|
|
50
|
-
validateProductBenchmarkRecord,
|
|
51
|
-
validateProductBenchmarkRun
|
|
52
|
-
} from "./chunk-FOUG2VVS.js";
|
|
12
|
+
} from "./chunk-QFGTU7MT.js";
|
|
53
13
|
import {
|
|
54
14
|
BENCHMARK_SPLIT_SEED,
|
|
55
15
|
benchmarks_exports,
|
|
56
16
|
deterministicSplit
|
|
57
17
|
} from "./chunk-T6W5ADLG.js";
|
|
18
|
+
import {
|
|
19
|
+
DEFAULT_RULES,
|
|
20
|
+
buildTrajectory,
|
|
21
|
+
classifyFailure,
|
|
22
|
+
compareToBaseline,
|
|
23
|
+
computeToolUseMetrics,
|
|
24
|
+
iqr,
|
|
25
|
+
welchsTTest
|
|
26
|
+
} from "./chunk-PLOMR3HP.js";
|
|
27
|
+
import {
|
|
28
|
+
analyzeSeries
|
|
29
|
+
} from "./chunk-BOD4O7OF.js";
|
|
30
|
+
import {
|
|
31
|
+
DockerSandboxDriver,
|
|
32
|
+
SandboxHarness,
|
|
33
|
+
SubprocessSandboxDriver,
|
|
34
|
+
composeParsers,
|
|
35
|
+
jestTestParser,
|
|
36
|
+
pytestTestParser,
|
|
37
|
+
runTestGradedScenario,
|
|
38
|
+
vitestTestParser
|
|
39
|
+
} from "./chunk-HZHNRYHK.js";
|
|
58
40
|
import {
|
|
59
41
|
CODING_HARNESSES,
|
|
60
42
|
HARNESS_NATIVE_MODEL,
|
|
@@ -77,7 +59,7 @@ import {
|
|
|
77
59
|
llmJudge,
|
|
78
60
|
parseCorrectnessResponse,
|
|
79
61
|
verifyCompletion
|
|
80
|
-
} from "./chunk-
|
|
62
|
+
} from "./chunk-RNB2NICW.js";
|
|
81
63
|
import {
|
|
82
64
|
DEFAULT_MUTATION_PRIMITIVES,
|
|
83
65
|
DEFAULT_RED_TEAM_CORPUS,
|
|
@@ -101,17 +83,7 @@ import {
|
|
|
101
83
|
scoreRedTeamOutput,
|
|
102
84
|
surfaceContentHash,
|
|
103
85
|
toolNamesForRun
|
|
104
|
-
} from "./chunk-
|
|
105
|
-
import {
|
|
106
|
-
BackendIntegrityError,
|
|
107
|
-
assertRealBackend,
|
|
108
|
-
cachedJudge,
|
|
109
|
-
canonicalJson,
|
|
110
|
-
contentHash,
|
|
111
|
-
fileVerdictCache,
|
|
112
|
-
inMemoryVerdictCache,
|
|
113
|
-
summarizeBackendIntegrity
|
|
114
|
-
} from "./chunk-3PFZBGMR.js";
|
|
86
|
+
} from "./chunk-GTERJI6Q.js";
|
|
115
87
|
import {
|
|
116
88
|
MODEL_PRICING,
|
|
117
89
|
MetricsCollector,
|
|
@@ -122,33 +94,20 @@ import {
|
|
|
122
94
|
resolveModelPricing
|
|
123
95
|
} from "./chunk-VI2UW6B6.js";
|
|
124
96
|
import {
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
} from "./chunk-BOD4O7OF.js";
|
|
135
|
-
import {
|
|
136
|
-
exportTrainingData,
|
|
137
|
-
toNdjson
|
|
138
|
-
} from "./chunk-KMPRBJK4.js";
|
|
139
|
-
import {
|
|
140
|
-
DockerSandboxDriver,
|
|
141
|
-
SandboxHarness,
|
|
142
|
-
SubprocessSandboxDriver,
|
|
143
|
-
composeParsers,
|
|
144
|
-
jestTestParser,
|
|
145
|
-
pytestTestParser,
|
|
146
|
-
runTestGradedScenario,
|
|
147
|
-
vitestTestParser
|
|
148
|
-
} from "./chunk-HZHNRYHK.js";
|
|
97
|
+
BackendIntegrityError,
|
|
98
|
+
assertRealBackend,
|
|
99
|
+
cachedJudge,
|
|
100
|
+
canonicalJson,
|
|
101
|
+
contentHash,
|
|
102
|
+
fileVerdictCache,
|
|
103
|
+
inMemoryVerdictCache,
|
|
104
|
+
summarizeBackendIntegrity
|
|
105
|
+
} from "./chunk-3PFZBGMR.js";
|
|
149
106
|
import {
|
|
150
107
|
DEFAULT_COMPLEXITY_WEIGHTS,
|
|
151
108
|
FindingsStore,
|
|
109
|
+
LockedJsonlAppender,
|
|
110
|
+
Mutex,
|
|
152
111
|
RunCritic,
|
|
153
112
|
SEMANTIC_CONCEPT_JUDGE_VERSION,
|
|
154
113
|
SKILL_USAGE_ANALYST,
|
|
@@ -159,15 +118,11 @@ import {
|
|
|
159
118
|
defaultIsMaterial,
|
|
160
119
|
diffFindings,
|
|
161
120
|
runSemanticConceptJudge
|
|
162
|
-
} from "./chunk-
|
|
163
|
-
import {
|
|
164
|
-
LockedJsonlAppender,
|
|
165
|
-
Mutex
|
|
166
|
-
} from "./chunk-DJWX3GVS.js";
|
|
121
|
+
} from "./chunk-XRGOKCMO.js";
|
|
167
122
|
import {
|
|
168
123
|
buildDefaultAnalystRegistry,
|
|
169
124
|
computeTraceMetrics
|
|
170
|
-
} from "./chunk-
|
|
125
|
+
} from "./chunk-OW47B5WA.js";
|
|
171
126
|
import {
|
|
172
127
|
DEFAULT_RUN_SCORE_WEIGHTS,
|
|
173
128
|
POLICY_EDIT_AXES,
|
|
@@ -184,7 +139,7 @@ import {
|
|
|
184
139
|
policyEditsFromFindings,
|
|
185
140
|
scorePolicyEditReadiness,
|
|
186
141
|
validatePolicyEdit
|
|
187
|
-
} from "./chunk-
|
|
142
|
+
} from "./chunk-2IY4ILP4.js";
|
|
188
143
|
import {
|
|
189
144
|
AnalystRegistry,
|
|
190
145
|
DEFAULT_TRACE_ANALYST_KINDS,
|
|
@@ -192,32 +147,32 @@ import {
|
|
|
192
147
|
IMPROVEMENT_KIND_SPEC,
|
|
193
148
|
KNOWLEDGE_GAP_KIND_SPEC,
|
|
194
149
|
KNOWLEDGE_POISONING_KIND_SPEC,
|
|
150
|
+
computeFindingId,
|
|
195
151
|
createTraceAnalystKind,
|
|
152
|
+
makeFinding,
|
|
196
153
|
renderPriorFindings
|
|
197
|
-
} from "./chunk-
|
|
154
|
+
} from "./chunk-7NX6ZSBG.js";
|
|
198
155
|
import {
|
|
156
|
+
allCriticalPassed,
|
|
199
157
|
controlFailureClassFromVerification,
|
|
200
158
|
controlRunToRunRecord,
|
|
201
159
|
createLlmReviewer,
|
|
160
|
+
errorStreakDetector,
|
|
202
161
|
evaluateActionPolicy,
|
|
203
162
|
inMemoryReviewStore,
|
|
204
163
|
jsonlReviewStore,
|
|
205
|
-
runProposeReview,
|
|
206
|
-
runProposeReviewAsControlLoop,
|
|
207
|
-
scoreFromEvals
|
|
208
|
-
} from "./chunk-K7QEIHHJ.js";
|
|
209
|
-
import {
|
|
210
|
-
allCriticalPassed,
|
|
211
|
-
errorStreakDetector,
|
|
212
164
|
noProgressDetector,
|
|
213
165
|
objectiveEval,
|
|
214
166
|
observeAll,
|
|
215
167
|
repeatedActionDetector,
|
|
216
168
|
runAgentControlLoop,
|
|
169
|
+
runProposeReview,
|
|
170
|
+
runProposeReviewAsControlLoop,
|
|
171
|
+
scoreFromEvals,
|
|
217
172
|
stopOnNoProgress,
|
|
218
173
|
stopOnRepeatedAction,
|
|
219
174
|
subjectiveEval
|
|
220
|
-
} from "./chunk-
|
|
175
|
+
} from "./chunk-IMWDSFUM.js";
|
|
221
176
|
import {
|
|
222
177
|
assertReleaseConfidence,
|
|
223
178
|
bootstrapCi,
|
|
@@ -227,20 +182,8 @@ import {
|
|
|
227
182
|
} from "./chunk-6HFYGEZ3.js";
|
|
228
183
|
import {
|
|
229
184
|
runEvalCampaign
|
|
230
|
-
} from "./chunk-
|
|
185
|
+
} from "./chunk-6PL5MGDL.js";
|
|
231
186
|
import "./chunk-N22ZO7FV.js";
|
|
232
|
-
import {
|
|
233
|
-
LlmCallError,
|
|
234
|
-
LlmClient,
|
|
235
|
-
LlmRouteAssertionError,
|
|
236
|
-
assertLlmRoute,
|
|
237
|
-
backoffMs,
|
|
238
|
-
callLlm,
|
|
239
|
-
callLlmJson,
|
|
240
|
-
isTransientLlmError,
|
|
241
|
-
probeLlm,
|
|
242
|
-
stripFencedJson
|
|
243
|
-
} from "./chunk-FUCQVFMU.js";
|
|
244
187
|
import {
|
|
245
188
|
evaluateInterimReleaseConfidence,
|
|
246
189
|
pairedEvalueSequence
|
|
@@ -252,17 +195,6 @@ import {
|
|
|
252
195
|
researchReport,
|
|
253
196
|
summaryTable
|
|
254
197
|
} from "./chunk-6MFFKPZ4.js";
|
|
255
|
-
import {
|
|
256
|
-
attributeCounterfactuals,
|
|
257
|
-
runCounterfactual
|
|
258
|
-
} from "./chunk-6SK5VFYK.js";
|
|
259
|
-
import {
|
|
260
|
-
buildTrajectory
|
|
261
|
-
} from "./chunk-RZTMDUO7.js";
|
|
262
|
-
import {
|
|
263
|
-
computeFindingId,
|
|
264
|
-
makeFinding
|
|
265
|
-
} from "./chunk-45EEMHTC.js";
|
|
266
198
|
import {
|
|
267
199
|
benjaminiHochberg,
|
|
268
200
|
bonferroni,
|
|
@@ -331,7 +263,24 @@ import {
|
|
|
331
263
|
scoreTraceInsightReadiness,
|
|
332
264
|
tokenizeDomainWords,
|
|
333
265
|
traceAnalystOnRunComplete
|
|
334
|
-
} from "./chunk-
|
|
266
|
+
} from "./chunk-RSVSSZKF.js";
|
|
267
|
+
import {
|
|
268
|
+
FAILURE_CLASSES,
|
|
269
|
+
TRACE_SCHEMA_VERSION,
|
|
270
|
+
aggregateLlm,
|
|
271
|
+
argHash,
|
|
272
|
+
groupBy,
|
|
273
|
+
isJudgeSpan,
|
|
274
|
+
isLlmSpan,
|
|
275
|
+
isRetrievalSpan,
|
|
276
|
+
isSandboxSpan,
|
|
277
|
+
isToolSpan,
|
|
278
|
+
judgeSpans,
|
|
279
|
+
llmSpans,
|
|
280
|
+
runFailureClass,
|
|
281
|
+
runsForScenario,
|
|
282
|
+
toolSpans
|
|
283
|
+
} from "./chunk-MHNQWM4I.js";
|
|
335
284
|
import {
|
|
336
285
|
TRACE_ANALYST_ACTOR_DESCRIPTION,
|
|
337
286
|
TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION,
|
|
@@ -344,25 +293,6 @@ import {
|
|
|
344
293
|
redactString,
|
|
345
294
|
redactValue
|
|
346
295
|
} from "./chunk-GGE4NNQT.js";
|
|
347
|
-
import {
|
|
348
|
-
aggregateLlm,
|
|
349
|
-
argHash,
|
|
350
|
-
groupBy,
|
|
351
|
-
judgeSpans,
|
|
352
|
-
llmSpans,
|
|
353
|
-
runFailureClass,
|
|
354
|
-
runsForScenario,
|
|
355
|
-
toolSpans
|
|
356
|
-
} from "./chunk-JZXGWLK5.js";
|
|
357
|
-
import {
|
|
358
|
-
FAILURE_CLASSES,
|
|
359
|
-
TRACE_SCHEMA_VERSION,
|
|
360
|
-
isJudgeSpan,
|
|
361
|
-
isLlmSpan,
|
|
362
|
-
isRetrievalSpan,
|
|
363
|
-
isSandboxSpan,
|
|
364
|
-
isToolSpan
|
|
365
|
-
} from "./chunk-5BKGXME7.js";
|
|
366
296
|
import {
|
|
367
297
|
DEFAULT_TRACE_ANALYST_BUDGETS,
|
|
368
298
|
LLM_CACHED_TOKENS,
|
|
@@ -403,12 +333,9 @@ import {
|
|
|
403
333
|
throwIfRunIncomplete
|
|
404
334
|
} from "./chunk-TT4KNT67.js";
|
|
405
335
|
import {
|
|
406
|
-
|
|
407
|
-
|
|
408
|
-
|
|
409
|
-
defaultProviderRedactor,
|
|
410
|
-
providerFromBaseUrl
|
|
411
|
-
} from "./chunk-PC4UYEBM.js";
|
|
336
|
+
TraceEmitter,
|
|
337
|
+
llmSpanFromProvider
|
|
338
|
+
} from "./chunk-TVVP3ZZQ.js";
|
|
412
339
|
import {
|
|
413
340
|
RunRecordValidationError,
|
|
414
341
|
isRunRecord,
|
|
@@ -417,10 +344,6 @@ import {
|
|
|
417
344
|
roundTripRunRecord,
|
|
418
345
|
validateRunRecord
|
|
419
346
|
} from "./chunk-VK6HBGAE.js";
|
|
420
|
-
import {
|
|
421
|
-
TraceEmitter,
|
|
422
|
-
llmSpanFromProvider
|
|
423
|
-
} from "./chunk-TVVP3ZZQ.js";
|
|
424
347
|
import {
|
|
425
348
|
AGENT_PROFILE_KINDS,
|
|
426
349
|
AgentProfileCellValidationError,
|
|
@@ -442,6 +365,25 @@ import {
|
|
|
442
365
|
signManifest,
|
|
443
366
|
verifyManifest
|
|
444
367
|
} from "./chunk-VSMTAMNK.js";
|
|
368
|
+
import {
|
|
369
|
+
LlmCallError,
|
|
370
|
+
LlmClient,
|
|
371
|
+
LlmRouteAssertionError,
|
|
372
|
+
assertLlmRoute,
|
|
373
|
+
backoffMs,
|
|
374
|
+
callLlm,
|
|
375
|
+
callLlmJson,
|
|
376
|
+
isTransientLlmError,
|
|
377
|
+
probeLlm,
|
|
378
|
+
stripFencedJson
|
|
379
|
+
} from "./chunk-FUCQVFMU.js";
|
|
380
|
+
import {
|
|
381
|
+
FileSystemRawProviderSink,
|
|
382
|
+
InMemoryRawProviderSink,
|
|
383
|
+
NoopRawProviderSink,
|
|
384
|
+
defaultProviderRedactor,
|
|
385
|
+
providerFromBaseUrl
|
|
386
|
+
} from "./chunk-PC4UYEBM.js";
|
|
445
387
|
import {
|
|
446
388
|
AgentEvalError,
|
|
447
389
|
CaptureIntegrityError,
|
|
@@ -654,12 +596,12 @@ function ghCliClient(opts = {}) {
|
|
|
654
596
|
await exec("git", ["branch", "-D", input.branchName], { cwd });
|
|
655
597
|
await run("git", ["checkout", "-b", input.branchName]);
|
|
656
598
|
const { mkdir, writeFile } = await import("fs/promises");
|
|
657
|
-
const { dirname:
|
|
599
|
+
const { dirname: dirname5, join: join6, resolve: resolve2 } = await import("path");
|
|
658
600
|
for (const change of input.fileChanges) {
|
|
659
|
-
const abs =
|
|
660
|
-
await mkdir(
|
|
601
|
+
const abs = resolve2(cwd, change.path);
|
|
602
|
+
await mkdir(dirname5(abs), { recursive: true });
|
|
661
603
|
await writeFile(abs, change.contents, "utf8");
|
|
662
|
-
await run("git", ["add",
|
|
604
|
+
await run("git", ["add", join6(change.path)]);
|
|
663
605
|
}
|
|
664
606
|
const env = {};
|
|
665
607
|
if (input.authorName) env.GIT_AUTHOR_NAME = input.authorName;
|
|
@@ -1121,7 +1063,7 @@ function capabilityHeadroom(rows, opts = {}) {
|
|
|
1121
1063
|
tasksWithGap: tasks.filter((t) => t.headroom === "gap").length,
|
|
1122
1064
|
tasksSaturated: tasks.filter((t) => t.headroom === "saturated").length,
|
|
1123
1065
|
tasksUnknown: tasks.filter((t) => t.headroom === "unknown").length,
|
|
1124
|
-
repsUnknown: tasks.reduce((
|
|
1066
|
+
repsUnknown: tasks.reduce((sum4, t) => sum4 + (t.n - t.nKnown), 0)
|
|
1125
1067
|
};
|
|
1126
1068
|
return { tasks, summary };
|
|
1127
1069
|
}
|
|
@@ -1711,10 +1653,10 @@ var FileSystemFeedbackTrajectoryStore = class {
|
|
|
1711
1653
|
}
|
|
1712
1654
|
async append(record) {
|
|
1713
1655
|
const { appendFile, mkdir } = await import("fs/promises");
|
|
1714
|
-
const { join:
|
|
1656
|
+
const { join: join6 } = await import("path");
|
|
1715
1657
|
await mkdir(this.dir, { recursive: true });
|
|
1716
1658
|
await appendFile(
|
|
1717
|
-
|
|
1659
|
+
join6(this.dir, "feedback-trajectories.ndjson"),
|
|
1718
1660
|
`${JSON.stringify(record)}
|
|
1719
1661
|
`,
|
|
1720
1662
|
"utf8"
|
|
@@ -1723,8 +1665,8 @@ var FileSystemFeedbackTrajectoryStore = class {
|
|
|
1723
1665
|
async load() {
|
|
1724
1666
|
if (this.loaded) return;
|
|
1725
1667
|
const { readFile: readFile2 } = await import("fs/promises");
|
|
1726
|
-
const { join:
|
|
1727
|
-
const file =
|
|
1668
|
+
const { join: join6 } = await import("path");
|
|
1669
|
+
const file = join6(this.dir, "feedback-trajectories.ndjson");
|
|
1728
1670
|
try {
|
|
1729
1671
|
const raw = await readFile2(file, "utf8");
|
|
1730
1672
|
for (const line of raw.split("\n")) {
|
|
@@ -1971,7 +1913,7 @@ function scoreFromLabels(labels) {
|
|
|
1971
1913
|
return void 0;
|
|
1972
1914
|
}).filter((value) => typeof value === "number");
|
|
1973
1915
|
if (!scored.length) return void 0;
|
|
1974
|
-
return Math.round(scored.reduce((
|
|
1916
|
+
return Math.round(scored.reduce((sum4, value) => sum4 + value, 0) / scored.length * 1e3) / 1e3;
|
|
1975
1917
|
}
|
|
1976
1918
|
function instructionFromLabel(trajectory, label) {
|
|
1977
1919
|
if (label.kind === "reject" && label.reason)
|
|
@@ -2260,6 +2202,190 @@ function assertCrossFamily(models, opts = {}) {
|
|
|
2260
2202
|
return list;
|
|
2261
2203
|
}
|
|
2262
2204
|
|
|
2205
|
+
// src/knowledge/readiness.ts
|
|
2206
|
+
function scoreKnowledgeReadiness(options) {
|
|
2207
|
+
const now = options.now ?? /* @__PURE__ */ new Date();
|
|
2208
|
+
const requirements = options.requirements.map(normalizeRequirement);
|
|
2209
|
+
const missing = requirements.filter((requirement) => isRequirementMissing(requirement, now));
|
|
2210
|
+
const blockingMissingRequirements = missing.filter(isBlockingGap);
|
|
2211
|
+
const nonBlockingGaps = missing.filter((requirement) => !isBlockingGap(requirement));
|
|
2212
|
+
const readinessScore = weightedReadinessAt(requirements, now);
|
|
2213
|
+
const bundle = {
|
|
2214
|
+
taskId: options.taskId,
|
|
2215
|
+
requirements,
|
|
2216
|
+
evidenceIds: unique([
|
|
2217
|
+
...options.evidenceIds ?? [],
|
|
2218
|
+
...requirements.flatMap((r) => r.evidenceIds)
|
|
2219
|
+
]),
|
|
2220
|
+
claimIds: unique(options.claimIds ?? []),
|
|
2221
|
+
wikiPageIds: unique(options.wikiPageIds ?? []),
|
|
2222
|
+
userAnswers: options.userAnswers ?? {},
|
|
2223
|
+
missing,
|
|
2224
|
+
readinessScore,
|
|
2225
|
+
metadata: options.metadata
|
|
2226
|
+
};
|
|
2227
|
+
const recommendedAction = chooseRecommendedAction(blockingMissingRequirements, nonBlockingGaps);
|
|
2228
|
+
const severity = blockingMissingRequirements.length > 0 ? "critical" : nonBlockingGaps.some((gap) => gap.importance === "high") ? "warning" : "info";
|
|
2229
|
+
const reason = blockingMissingRequirements.length > 0 ? `${blockingMissingRequirements.length} blocking knowledge requirement(s) are missing.` : nonBlockingGaps.length > 0 ? `${nonBlockingGaps.length} non-blocking knowledge gap(s) remain.` : "All declared knowledge requirements are ready.";
|
|
2230
|
+
return {
|
|
2231
|
+
taskId: options.taskId,
|
|
2232
|
+
readinessScore,
|
|
2233
|
+
blockingMissingRequirements,
|
|
2234
|
+
nonBlockingGaps,
|
|
2235
|
+
recommendedAction,
|
|
2236
|
+
bundle,
|
|
2237
|
+
severity,
|
|
2238
|
+
reason
|
|
2239
|
+
};
|
|
2240
|
+
}
|
|
2241
|
+
function blockingKnowledgeEval(report, options = {}) {
|
|
2242
|
+
const minimumScore = options.minimumScore ?? 0.7;
|
|
2243
|
+
const passed = report.blockingMissingRequirements.length === 0 && report.readinessScore >= minimumScore;
|
|
2244
|
+
if (options.emitter) {
|
|
2245
|
+
void options.emitter.emit({
|
|
2246
|
+
kind: "custom",
|
|
2247
|
+
payload: knowledgeReadinessTracePayload(report, { passed, minimumScore })
|
|
2248
|
+
}).catch(() => void 0);
|
|
2249
|
+
}
|
|
2250
|
+
return objectiveEval({
|
|
2251
|
+
id: options.id ?? "knowledge-ready",
|
|
2252
|
+
passed,
|
|
2253
|
+
score: report.readinessScore,
|
|
2254
|
+
severity: passed ? "info" : report.severity,
|
|
2255
|
+
detail: report.reason,
|
|
2256
|
+
evidence: report.blockingMissingRequirements.map((r) => r.id).join(", ") || void 0,
|
|
2257
|
+
metadata: { knowledgeReadiness: report }
|
|
2258
|
+
});
|
|
2259
|
+
}
|
|
2260
|
+
function knowledgeReadinessTracePayload(report, options = {}) {
|
|
2261
|
+
return {
|
|
2262
|
+
kind: "readiness_scored",
|
|
2263
|
+
taskId: report.taskId,
|
|
2264
|
+
passed: options.passed ?? report.blockingMissingRequirements.length === 0,
|
|
2265
|
+
readinessScore: report.readinessScore,
|
|
2266
|
+
minimumScore: options.minimumScore,
|
|
2267
|
+
blockingRequirementIds: report.blockingMissingRequirements.map((r) => r.id),
|
|
2268
|
+
nonBlockingRequirementIds: report.nonBlockingGaps.map((r) => r.id),
|
|
2269
|
+
recommendedAction: report.recommendedAction,
|
|
2270
|
+
severity: report.severity,
|
|
2271
|
+
reason: report.reason
|
|
2272
|
+
};
|
|
2273
|
+
}
|
|
2274
|
+
function userQuestionsForKnowledgeGaps(gaps) {
|
|
2275
|
+
return gaps.filter((gap) => gap.acquisitionMode === "ask_user" || gap.fallbackPolicy === "ask").map((gap) => ({
|
|
2276
|
+
id: `question_${gap.id}`,
|
|
2277
|
+
question: `Please provide: ${gap.description}`,
|
|
2278
|
+
reason: `Required for ${gap.requiredFor.join(", ") || "the task"}.`,
|
|
2279
|
+
requirementId: gap.id,
|
|
2280
|
+
importance: gap.importance,
|
|
2281
|
+
answerType: gap.sensitivity === "secret" ? "credential" : "free_text",
|
|
2282
|
+
impactIfUnknown: impactFor(gap)
|
|
2283
|
+
}));
|
|
2284
|
+
}
|
|
2285
|
+
function acquisitionPlansForKnowledgeGaps(gaps) {
|
|
2286
|
+
const byMode = /* @__PURE__ */ new Map();
|
|
2287
|
+
for (const gap of gaps) {
|
|
2288
|
+
const mode = planMode(gap.acquisitionMode);
|
|
2289
|
+
if (!mode) continue;
|
|
2290
|
+
const bucket = byMode.get(mode) ?? [];
|
|
2291
|
+
bucket.push(gap);
|
|
2292
|
+
byMode.set(mode, bucket);
|
|
2293
|
+
}
|
|
2294
|
+
return [...byMode.entries()].map(([mode, requirements]) => ({
|
|
2295
|
+
id: `acquire_${mode}`,
|
|
2296
|
+
requirementIds: requirements.map((r) => r.id),
|
|
2297
|
+
mode,
|
|
2298
|
+
description: descriptionForPlan(mode, requirements),
|
|
2299
|
+
priority: maxImportance(requirements.map((r) => r.importance)),
|
|
2300
|
+
questions: mode === "ask_user" ? userQuestionsForKnowledgeGaps(requirements) : void 0
|
|
2301
|
+
}));
|
|
2302
|
+
}
|
|
2303
|
+
function normalizeRequirement(requirement) {
|
|
2304
|
+
return {
|
|
2305
|
+
...requirement,
|
|
2306
|
+
confidenceNeeded: clamp012(requirement.confidenceNeeded),
|
|
2307
|
+
currentConfidence: clamp012(requirement.currentConfidence),
|
|
2308
|
+
evidenceIds: unique(requirement.evidenceIds)
|
|
2309
|
+
};
|
|
2310
|
+
}
|
|
2311
|
+
function weightedReadinessAt(requirements, now) {
|
|
2312
|
+
if (requirements.length === 0) return 1;
|
|
2313
|
+
let weightSum = 0;
|
|
2314
|
+
let scoreSum = 0;
|
|
2315
|
+
for (const requirement of requirements) {
|
|
2316
|
+
const weight = importanceWeight(requirement.importance);
|
|
2317
|
+
const score = isExpired(requirement, now) ? 0 : requirement.confidenceNeeded <= 0 ? 1 : Math.min(1, requirement.currentConfidence / requirement.confidenceNeeded);
|
|
2318
|
+
weightSum += weight;
|
|
2319
|
+
scoreSum += weight * score;
|
|
2320
|
+
}
|
|
2321
|
+
return clamp012(scoreSum / weightSum);
|
|
2322
|
+
}
|
|
2323
|
+
function isRequirementMissing(requirement, now) {
|
|
2324
|
+
return isExpired(requirement, now) || requirement.currentConfidence < requirement.confidenceNeeded;
|
|
2325
|
+
}
|
|
2326
|
+
function isExpired(requirement, now) {
|
|
2327
|
+
if (!requirement.validUntil) return false;
|
|
2328
|
+
const deadline = Date.parse(requirement.validUntil);
|
|
2329
|
+
if (!Number.isFinite(deadline)) return true;
|
|
2330
|
+
return deadline <= now.getTime();
|
|
2331
|
+
}
|
|
2332
|
+
function isBlockingGap(requirement) {
|
|
2333
|
+
return requirement.importance === "blocking" || requirement.fallbackPolicy === "block" || requirement.sensitivity === "secret";
|
|
2334
|
+
}
|
|
2335
|
+
function chooseRecommendedAction(blocking, nonBlocking) {
|
|
2336
|
+
const gaps = blocking.length > 0 ? blocking : nonBlocking;
|
|
2337
|
+
if (gaps.length === 0) return "run_agent";
|
|
2338
|
+
if (gaps.some((gap) => gap.acquisitionMode === "ask_user" || gap.fallbackPolicy === "ask"))
|
|
2339
|
+
return "ask_user";
|
|
2340
|
+
if (gaps.some((gap) => gap.acquisitionMode === "query_connector")) return "query_connectors";
|
|
2341
|
+
if (gaps.some(
|
|
2342
|
+
(gap) => gap.acquisitionMode === "inspect_repo" || gap.acquisitionMode === "run_command"
|
|
2343
|
+
))
|
|
2344
|
+
return "inspect_repo";
|
|
2345
|
+
if (gaps.some((gap) => gap.acquisitionMode === "search_web")) return "collect_web_data";
|
|
2346
|
+
if (gaps.some((gap) => gap.acquisitionMode === "not_available")) return "abort_or_rescope";
|
|
2347
|
+
if (nonBlocking.some((gap) => gap.importance === "high")) return "build_domain_wiki";
|
|
2348
|
+
return "continue_with_caveat";
|
|
2349
|
+
}
|
|
2350
|
+
function planMode(mode) {
|
|
2351
|
+
if (mode === "infer_low_confidence" || mode === "not_available") return null;
|
|
2352
|
+
return mode;
|
|
2353
|
+
}
|
|
2354
|
+
function descriptionForPlan(mode, requirements) {
|
|
2355
|
+
const labels = requirements.map((r) => r.description).join("; ");
|
|
2356
|
+
if (mode === "ask_user") return `Ask the user for: ${labels}`;
|
|
2357
|
+
if (mode === "search_web") return `Search web or documentation sources for: ${labels}`;
|
|
2358
|
+
if (mode === "query_connector") return `Query configured connectors for: ${labels}`;
|
|
2359
|
+
if (mode === "inspect_repo") return `Inspect repository context for: ${labels}`;
|
|
2360
|
+
if (mode === "run_command") return `Run local commands to collect: ${labels}`;
|
|
2361
|
+
return `Build domain wiki evidence for: ${labels}`;
|
|
2362
|
+
}
|
|
2363
|
+
function impactFor(requirement) {
|
|
2364
|
+
if (requirement.fallbackPolicy === "block") return "The agent should not run until this is known.";
|
|
2365
|
+
if (requirement.fallbackPolicy === "continue_with_caveat")
|
|
2366
|
+
return "The agent may continue, but must disclose uncertainty.";
|
|
2367
|
+
if (requirement.fallbackPolicy === "use_default")
|
|
2368
|
+
return "The agent will use the configured default if skipped.";
|
|
2369
|
+
return "The agent should ask before continuing.";
|
|
2370
|
+
}
|
|
2371
|
+
function maxImportance(values) {
|
|
2372
|
+
const order = ["blocking", "high", "medium", "low"];
|
|
2373
|
+
return order.find((value) => values.includes(value)) ?? "low";
|
|
2374
|
+
}
|
|
2375
|
+
function importanceWeight(importance) {
|
|
2376
|
+
if (importance === "blocking") return 8;
|
|
2377
|
+
if (importance === "high") return 4;
|
|
2378
|
+
if (importance === "medium") return 2;
|
|
2379
|
+
return 1;
|
|
2380
|
+
}
|
|
2381
|
+
function clamp012(value) {
|
|
2382
|
+
if (!Number.isFinite(value)) return 0;
|
|
2383
|
+
return Math.max(0, Math.min(1, value));
|
|
2384
|
+
}
|
|
2385
|
+
function unique(items) {
|
|
2386
|
+
return [...new Set(items)];
|
|
2387
|
+
}
|
|
2388
|
+
|
|
2263
2389
|
// src/live-proof.ts
|
|
2264
2390
|
async function runLiveProof(config) {
|
|
2265
2391
|
const startedAt = Date.now();
|
|
@@ -3520,7 +3646,7 @@ async function mapLimit(items, limit, fn) {
|
|
|
3520
3646
|
return results;
|
|
3521
3647
|
}
|
|
3522
3648
|
function mean2(values) {
|
|
3523
|
-
return values.length ? values.reduce((
|
|
3649
|
+
return values.length ? values.reduce((sum4, value) => sum4 + value, 0) / values.length : 0;
|
|
3524
3650
|
}
|
|
3525
3651
|
function meanRunScore(scores2) {
|
|
3526
3652
|
return {
|
|
@@ -3612,7 +3738,7 @@ function aggregateJudgeVerdicts(verdicts, dimensionKeys, weights) {
|
|
|
3612
3738
|
var DEFAULT_MAX_ATTEMPTS = 3;
|
|
3613
3739
|
var DEFAULT_TIMEOUT_MS = 3e5;
|
|
3614
3740
|
function sleep(ms) {
|
|
3615
|
-
return new Promise((
|
|
3741
|
+
return new Promise((resolve2) => setTimeout(resolve2, ms));
|
|
3616
3742
|
}
|
|
3617
3743
|
async function withJudgeRetry(judgeFn, policy = {}) {
|
|
3618
3744
|
const maxAttempts = policy.maxAttempts ?? DEFAULT_MAX_ATTEMPTS;
|
|
@@ -4041,7 +4167,7 @@ function rankRows(rows, weights) {
|
|
|
4041
4167
|
}
|
|
4042
4168
|
return [...buckets.entries()].map(([variantId, values]) => ({
|
|
4043
4169
|
variantId,
|
|
4044
|
-
mean: values.reduce((
|
|
4170
|
+
mean: values.reduce((sum4, value) => sum4 + value, 0) / values.length,
|
|
4045
4171
|
runs: values.length
|
|
4046
4172
|
})).sort((a, b) => b.mean - a.mean);
|
|
4047
4173
|
}
|
|
@@ -4295,13 +4421,13 @@ var defaultBlendWeights = { heldout: 0.7, judge: 0.3 };
|
|
|
4295
4421
|
function normalizeWeights(weights) {
|
|
4296
4422
|
const h = Number.isFinite(weights.heldout) && weights.heldout >= 0 ? weights.heldout : 0;
|
|
4297
4423
|
const j = Number.isFinite(weights.judge) && weights.judge >= 0 ? weights.judge : 0;
|
|
4298
|
-
const
|
|
4299
|
-
if (
|
|
4424
|
+
const sum4 = h + j;
|
|
4425
|
+
if (sum4 <= 0) {
|
|
4300
4426
|
throw new ValidationError(
|
|
4301
4427
|
"blend weights must have a positive sum (got heldout+judge <= 0) \u2014 cannot weight a composite by zero"
|
|
4302
4428
|
);
|
|
4303
4429
|
}
|
|
4304
|
-
return { heldout: h /
|
|
4430
|
+
return { heldout: h / sum4, judge: j / sum4 };
|
|
4305
4431
|
}
|
|
4306
4432
|
function blendHeldout(heldoutPassRate, judgeScore, weights = defaultBlendWeights) {
|
|
4307
4433
|
const w = normalizeWeights(weights);
|
|
@@ -6705,7 +6831,7 @@ async function commitBisect(options) {
|
|
|
6705
6831
|
}
|
|
6706
6832
|
async function promptBisect(options) {
|
|
6707
6833
|
const split = options.paragraphSplitter ?? ((p) => p.split(/\n\s*\n/));
|
|
6708
|
-
const
|
|
6834
|
+
const join6 = (paragraphs) => paragraphs.join("\n\n");
|
|
6709
6835
|
const goodParas = split(options.good);
|
|
6710
6836
|
const badParas = split(options.bad);
|
|
6711
6837
|
if (goodParas.length !== badParas.length) {
|
|
@@ -6725,7 +6851,7 @@ async function promptBisect(options) {
|
|
|
6725
6851
|
const result = await bisect({
|
|
6726
6852
|
good: goodMask,
|
|
6727
6853
|
bad: badMask,
|
|
6728
|
-
runEval: (mask) => options.runEval(
|
|
6854
|
+
runEval: (mask) => options.runEval(join6(paragraphsFor(mask))),
|
|
6729
6855
|
maxIterations: options.maxIterations ?? n + 5,
|
|
6730
6856
|
halfway: (g, b) => {
|
|
6731
6857
|
for (let i = 0; i < g.length; i++) {
|
|
@@ -6756,12 +6882,12 @@ async function promptBisect(options) {
|
|
|
6756
6882
|
}
|
|
6757
6883
|
}
|
|
6758
6884
|
const materializedPath = result.path.map((s) => ({
|
|
6759
|
-
state:
|
|
6885
|
+
state: join6(paragraphsFor(s.state)),
|
|
6760
6886
|
score: s.score,
|
|
6761
6887
|
pass: s.pass
|
|
6762
6888
|
}));
|
|
6763
6889
|
return {
|
|
6764
|
-
culprit:
|
|
6890
|
+
culprit: join6(paragraphsFor(culprit)),
|
|
6765
6891
|
path: materializedPath,
|
|
6766
6892
|
converged: result.converged,
|
|
6767
6893
|
inputInconsistent: result.inputInconsistent,
|
|
@@ -6769,6 +6895,90 @@ async function promptBisect(options) {
|
|
|
6769
6895
|
};
|
|
6770
6896
|
}
|
|
6771
6897
|
|
|
6898
|
+
// src/counterfactual.ts
|
|
6899
|
+
async function runCounterfactual(store, originalRunId, mutation, runner) {
|
|
6900
|
+
const originalRun = await store.getRun(originalRunId);
|
|
6901
|
+
if (!originalRun) throw new NotFoundError(`counterfactual: run ${originalRunId} not found`);
|
|
6902
|
+
const trajectory = await buildTrajectory(store, originalRunId);
|
|
6903
|
+
if (mutation.at < 0 || mutation.at >= trajectory.steps.length) {
|
|
6904
|
+
throw new ValidationError(
|
|
6905
|
+
`counterfactual: mutation.at=${mutation.at} out of range [0, ${trajectory.steps.length})`
|
|
6906
|
+
);
|
|
6907
|
+
}
|
|
6908
|
+
const targetStep = trajectory.steps[mutation.at];
|
|
6909
|
+
const mutatedStep = applyMutation(targetStep, mutation);
|
|
6910
|
+
const cfEmitter = new TraceEmitter(store);
|
|
6911
|
+
await cfEmitter.startRun({
|
|
6912
|
+
scenarioId: originalRun.scenarioId,
|
|
6913
|
+
variantId: originalRun.variantId ? `${originalRun.variantId}+cf:${mutation.kind}@${mutation.at}` : `cf:${mutation.kind}@${mutation.at}`,
|
|
6914
|
+
projectId: originalRun.projectId,
|
|
6915
|
+
parentRunId: originalRunId,
|
|
6916
|
+
layer: "meta",
|
|
6917
|
+
tags: { counterfactual: "true", mutationKind: mutation.kind, mutationAt: String(mutation.at) }
|
|
6918
|
+
});
|
|
6919
|
+
await runner.executeFrom(
|
|
6920
|
+
{
|
|
6921
|
+
originalRunId,
|
|
6922
|
+
originalTrajectory: trajectory,
|
|
6923
|
+
prefix: trajectory.steps.slice(0, mutation.at),
|
|
6924
|
+
mutation,
|
|
6925
|
+
mutatedStep
|
|
6926
|
+
},
|
|
6927
|
+
cfEmitter
|
|
6928
|
+
);
|
|
6929
|
+
const counterfactual = await store.getRun(cfEmitter.runId);
|
|
6930
|
+
const delta = {
|
|
6931
|
+
originalOutcomeScore: originalRun.outcome?.score ?? null,
|
|
6932
|
+
counterfactualOutcomeScore: counterfactual?.outcome?.score ?? null,
|
|
6933
|
+
deltaScore: originalRun.outcome?.score !== void 0 && counterfactual?.outcome?.score !== void 0 ? counterfactual.outcome.score - originalRun.outcome.score : null
|
|
6934
|
+
};
|
|
6935
|
+
return { counterfactualRunId: cfEmitter.runId, originalRunId, mutation, delta };
|
|
6936
|
+
}
|
|
6937
|
+
function applyMutation(step, mutation) {
|
|
6938
|
+
if (mutation.kind === "swap-model" && step.span.kind === "llm") {
|
|
6939
|
+
const llm = step.span;
|
|
6940
|
+
return { ...step, span: { ...llm, model: mutation.newModel } };
|
|
6941
|
+
}
|
|
6942
|
+
if (mutation.kind === "swap-tool-result" && step.span.kind === "tool") {
|
|
6943
|
+
const tool = step.span;
|
|
6944
|
+
return { ...step, span: { ...tool, result: mutation.newResult } };
|
|
6945
|
+
}
|
|
6946
|
+
if (mutation.kind === "inject-system-message" && step.span.kind === "llm") {
|
|
6947
|
+
const llm = step.span;
|
|
6948
|
+
return {
|
|
6949
|
+
...step,
|
|
6950
|
+
span: {
|
|
6951
|
+
...llm,
|
|
6952
|
+
messages: [{ role: "system", content: mutation.content }, ...llm.messages]
|
|
6953
|
+
}
|
|
6954
|
+
};
|
|
6955
|
+
}
|
|
6956
|
+
if (mutation.kind === "custom") return mutation.apply(step);
|
|
6957
|
+
return step;
|
|
6958
|
+
}
|
|
6959
|
+
function attributeCounterfactuals(results) {
|
|
6960
|
+
const grouped = /* @__PURE__ */ new Map();
|
|
6961
|
+
for (const r of results) {
|
|
6962
|
+
const arr = grouped.get(r.mutation.kind) ?? [];
|
|
6963
|
+
arr.push(r);
|
|
6964
|
+
grouped.set(r.mutation.kind, arr);
|
|
6965
|
+
}
|
|
6966
|
+
const out = [];
|
|
6967
|
+
for (const [kind, items] of grouped) {
|
|
6968
|
+
const deltas = items.map((i) => i.delta.deltaScore).filter((d) => typeof d === "number");
|
|
6969
|
+
if (deltas.length === 0) continue;
|
|
6970
|
+
const meanAbs = deltas.reduce((a, b) => a + Math.abs(b), 0) / deltas.length;
|
|
6971
|
+
const meanSigned = deltas.reduce((a, b) => a + b, 0) / deltas.length;
|
|
6972
|
+
out.push({
|
|
6973
|
+
mutationKind: kind,
|
|
6974
|
+
n: deltas.length,
|
|
6975
|
+
meanAbsDelta: meanAbs,
|
|
6976
|
+
meanSignedDelta: meanSigned
|
|
6977
|
+
});
|
|
6978
|
+
}
|
|
6979
|
+
return out.sort((a, b) => b.meanAbsDelta - a.meanAbsDelta);
|
|
6980
|
+
}
|
|
6981
|
+
|
|
6772
6982
|
// src/cross-trace-diff.ts
|
|
6773
6983
|
async function crossTraceDiff(store, runA, runB, options = {}) {
|
|
6774
6984
|
const [a, b] = await Promise.all([buildTrajectory(store, runA), buildTrajectory(store, runB)]);
|
|
@@ -7051,6 +7261,67 @@ function groupBy2(items, key) {
|
|
|
7051
7261
|
return m;
|
|
7052
7262
|
}
|
|
7053
7263
|
|
|
7264
|
+
// src/prm/training-export.ts
|
|
7265
|
+
async function exportTrainingData(store, graded, options = {}) {
|
|
7266
|
+
const window = options.contextWindow ?? 5;
|
|
7267
|
+
const out = [];
|
|
7268
|
+
for (const g of graded) {
|
|
7269
|
+
const trajectory = await buildTrajectory(store, g.runId);
|
|
7270
|
+
const spanById = new Map(trajectory.steps.map((s) => [s.span.spanId, s]));
|
|
7271
|
+
for (const gs of g.steps) {
|
|
7272
|
+
const node = spanById.get(gs.spanId);
|
|
7273
|
+
if (!node) continue;
|
|
7274
|
+
const idx = trajectory.steps.indexOf(node);
|
|
7275
|
+
const priorSpans = trajectory.steps.slice(Math.max(0, idx - window), idx).map((s) => s.span);
|
|
7276
|
+
out.push({
|
|
7277
|
+
runId: g.runId,
|
|
7278
|
+
spanId: gs.spanId,
|
|
7279
|
+
rubricId: gs.rubricId,
|
|
7280
|
+
score: gs.score,
|
|
7281
|
+
context: {
|
|
7282
|
+
priorTurns: priorSpans.map(spanToTurn).filter((t) => t !== null),
|
|
7283
|
+
step: { kind: node.span.kind, text: spanToText(node.span) }
|
|
7284
|
+
},
|
|
7285
|
+
rationale: gs.rationale,
|
|
7286
|
+
evidence: gs.evidence
|
|
7287
|
+
});
|
|
7288
|
+
}
|
|
7289
|
+
}
|
|
7290
|
+
return out;
|
|
7291
|
+
}
|
|
7292
|
+
function toNdjson(samples) {
|
|
7293
|
+
return `${samples.map((s) => JSON.stringify(s)).join("\n")}
|
|
7294
|
+
`;
|
|
7295
|
+
}
|
|
7296
|
+
function spanToTurn(span) {
|
|
7297
|
+
if (isLlmSpan(span)) {
|
|
7298
|
+
const text = span.output ?? span.messages.map((m) => `${m.role}: ${m.content}`).join("\n");
|
|
7299
|
+
return { role: "assistant", content: text };
|
|
7300
|
+
}
|
|
7301
|
+
if (isToolSpan(span)) {
|
|
7302
|
+
return {
|
|
7303
|
+
role: "tool",
|
|
7304
|
+
content: `${span.toolName}(${safeStringify(span.args)}) \u2192 ${safeStringify(span.result)}`
|
|
7305
|
+
};
|
|
7306
|
+
}
|
|
7307
|
+
return null;
|
|
7308
|
+
}
|
|
7309
|
+
function spanToText(span) {
|
|
7310
|
+
if (isLlmSpan(span)) return span.output ?? "";
|
|
7311
|
+
if (isToolSpan(span))
|
|
7312
|
+
return `${span.toolName}(${safeStringify(span.args)}) \u2192 ${safeStringify(span.result)}`;
|
|
7313
|
+
return span.name;
|
|
7314
|
+
}
|
|
7315
|
+
function safeStringify(v) {
|
|
7316
|
+
if (v === null || v === void 0) return "";
|
|
7317
|
+
if (typeof v === "string") return v;
|
|
7318
|
+
try {
|
|
7319
|
+
return JSON.stringify(v);
|
|
7320
|
+
} catch {
|
|
7321
|
+
return String(v);
|
|
7322
|
+
}
|
|
7323
|
+
}
|
|
7324
|
+
|
|
7054
7325
|
// src/reward-model-export.ts
|
|
7055
7326
|
async function exportRewardModel(store, grader, runIds) {
|
|
7056
7327
|
const graded = await Promise.all(runIds.map((id) => grader.grade(store, id)));
|
|
@@ -7481,7 +7752,7 @@ function extractErrorCount(text, opts = {}) {
|
|
|
7481
7752
|
for (const p of patterns) {
|
|
7482
7753
|
const matches2 = Array.from(text.matchAll(p.regex));
|
|
7483
7754
|
if (matches2.length === 0) continue;
|
|
7484
|
-
const count = p.transform ? matches2.reduce((
|
|
7755
|
+
const count = p.transform ? matches2.reduce((sum4, m) => sum4 + p.transform(m), 0) : matches2.length;
|
|
7485
7756
|
return {
|
|
7486
7757
|
count,
|
|
7487
7758
|
matched: p.name,
|
|
@@ -8419,7 +8690,7 @@ function defaultReferenceReplayMatcher(reference, candidate) {
|
|
|
8419
8690
|
const textScore = tokenJaccard(referenceText, candidateText);
|
|
8420
8691
|
const severityScore = reference.severity && candidate.severity ? normalize(reference.severity) === normalize(candidate.severity) ? 0.1 : -0.05 : 0;
|
|
8421
8692
|
const tagScore = tagOverlap(reference.tags, candidate.tags) * 0.15;
|
|
8422
|
-
const score =
|
|
8693
|
+
const score = clamp013(textScore * 0.85 + tagScore + severityScore);
|
|
8423
8694
|
return {
|
|
8424
8695
|
score,
|
|
8425
8696
|
reason: `token=${textScore.toFixed(2)} tags=${tagScore.toFixed(2)} severity=${severityScore.toFixed(2)}`
|
|
@@ -8525,13 +8796,13 @@ function scorePair(scenario, matcher, reference, candidate) {
|
|
|
8525
8796
|
`reference replay matcher returned non-finite score for ${scenario.id}:${reference.id}:${candidate.id}`
|
|
8526
8797
|
);
|
|
8527
8798
|
}
|
|
8528
|
-
return { score:
|
|
8799
|
+
return { score: clamp013(result.score), reason: result.reason ?? "" };
|
|
8529
8800
|
}
|
|
8530
8801
|
function buildScenarioScore(scenario, matches2, falsePositives) {
|
|
8531
8802
|
const matched = matches2.filter((match) => match.matched).length;
|
|
8532
8803
|
const total = scenario.references.length;
|
|
8533
|
-
const matchedWeight = matches2.filter((match) => match.matched).reduce((
|
|
8534
|
-
const totalWeight = matches2.reduce((
|
|
8804
|
+
const matchedWeight = matches2.filter((match) => match.matched).reduce((sum4, match) => sum4 + match.weight, 0);
|
|
8805
|
+
const totalWeight = matches2.reduce((sum4, match) => sum4 + match.weight, 0);
|
|
8535
8806
|
const precision2 = ratio(matched, matched + falsePositives);
|
|
8536
8807
|
const recall = ratio(matched, total);
|
|
8537
8808
|
return {
|
|
@@ -8624,7 +8895,7 @@ function tokens(text) {
|
|
|
8624
8895
|
function normalize(text) {
|
|
8625
8896
|
return text.toLowerCase().replace(/[^a-z0-9]+/g, " ").trim();
|
|
8626
8897
|
}
|
|
8627
|
-
function
|
|
8898
|
+
function clamp013(value) {
|
|
8628
8899
|
if (!Number.isFinite(value)) return 0;
|
|
8629
8900
|
return Math.max(0, Math.min(1, value));
|
|
8630
8901
|
}
|
|
@@ -9428,8 +9699,8 @@ function createSandboxPool(opts) {
|
|
|
9428
9699
|
});
|
|
9429
9700
|
return state;
|
|
9430
9701
|
}
|
|
9431
|
-
return new Promise((
|
|
9432
|
-
waiters.push({ resolve, reject });
|
|
9702
|
+
return new Promise((resolve2, reject) => {
|
|
9703
|
+
waiters.push({ resolve: resolve2, reject });
|
|
9433
9704
|
});
|
|
9434
9705
|
}
|
|
9435
9706
|
function handOffCleanSlot(state) {
|
|
@@ -9617,7 +9888,7 @@ function traceJudge(judge, judgeName, opts) {
|
|
|
9617
9888
|
});
|
|
9618
9889
|
try {
|
|
9619
9890
|
const scores2 = await judge(tc, input);
|
|
9620
|
-
const composite = scores2.length > 0 ? scores2.reduce((
|
|
9891
|
+
const composite = scores2.length > 0 ? scores2.reduce((sum4, s) => sum4 + s.score, 0) / scores2.length : 0;
|
|
9621
9892
|
await span.end({
|
|
9622
9893
|
attributes: {
|
|
9623
9894
|
"judge.name": judgeName,
|
|
@@ -9662,7 +9933,7 @@ function traceJudgeEnsemble(judges, judgeNames, opts) {
|
|
|
9662
9933
|
failedJudges++;
|
|
9663
9934
|
}
|
|
9664
9935
|
}
|
|
9665
|
-
const composite = allScores.length > 0 ? allScores.reduce((
|
|
9936
|
+
const composite = allScores.length > 0 ? allScores.reduce((sum4, s) => sum4 + s.score, 0) / allScores.length : 0;
|
|
9666
9937
|
await ensembleSpan.end({
|
|
9667
9938
|
attributes: {
|
|
9668
9939
|
"judge.ensemble_size": judges.length,
|
|
@@ -10119,7 +10390,7 @@ function costReport(ledger) {
|
|
|
10119
10390
|
perChannel: summary.byChannel,
|
|
10120
10391
|
total: {
|
|
10121
10392
|
usd: summary.totalCostUsd,
|
|
10122
|
-
unknownEntries: summary.byChannel.reduce((
|
|
10393
|
+
unknownEntries: summary.byChannel.reduce((sum4, c) => sum4 + c.unpricedCalls, 0)
|
|
10123
10394
|
},
|
|
10124
10395
|
perModel: [...perModel.values()].sort((a, b) => a.model.localeCompare(b.model))
|
|
10125
10396
|
};
|
|
@@ -10217,6 +10488,839 @@ function verifyAttestation(report, attested) {
|
|
|
10217
10488
|
}
|
|
10218
10489
|
return { valid: true };
|
|
10219
10490
|
}
|
|
10491
|
+
|
|
10492
|
+
// src/product-benchmark/index.ts
|
|
10493
|
+
import { existsSync as existsSync6, readFileSync as readFileSync7, statSync as statSync3 } from "fs";
|
|
10494
|
+
import { dirname as dirname4, join as join5 } from "path";
|
|
10495
|
+
|
|
10496
|
+
// src/product-benchmark/export.ts
|
|
10497
|
+
import { spawnSync as spawnSync2 } from "child_process";
|
|
10498
|
+
import { createHash } from "crypto";
|
|
10499
|
+
import { cpSync, existsSync as existsSync5, mkdirSync as mkdirSync3, readFileSync as readFileSync6, writeFileSync } from "fs";
|
|
10500
|
+
import { basename as basename2, dirname as dirname3, isAbsolute, join as join4, relative, resolve } from "path";
|
|
10501
|
+
var productBenchmarkMutableSurfaces = [
|
|
10502
|
+
"prompt",
|
|
10503
|
+
"resources.files",
|
|
10504
|
+
"tools",
|
|
10505
|
+
"mcp",
|
|
10506
|
+
"hooks",
|
|
10507
|
+
"subagents"
|
|
10508
|
+
];
|
|
10509
|
+
function sha256(value) {
|
|
10510
|
+
return `sha256:${createHash("sha256").update(value).digest("hex")}`;
|
|
10511
|
+
}
|
|
10512
|
+
function safePathPart(value) {
|
|
10513
|
+
return value.replace(/[^a-zA-Z0-9_.-]+/g, "-").replace(/^-+|-+$/g, "").slice(0, 80) || "artifact";
|
|
10514
|
+
}
|
|
10515
|
+
function git(args, fallback) {
|
|
10516
|
+
const res = spawnSync2("git", args, { cwd: process.cwd(), encoding: "utf8" });
|
|
10517
|
+
const value = res.status === 0 ? res.stdout.trim() : "";
|
|
10518
|
+
return value.length > 0 ? value : fallback;
|
|
10519
|
+
}
|
|
10520
|
+
function knownEnv(name) {
|
|
10521
|
+
const value = process.env[name]?.trim();
|
|
10522
|
+
return value && value !== "unknown" ? value : void 0;
|
|
10523
|
+
}
|
|
10524
|
+
function ciBranchName() {
|
|
10525
|
+
const direct = knownEnv("GITHUB_HEAD_REF") ?? knownEnv("GITHUB_REF_NAME") ?? knownEnv("VERCEL_GIT_COMMIT_REF");
|
|
10526
|
+
if (direct) return direct;
|
|
10527
|
+
const ref = knownEnv("GITHUB_REF");
|
|
10528
|
+
return ref?.startsWith("refs/heads/") ? ref.slice("refs/heads/".length) : void 0;
|
|
10529
|
+
}
|
|
10530
|
+
function repoBranch() {
|
|
10531
|
+
const ci = ciBranchName();
|
|
10532
|
+
if (ci) return ci;
|
|
10533
|
+
const branch = git(["branch", "--show-current"], "");
|
|
10534
|
+
if (branch) return branch;
|
|
10535
|
+
const name = git(["name-rev", "--name-only", "--exclude=tags/*", "HEAD"], "");
|
|
10536
|
+
if (name && name !== "undefined") return name;
|
|
10537
|
+
return `detached:${git(["rev-parse", "--short", "HEAD"], "unknown")}`;
|
|
10538
|
+
}
|
|
10539
|
+
function productBenchmarkRepoIdentity() {
|
|
10540
|
+
return {
|
|
10541
|
+
url: git(["config", "--get", "remote.origin.url"], "unknown"),
|
|
10542
|
+
commit: git(["rev-parse", "HEAD"], "unknown"),
|
|
10543
|
+
branch: repoBranch()
|
|
10544
|
+
};
|
|
10545
|
+
}
|
|
10546
|
+
function packageVersion(name) {
|
|
10547
|
+
const pkg = JSON.parse(readFileSync6(resolve("package.json"), "utf8"));
|
|
10548
|
+
if (pkg.name === name && pkg.version) return pkg.version;
|
|
10549
|
+
const declared = pkg.dependencies?.[name] ?? pkg.devDependencies?.[name];
|
|
10550
|
+
if (declared) return declared;
|
|
10551
|
+
const installed = resolve("node_modules", name, "package.json");
|
|
10552
|
+
if (existsSync5(installed)) {
|
|
10553
|
+
const installedPkg = JSON.parse(readFileSync6(installed, "utf8"));
|
|
10554
|
+
if (installedPkg.version) return installedPkg.version;
|
|
10555
|
+
}
|
|
10556
|
+
return "unknown";
|
|
10557
|
+
}
|
|
10558
|
+
function readRunRecords(path) {
|
|
10559
|
+
const lines = readFileSync6(path, "utf8").split("\n").map((line) => line.trim()).filter(Boolean);
|
|
10560
|
+
return lines.map((line, index) => {
|
|
10561
|
+
let parsed;
|
|
10562
|
+
try {
|
|
10563
|
+
parsed = JSON.parse(line);
|
|
10564
|
+
} catch (err) {
|
|
10565
|
+
throw new ValidationError(
|
|
10566
|
+
`${path}:${index + 1}: ${err instanceof Error ? err.message : String(err)}`,
|
|
10567
|
+
{ cause: err }
|
|
10568
|
+
);
|
|
10569
|
+
}
|
|
10570
|
+
return expectRunRecordShape(parsed, `${path}:${index + 1}`);
|
|
10571
|
+
});
|
|
10572
|
+
}
|
|
10573
|
+
function expectRunRecordShape(value, path) {
|
|
10574
|
+
if (!value || typeof value !== "object" || Array.isArray(value)) {
|
|
10575
|
+
throw new ValidationError(`${path}: run record must be an object`);
|
|
10576
|
+
}
|
|
10577
|
+
const obj = value;
|
|
10578
|
+
for (const key of ["runId", "model"]) {
|
|
10579
|
+
if (typeof obj[key] !== "string" || obj[key].length === 0) {
|
|
10580
|
+
throw new ValidationError(`${path}: run record ${key} must be a non-empty string`);
|
|
10581
|
+
}
|
|
10582
|
+
}
|
|
10583
|
+
for (const key of ["costUsd", "wallMs"]) {
|
|
10584
|
+
if (typeof obj[key] !== "number" || !Number.isFinite(obj[key])) {
|
|
10585
|
+
throw new ValidationError(`${path}: run record ${key} must be a finite number`);
|
|
10586
|
+
}
|
|
10587
|
+
}
|
|
10588
|
+
const tokenUsage = obj.tokenUsage;
|
|
10589
|
+
if (!tokenUsage || typeof tokenUsage.input !== "number" || typeof tokenUsage.output !== "number") {
|
|
10590
|
+
throw new ValidationError(`${path}: run record tokenUsage.input/output must be numbers`);
|
|
10591
|
+
}
|
|
10592
|
+
const outcome = obj.outcome;
|
|
10593
|
+
if (!outcome || !outcome.raw || typeof outcome.raw !== "object") {
|
|
10594
|
+
throw new ValidationError(`${path}: run record outcome.raw must be an object`);
|
|
10595
|
+
}
|
|
10596
|
+
return value;
|
|
10597
|
+
}
|
|
10598
|
+
function clamp014(value) {
|
|
10599
|
+
return Math.max(0, Math.min(1, value));
|
|
10600
|
+
}
|
|
10601
|
+
function splitOf(record, opts) {
|
|
10602
|
+
const custom = opts.classifySplit?.(record);
|
|
10603
|
+
if (custom) return custom;
|
|
10604
|
+
if (record.outcome.raw.safety === 1) return "safety";
|
|
10605
|
+
if (record.splitTag === "holdout") return "holdout";
|
|
10606
|
+
if (record.splitTag === "dev") return "dev";
|
|
10607
|
+
return "practice";
|
|
10608
|
+
}
|
|
10609
|
+
function scoreOf(record) {
|
|
10610
|
+
const score = record.outcome.holdoutScore ?? record.outcome.searchScore;
|
|
10611
|
+
if (typeof score === "number" && Number.isFinite(score)) return clamp014(score);
|
|
10612
|
+
const rawScore = record.outcome.raw.score ?? record.outcome.raw.composite;
|
|
10613
|
+
return typeof rawScore === "number" && Number.isFinite(rawScore) ? clamp014(rawScore) : 0;
|
|
10614
|
+
}
|
|
10615
|
+
function rawPassOf(record) {
|
|
10616
|
+
const rawPass = record.outcome.raw.pass;
|
|
10617
|
+
if (typeof rawPass === "boolean") return rawPass;
|
|
10618
|
+
if (typeof rawPass === "number" && Number.isFinite(rawPass)) return rawPass >= 1;
|
|
10619
|
+
return null;
|
|
10620
|
+
}
|
|
10621
|
+
function passOf(record, score, threshold) {
|
|
10622
|
+
const rawPass = rawPassOf(record);
|
|
10623
|
+
if (rawPass !== null) return rawPass && !record.failureMode;
|
|
10624
|
+
return score >= threshold && !record.failureMode;
|
|
10625
|
+
}
|
|
10626
|
+
function failureModeOf(record, score, threshold) {
|
|
10627
|
+
if (record.failureMode) return record.failureMode;
|
|
10628
|
+
const belowThreshold = `quality-below-threshold: ${Math.round(score * 100)}% < ${Math.round(threshold * 100)}%`;
|
|
10629
|
+
if (rawPassOf(record) === false) {
|
|
10630
|
+
return score < threshold ? belowThreshold : "product-pass-failed";
|
|
10631
|
+
}
|
|
10632
|
+
if (passOf(record, score, threshold)) return null;
|
|
10633
|
+
return belowThreshold;
|
|
10634
|
+
}
|
|
10635
|
+
function numericDimensions(record) {
|
|
10636
|
+
const dimensions = {};
|
|
10637
|
+
for (const [key, value] of Object.entries(record.outcome.raw ?? {})) {
|
|
10638
|
+
if (typeof value === "number" && Number.isFinite(value)) dimensions[key] = value;
|
|
10639
|
+
}
|
|
10640
|
+
if (Object.keys(dimensions).length === 0) dimensions.score = scoreOf(record);
|
|
10641
|
+
return dimensions;
|
|
10642
|
+
}
|
|
10643
|
+
function armIdOf(record) {
|
|
10644
|
+
if (record.candidateId) return record.candidateId;
|
|
10645
|
+
const variant = record.agentProfile?.dimensions?.variant;
|
|
10646
|
+
if (typeof variant === "string" && variant.length > 0) return variant;
|
|
10647
|
+
const variantId = record.agentProfile?.dimensions?.variantId;
|
|
10648
|
+
if (typeof variantId === "string" && variantId.length > 0) return variantId;
|
|
10649
|
+
return "production-profile";
|
|
10650
|
+
}
|
|
10651
|
+
function backendOf(record) {
|
|
10652
|
+
const backend = record.agentProfile?.dimensions?.backend;
|
|
10653
|
+
return typeof backend === "string" && backend.length > 0 ? backend : "unknown";
|
|
10654
|
+
}
|
|
10655
|
+
function modelProvider(record) {
|
|
10656
|
+
const backend = backendOf(record);
|
|
10657
|
+
if (backend === "cli-bridge") return "cli-bridge";
|
|
10658
|
+
if (backend === "sandbox") return "router";
|
|
10659
|
+
if (record.model.startsWith("router/")) return "router";
|
|
10660
|
+
return backend === "unknown" ? "router" : backend;
|
|
10661
|
+
}
|
|
10662
|
+
function runtimeResolution(record, opts) {
|
|
10663
|
+
const reasoningEffort = record.agentProfile?.dimensions?.reasoningLevel;
|
|
10664
|
+
return {
|
|
10665
|
+
model: record.agentProfile?.model ?? record.model,
|
|
10666
|
+
harness: record.agentProfile?.harness?.id ?? `${opts.projectId}-canonical-eval`,
|
|
10667
|
+
backend: backendOf(record),
|
|
10668
|
+
...typeof reasoningEffort === "string" && reasoningEffort.length > 0 ? { reasoningEffort } : {}
|
|
10669
|
+
};
|
|
10670
|
+
}
|
|
10671
|
+
function sourceProfileHash(record) {
|
|
10672
|
+
return record.agentProfile?.sourceProfile?.hash ?? record.agentProfile?.cellId ?? sha256(
|
|
10673
|
+
JSON.stringify({
|
|
10674
|
+
candidateId: record.candidateId,
|
|
10675
|
+
promptHash: record.promptHash,
|
|
10676
|
+
configHash: record.configHash
|
|
10677
|
+
})
|
|
10678
|
+
);
|
|
10679
|
+
}
|
|
10680
|
+
function materializedProfileHash(record, runtime) {
|
|
10681
|
+
return sha256(
|
|
10682
|
+
JSON.stringify({ sourceProfileHash: sourceProfileHash(record), model: runtime.model })
|
|
10683
|
+
);
|
|
10684
|
+
}
|
|
10685
|
+
function profileIdOf(record, armId, runtime, opts) {
|
|
10686
|
+
const base = record.agentProfile?.profileId ?? opts.fallbackProfileId ?? armId;
|
|
10687
|
+
const withArm = base.endsWith(`:${armId}`) ? base : `${base}:${armId}`;
|
|
10688
|
+
return `${withArm}:${runtime.model}`;
|
|
10689
|
+
}
|
|
10690
|
+
function toolCallsOf(record, runDir, opts) {
|
|
10691
|
+
const raw = record.outcome.raw;
|
|
10692
|
+
const primary = Number(raw.tool_call_count ?? raw.tool_calls ?? raw.toolCalls ?? 0);
|
|
10693
|
+
if (Number.isFinite(primary) && primary > 0) return primary;
|
|
10694
|
+
const fallback = opts.toolCallFallback?.(record, runDir);
|
|
10695
|
+
if (fallback !== void 0 && Number.isFinite(fallback) && fallback > 0) return fallback;
|
|
10696
|
+
return Number.isFinite(primary) ? primary : 0;
|
|
10697
|
+
}
|
|
10698
|
+
var TRACE_CANDIDATES = ["traces.jsonl", "traces", "trace", "trace-store"];
|
|
10699
|
+
var RAW_CANDIDATES = ["raws.jsonl", "raw-events"];
|
|
10700
|
+
var SCORE_CANDIDATES = ["scores.json", "manifest.json"];
|
|
10701
|
+
function firstExisting(runDir, candidates) {
|
|
10702
|
+
return candidates.find((candidate) => existsSync5(join4(runDir, candidate))) ?? candidates[0];
|
|
10703
|
+
}
|
|
10704
|
+
function relativeArtifacts(runDir) {
|
|
10705
|
+
return {
|
|
10706
|
+
records: "records.jsonl",
|
|
10707
|
+
traces: firstExisting(runDir, TRACE_CANDIDATES),
|
|
10708
|
+
raws: firstExisting(runDir, RAW_CANDIDATES),
|
|
10709
|
+
scores: firstExisting(runDir, SCORE_CANDIDATES),
|
|
10710
|
+
workspace: existsSync5(join4(runDir, "workspace")) ? "workspace" : "."
|
|
10711
|
+
};
|
|
10712
|
+
}
|
|
10713
|
+
function absoluteArtifacts(runDir) {
|
|
10714
|
+
const rel = relativeArtifacts(runDir);
|
|
10715
|
+
return {
|
|
10716
|
+
records: resolve(runDir, rel.records),
|
|
10717
|
+
traces: resolve(runDir, rel.traces),
|
|
10718
|
+
raws: resolve(runDir, rel.raws),
|
|
10719
|
+
scores: resolve(runDir, rel.scores),
|
|
10720
|
+
workspace: resolve(runDir, rel.workspace)
|
|
10721
|
+
};
|
|
10722
|
+
}
|
|
10723
|
+
function materializeRunArtifacts(runDir, outDir, index) {
|
|
10724
|
+
if (resolve(runDir) === resolve(outDir)) return relativeArtifacts(runDir);
|
|
10725
|
+
const rel = relative(runDir, outDir);
|
|
10726
|
+
if (rel && !rel.startsWith("..") && !isAbsolute(rel)) {
|
|
10727
|
+
throw new ValidationError(`outDir must not be nested inside runDir: ${outDir}`);
|
|
10728
|
+
}
|
|
10729
|
+
const id = `${String(index + 1).padStart(2, "0")}-${safePathPart(basename2(runDir))}-${sha256(runDir).slice(7, 19)}`;
|
|
10730
|
+
const destRel = join4("source-runs", id);
|
|
10731
|
+
const destAbs = join4(outDir, destRel);
|
|
10732
|
+
mkdirSync3(dirname3(destAbs), { recursive: true });
|
|
10733
|
+
cpSync(runDir, destAbs, { recursive: true, force: true });
|
|
10734
|
+
const source = relativeArtifacts(runDir);
|
|
10735
|
+
return {
|
|
10736
|
+
records: join4(destRel, source.records),
|
|
10737
|
+
traces: join4(destRel, source.traces),
|
|
10738
|
+
raws: join4(destRel, source.raws),
|
|
10739
|
+
scores: join4(destRel, source.scores),
|
|
10740
|
+
workspace: source.workspace === "." ? destRel : join4(destRel, source.workspace)
|
|
10741
|
+
};
|
|
10742
|
+
}
|
|
10743
|
+
function existsArtifact(artifactRoot, value) {
|
|
10744
|
+
return existsSync5(resolve(artifactRoot, value));
|
|
10745
|
+
}
|
|
10746
|
+
function runRecordToProductBenchmarkRecord(record, runDir, artifactRoot, artifacts, options) {
|
|
10747
|
+
const opts = resolveOptions(options);
|
|
10748
|
+
const runtime = runtimeResolution(record, opts);
|
|
10749
|
+
const score = scoreOf(record);
|
|
10750
|
+
const armId = armIdOf(record);
|
|
10751
|
+
const inputTokens = record.tokenUsage.input;
|
|
10752
|
+
const outputTokens = record.tokenUsage.output;
|
|
10753
|
+
const toolCallCount = toolCallsOf(record, runDir, opts);
|
|
10754
|
+
const dimensions = numericDimensions(record);
|
|
10755
|
+
if (!("tool_calls" in dimensions)) dimensions.tool_calls = toolCallCount;
|
|
10756
|
+
const product = {
|
|
10757
|
+
schemaVersion: 1,
|
|
10758
|
+
projectId: opts.projectId,
|
|
10759
|
+
benchmarkId: opts.benchmarkId,
|
|
10760
|
+
runId: record.runId,
|
|
10761
|
+
scenarioId: record.scenarioId ?? record.experimentId,
|
|
10762
|
+
split: splitOf(record, opts),
|
|
10763
|
+
armId,
|
|
10764
|
+
rep: Number(record.seed ?? 0) + 1,
|
|
10765
|
+
agentProfile: {
|
|
10766
|
+
id: profileIdOf(record, armId, runtime, opts),
|
|
10767
|
+
hash: materializedProfileHash(record, runtime),
|
|
10768
|
+
path: opts.agentProfilePath,
|
|
10769
|
+
declared: runtime,
|
|
10770
|
+
resolved: runtime
|
|
10771
|
+
},
|
|
10772
|
+
model: { provider: modelProvider(record), id: record.model },
|
|
10773
|
+
backend: { kind: backendOf(record), version: opts.backendVersion },
|
|
10774
|
+
outcome: {
|
|
10775
|
+
pass: passOf(record, score, opts.passThreshold),
|
|
10776
|
+
score,
|
|
10777
|
+
dimensions,
|
|
10778
|
+
failureMode: failureModeOf(record, score, opts.passThreshold)
|
|
10779
|
+
},
|
|
10780
|
+
usage: {
|
|
10781
|
+
inputTokens,
|
|
10782
|
+
outputTokens,
|
|
10783
|
+
costUsd: record.costUsd,
|
|
10784
|
+
// Rounded: the bundle contract requires integer milliseconds.
|
|
10785
|
+
wallMs: Math.round(record.wallMs),
|
|
10786
|
+
toolCalls: toolCallCount
|
|
10787
|
+
},
|
|
10788
|
+
integrity: {
|
|
10789
|
+
realBackend: inputTokens + outputTokens > 0,
|
|
10790
|
+
rawCapture: existsArtifact(artifactRoot, artifacts.raws),
|
|
10791
|
+
traceCapture: existsArtifact(artifactRoot, artifacts.traces),
|
|
10792
|
+
noStubRows: inputTokens + outputTokens > 0,
|
|
10793
|
+
priced: record.costUsd > 0,
|
|
10794
|
+
profileMaterialized: Boolean(record.agentProfile?.cellId)
|
|
10795
|
+
},
|
|
10796
|
+
artifacts
|
|
10797
|
+
};
|
|
10798
|
+
return validateProductBenchmarkRecord(product);
|
|
10799
|
+
}
|
|
10800
|
+
function uniqueArmRecords(records) {
|
|
10801
|
+
const byArm = /* @__PURE__ */ new Map();
|
|
10802
|
+
for (const record of records) {
|
|
10803
|
+
const prior = byArm.get(record.armId);
|
|
10804
|
+
if (!prior) {
|
|
10805
|
+
byArm.set(record.armId, record);
|
|
10806
|
+
continue;
|
|
10807
|
+
}
|
|
10808
|
+
const fields = [
|
|
10809
|
+
["profileId", prior.agentProfile.id, record.agentProfile.id],
|
|
10810
|
+
["model", prior.agentProfile.resolved.model, record.agentProfile.resolved.model],
|
|
10811
|
+
["backend", prior.agentProfile.resolved.backend, record.agentProfile.resolved.backend],
|
|
10812
|
+
["harness", prior.agentProfile.resolved.harness, record.agentProfile.resolved.harness],
|
|
10813
|
+
[
|
|
10814
|
+
"reasoningEffort",
|
|
10815
|
+
prior.agentProfile.resolved.reasoningEffort,
|
|
10816
|
+
record.agentProfile.resolved.reasoningEffort
|
|
10817
|
+
]
|
|
10818
|
+
];
|
|
10819
|
+
for (const [name, a, b] of fields) {
|
|
10820
|
+
if (a !== b) {
|
|
10821
|
+
throw new ValidationError(
|
|
10822
|
+
`records for arm '${record.armId}' disagree on ${name} (${String(a)} vs ${String(b)}) \u2014 one arm id must map to one policy`
|
|
10823
|
+
);
|
|
10824
|
+
}
|
|
10825
|
+
}
|
|
10826
|
+
}
|
|
10827
|
+
return [...byArm.values()];
|
|
10828
|
+
}
|
|
10829
|
+
function uniqueScenarioRecords(records) {
|
|
10830
|
+
const byScenario = /* @__PURE__ */ new Map();
|
|
10831
|
+
for (const record of records) {
|
|
10832
|
+
const prior = byScenario.get(record.scenarioId);
|
|
10833
|
+
if (!prior) {
|
|
10834
|
+
byScenario.set(record.scenarioId, record);
|
|
10835
|
+
continue;
|
|
10836
|
+
}
|
|
10837
|
+
if (prior.split !== record.split) {
|
|
10838
|
+
throw new ValidationError(
|
|
10839
|
+
`records for scenario '${record.scenarioId}' disagree on split (${prior.split} vs ${record.split})`
|
|
10840
|
+
);
|
|
10841
|
+
}
|
|
10842
|
+
}
|
|
10843
|
+
return [...byScenario.values()];
|
|
10844
|
+
}
|
|
10845
|
+
function buildProductBenchmarkManifest(records, options) {
|
|
10846
|
+
if (records.length === 0) {
|
|
10847
|
+
throw new ValidationError("cannot build a product benchmark manifest from zero records");
|
|
10848
|
+
}
|
|
10849
|
+
const scenarioTagPrefix = options.scenarioTagPrefix ?? defaultScenarioTagPrefix(options.projectId);
|
|
10850
|
+
const mutableSurfaces = options.mutableSurfaces ?? productBenchmarkMutableSurfaces;
|
|
10851
|
+
const byProfile = /* @__PURE__ */ new Map();
|
|
10852
|
+
for (const record of records) {
|
|
10853
|
+
byProfile.set(record.agentProfile.id, {
|
|
10854
|
+
id: record.agentProfile.id,
|
|
10855
|
+
profileHash: record.agentProfile.hash,
|
|
10856
|
+
agentProfilePath: record.agentProfile.path
|
|
10857
|
+
});
|
|
10858
|
+
}
|
|
10859
|
+
const manifest = {
|
|
10860
|
+
schemaVersion: 1,
|
|
10861
|
+
projectId: options.projectId,
|
|
10862
|
+
benchmarkId: options.benchmarkId,
|
|
10863
|
+
repo: productBenchmarkRepoIdentity(),
|
|
10864
|
+
substrate: {
|
|
10865
|
+
agentEval: packageVersion("@tangle-network/agent-eval"),
|
|
10866
|
+
agentRuntime: packageVersion("@tangle-network/agent-runtime"),
|
|
10867
|
+
agentInterface: packageVersion("@tangle-network/agent-interface"),
|
|
10868
|
+
sandbox: packageVersion("@tangle-network/sandbox"),
|
|
10869
|
+
...options.substrate
|
|
10870
|
+
},
|
|
10871
|
+
profiles: [...byProfile.values()],
|
|
10872
|
+
arms: uniqueArmRecords(records).map((record) => ({
|
|
10873
|
+
id: record.armId,
|
|
10874
|
+
profileId: record.agentProfile.id,
|
|
10875
|
+
mutableSurfaces: [...mutableSurfaces],
|
|
10876
|
+
policyAxes: {
|
|
10877
|
+
carrier: record.armId.includes("policy") ? "resource-file" : "profile",
|
|
10878
|
+
model: record.agentProfile.resolved.model,
|
|
10879
|
+
backend: record.agentProfile.resolved.backend,
|
|
10880
|
+
harness: record.agentProfile.resolved.harness,
|
|
10881
|
+
...record.agentProfile.resolved.reasoningEffort !== void 0 ? { reasoningEffort: record.agentProfile.resolved.reasoningEffort } : {}
|
|
10882
|
+
}
|
|
10883
|
+
})),
|
|
10884
|
+
scenarios: uniqueScenarioRecords(records).map((record) => ({
|
|
10885
|
+
id: record.scenarioId,
|
|
10886
|
+
split: record.split,
|
|
10887
|
+
tags: [scenarioTagPrefix, options.benchmarkId, record.split],
|
|
10888
|
+
sourceAllowedForSynthesis: false
|
|
10889
|
+
})),
|
|
10890
|
+
budgets: {
|
|
10891
|
+
maxUsd: records.reduce((sum4, record) => sum4 + record.usage.costUsd, 0),
|
|
10892
|
+
maxCells: records.length,
|
|
10893
|
+
maxWallMs: records.reduce((sum4, record) => sum4 + record.usage.wallMs, 0)
|
|
10894
|
+
},
|
|
10895
|
+
expectedArtifactDir: resolve(options.outDir)
|
|
10896
|
+
};
|
|
10897
|
+
return validateProductBenchmarkManifest(manifest);
|
|
10898
|
+
}
|
|
10899
|
+
function exportProductBenchmark(options) {
|
|
10900
|
+
const { runDir, ...rest } = options;
|
|
10901
|
+
return exportProductBenchmarkRuns({ ...rest, runDirs: [runDir] });
|
|
10902
|
+
}
|
|
10903
|
+
function exportProductBenchmarkRuns(options) {
|
|
10904
|
+
const opts = resolveOptions(options);
|
|
10905
|
+
const outDir = resolve(options.outDir);
|
|
10906
|
+
const runDirs = options.runDirs.map((runDir) => resolve(runDir));
|
|
10907
|
+
if (runDirs.length === 0) throw new ValidationError("export requires at least one run directory");
|
|
10908
|
+
const materialize = options.materializeSourceRuns !== false;
|
|
10909
|
+
const rows = runDirs.flatMap((runDir) => {
|
|
10910
|
+
const sourceRecordsPath = join4(runDir, "records.jsonl");
|
|
10911
|
+
if (!existsSync5(sourceRecordsPath)) throw new ValidationError(`missing ${sourceRecordsPath}`);
|
|
10912
|
+
const records = readRunRecords(sourceRecordsPath);
|
|
10913
|
+
if (records.length === 0) throw new ValidationError(`${sourceRecordsPath} is empty`);
|
|
10914
|
+
return records.map((record) => ({ record, runDir }));
|
|
10915
|
+
});
|
|
10916
|
+
mkdirSync3(outDir, { recursive: true });
|
|
10917
|
+
const artifactsByRunDir = /* @__PURE__ */ new Map();
|
|
10918
|
+
for (const [index, runDir] of runDirs.entries()) {
|
|
10919
|
+
artifactsByRunDir.set(
|
|
10920
|
+
runDir,
|
|
10921
|
+
materialize ? materializeRunArtifacts(runDir, outDir, index) : absoluteArtifacts(runDir)
|
|
10922
|
+
);
|
|
10923
|
+
}
|
|
10924
|
+
const artifactRoot = materialize ? outDir : "/";
|
|
10925
|
+
const normalized = rows.map(
|
|
10926
|
+
({ record, runDir }) => runRecordToProductBenchmarkRecord(
|
|
10927
|
+
record,
|
|
10928
|
+
runDir,
|
|
10929
|
+
artifactRoot,
|
|
10930
|
+
artifactsByRunDir.get(runDir),
|
|
10931
|
+
{ ...options, runDirs, backendVersion: opts.backendVersion }
|
|
10932
|
+
)
|
|
10933
|
+
);
|
|
10934
|
+
const manifest = buildProductBenchmarkManifest(normalized, {
|
|
10935
|
+
outDir,
|
|
10936
|
+
projectId: opts.projectId,
|
|
10937
|
+
benchmarkId: opts.benchmarkId,
|
|
10938
|
+
scenarioTagPrefix: opts.scenarioTagPrefix,
|
|
10939
|
+
mutableSurfaces: opts.mutableSurfaces,
|
|
10940
|
+
...options.substrate !== void 0 ? { substrate: options.substrate } : {}
|
|
10941
|
+
});
|
|
10942
|
+
const manifestPath = join4(outDir, "product-benchmark-manifest.json");
|
|
10943
|
+
const recordsPath = join4(outDir, "product-benchmark-records.jsonl");
|
|
10944
|
+
writeFileSync(manifestPath, `${JSON.stringify(manifest, null, 2)}
|
|
10945
|
+
`);
|
|
10946
|
+
writeFileSync(recordsPath, `${normalized.map((record) => JSON.stringify(record)).join("\n")}
|
|
10947
|
+
`);
|
|
10948
|
+
return { manifestPath, recordsPath, records: normalized.length };
|
|
10949
|
+
}
|
|
10950
|
+
function defaultScenarioTagPrefix(projectId) {
|
|
10951
|
+
return projectId.replace(/-agent$/, "") || projectId;
|
|
10952
|
+
}
|
|
10953
|
+
function resolveOptions(options) {
|
|
10954
|
+
return {
|
|
10955
|
+
projectId: options.projectId,
|
|
10956
|
+
benchmarkId: options.benchmarkId,
|
|
10957
|
+
agentProfilePath: options.agentProfilePath,
|
|
10958
|
+
passThreshold: options.passThreshold ?? 0.7,
|
|
10959
|
+
scenarioTagPrefix: options.scenarioTagPrefix ?? defaultScenarioTagPrefix(options.projectId),
|
|
10960
|
+
...options.fallbackProfileId !== void 0 ? { fallbackProfileId: options.fallbackProfileId } : {},
|
|
10961
|
+
mutableSurfaces: options.mutableSurfaces ?? productBenchmarkMutableSurfaces,
|
|
10962
|
+
...options.classifySplit !== void 0 ? { classifySplit: options.classifySplit } : {},
|
|
10963
|
+
...options.toolCallFallback !== void 0 ? { toolCallFallback: options.toolCallFallback } : {},
|
|
10964
|
+
backendVersion: options.backendVersion ?? packageVersion("@tangle-network/sandbox")
|
|
10965
|
+
};
|
|
10966
|
+
}
|
|
10967
|
+
|
|
10968
|
+
// src/product-benchmark/index.ts
|
|
10969
|
+
var productBenchmarkSplits = ["practice", "dev", "holdout", "safety", "sentinel"];
|
|
10970
|
+
function isObject2(value) {
|
|
10971
|
+
return Boolean(value) && typeof value === "object" && !Array.isArray(value);
|
|
10972
|
+
}
|
|
10973
|
+
function fail(path, message) {
|
|
10974
|
+
throw new ValidationError(`${path}: ${message}`);
|
|
10975
|
+
}
|
|
10976
|
+
function wrapValidationError(path, err) {
|
|
10977
|
+
if (err instanceof ValidationError) {
|
|
10978
|
+
throw new ValidationError(`${path}: ${err.message}`, { cause: err });
|
|
10979
|
+
}
|
|
10980
|
+
throw new ValidationError(`${path}: ${err instanceof Error ? err.message : String(err)}`, {
|
|
10981
|
+
cause: err
|
|
10982
|
+
});
|
|
10983
|
+
}
|
|
10984
|
+
function expectObject(value, path) {
|
|
10985
|
+
if (!isObject2(value)) fail(path, "must be an object");
|
|
10986
|
+
return value;
|
|
10987
|
+
}
|
|
10988
|
+
function expectString(value, path) {
|
|
10989
|
+
if (typeof value !== "string" || value.trim().length === 0)
|
|
10990
|
+
fail(path, "must be a non-empty string");
|
|
10991
|
+
return value;
|
|
10992
|
+
}
|
|
10993
|
+
function expectBoolean(value, path) {
|
|
10994
|
+
if (typeof value !== "boolean") fail(path, "must be a boolean");
|
|
10995
|
+
return value;
|
|
10996
|
+
}
|
|
10997
|
+
function expectNumber(value, path, opts = {}) {
|
|
10998
|
+
if (typeof value !== "number" || !Number.isFinite(value)) fail(path, "must be a finite number");
|
|
10999
|
+
if (opts.integer && !Number.isInteger(value)) fail(path, "must be an integer");
|
|
11000
|
+
if (opts.min !== void 0 && value < opts.min) fail(path, `must be >= ${opts.min}`);
|
|
11001
|
+
if (opts.max !== void 0 && value > opts.max) fail(path, `must be <= ${opts.max}`);
|
|
11002
|
+
return value;
|
|
11003
|
+
}
|
|
11004
|
+
function expectStringArray(value, path) {
|
|
11005
|
+
if (!Array.isArray(value)) fail(path, "must be an array");
|
|
11006
|
+
return value.map((entry, index) => expectString(entry, `${path}[${index}]`));
|
|
11007
|
+
}
|
|
11008
|
+
function expectObjectArray(value, path) {
|
|
11009
|
+
if (!Array.isArray(value)) fail(path, "must be an array");
|
|
11010
|
+
return value.map((entry, index) => expectObject(entry, `${path}[${index}]`));
|
|
11011
|
+
}
|
|
11012
|
+
function expectSplit(value, path) {
|
|
11013
|
+
const split = expectString(value, path);
|
|
11014
|
+
if (!productBenchmarkSplits.includes(split)) {
|
|
11015
|
+
fail(path, `must be one of ${productBenchmarkSplits.join(", ")}`);
|
|
11016
|
+
}
|
|
11017
|
+
return split;
|
|
11018
|
+
}
|
|
11019
|
+
function optionalString(value, path) {
|
|
11020
|
+
if (value === void 0) return void 0;
|
|
11021
|
+
return expectString(value, path);
|
|
11022
|
+
}
|
|
11023
|
+
function expectDimensions(value, path) {
|
|
11024
|
+
const obj = expectObject(value, path);
|
|
11025
|
+
const out = {};
|
|
11026
|
+
for (const [key, raw] of Object.entries(obj)) out[key] = expectNumber(raw, `${path}.${key}`);
|
|
11027
|
+
return out;
|
|
11028
|
+
}
|
|
11029
|
+
function expectRuntimeResolution(value, path) {
|
|
11030
|
+
const obj = expectObject(value, path);
|
|
11031
|
+
return {
|
|
11032
|
+
model: expectString(obj.model, `${path}.model`),
|
|
11033
|
+
harness: expectString(obj.harness, `${path}.harness`),
|
|
11034
|
+
backend: expectString(obj.backend, `${path}.backend`),
|
|
11035
|
+
...obj.reasoningEffort !== void 0 ? { reasoningEffort: optionalString(obj.reasoningEffort, `${path}.reasoningEffort`) } : {}
|
|
11036
|
+
};
|
|
11037
|
+
}
|
|
11038
|
+
function validateProductBenchmarkManifest(value) {
|
|
11039
|
+
const obj = expectObject(value, "manifest");
|
|
11040
|
+
if (obj.schemaVersion !== 1) fail("manifest.schemaVersion", "must be 1");
|
|
11041
|
+
const repo = expectObject(obj.repo, "manifest.repo");
|
|
11042
|
+
const substrate = expectObject(obj.substrate, "manifest.substrate");
|
|
11043
|
+
const profiles = expectObjectArray(obj.profiles, "manifest.profiles").map((profile, index) => ({
|
|
11044
|
+
id: expectString(profile.id, `manifest.profiles[${index}].id`),
|
|
11045
|
+
profileHash: expectString(profile.profileHash, `manifest.profiles[${index}].profileHash`),
|
|
11046
|
+
agentProfilePath: expectString(
|
|
11047
|
+
profile.agentProfilePath,
|
|
11048
|
+
`manifest.profiles[${index}].agentProfilePath`
|
|
11049
|
+
)
|
|
11050
|
+
}));
|
|
11051
|
+
const arms = expectObjectArray(obj.arms, "manifest.arms").map((arm, index) => ({
|
|
11052
|
+
id: expectString(arm.id, `manifest.arms[${index}].id`),
|
|
11053
|
+
profileId: expectString(arm.profileId, `manifest.arms[${index}].profileId`),
|
|
11054
|
+
mutableSurfaces: expectStringArray(
|
|
11055
|
+
arm.mutableSurfaces,
|
|
11056
|
+
`manifest.arms[${index}].mutableSurfaces`
|
|
11057
|
+
),
|
|
11058
|
+
policyAxes: expectObject(arm.policyAxes, `manifest.arms[${index}].policyAxes`)
|
|
11059
|
+
}));
|
|
11060
|
+
const scenarios = expectObjectArray(obj.scenarios, "manifest.scenarios").map(
|
|
11061
|
+
(scenario, index) => ({
|
|
11062
|
+
id: expectString(scenario.id, `manifest.scenarios[${index}].id`),
|
|
11063
|
+
split: expectSplit(scenario.split, `manifest.scenarios[${index}].split`),
|
|
11064
|
+
tags: expectStringArray(scenario.tags, `manifest.scenarios[${index}].tags`),
|
|
11065
|
+
sourceAllowedForSynthesis: expectBoolean(
|
|
11066
|
+
scenario.sourceAllowedForSynthesis,
|
|
11067
|
+
`manifest.scenarios[${index}].sourceAllowedForSynthesis`
|
|
11068
|
+
)
|
|
11069
|
+
})
|
|
11070
|
+
);
|
|
11071
|
+
const budgets = expectObject(obj.budgets, "manifest.budgets");
|
|
11072
|
+
if (profiles.length === 0) fail("manifest.profiles", "must contain at least one profile");
|
|
11073
|
+
if (arms.length === 0) fail("manifest.arms", "must contain at least one arm");
|
|
11074
|
+
if (scenarios.length === 0) fail("manifest.scenarios", "must contain at least one scenario");
|
|
11075
|
+
assertUnique(
|
|
11076
|
+
profiles.map((profile) => profile.id),
|
|
11077
|
+
"manifest.profiles.id"
|
|
11078
|
+
);
|
|
11079
|
+
assertUnique(
|
|
11080
|
+
arms.map((arm) => arm.id),
|
|
11081
|
+
"manifest.arms.id"
|
|
11082
|
+
);
|
|
11083
|
+
assertUnique(
|
|
11084
|
+
scenarios.map((scenario) => scenario.id),
|
|
11085
|
+
"manifest.scenarios.id"
|
|
11086
|
+
);
|
|
11087
|
+
const profileIds = new Set(profiles.map((profile) => profile.id));
|
|
11088
|
+
for (const arm of arms) {
|
|
11089
|
+
if (!profileIds.has(arm.profileId))
|
|
11090
|
+
fail(`manifest.arms.${arm.id}.profileId`, `unknown profile ${arm.profileId}`);
|
|
11091
|
+
}
|
|
11092
|
+
return {
|
|
11093
|
+
schemaVersion: 1,
|
|
11094
|
+
projectId: expectString(obj.projectId, "manifest.projectId"),
|
|
11095
|
+
benchmarkId: expectString(obj.benchmarkId, "manifest.benchmarkId"),
|
|
11096
|
+
repo: {
|
|
11097
|
+
url: expectString(repo.url, "manifest.repo.url"),
|
|
11098
|
+
commit: expectString(repo.commit, "manifest.repo.commit"),
|
|
11099
|
+
branch: expectString(repo.branch, "manifest.repo.branch")
|
|
11100
|
+
},
|
|
11101
|
+
substrate: {
|
|
11102
|
+
agentEval: expectString(substrate.agentEval, "manifest.substrate.agentEval"),
|
|
11103
|
+
agentRuntime: expectString(substrate.agentRuntime, "manifest.substrate.agentRuntime"),
|
|
11104
|
+
agentInterface: expectString(substrate.agentInterface, "manifest.substrate.agentInterface"),
|
|
11105
|
+
sandbox: expectString(substrate.sandbox, "manifest.substrate.sandbox"),
|
|
11106
|
+
...substrate.agentBench !== void 0 ? { agentBench: expectString(substrate.agentBench, "manifest.substrate.agentBench") } : {}
|
|
11107
|
+
},
|
|
11108
|
+
profiles,
|
|
11109
|
+
arms,
|
|
11110
|
+
scenarios,
|
|
11111
|
+
budgets: {
|
|
11112
|
+
maxUsd: expectNumber(budgets.maxUsd, "manifest.budgets.maxUsd", { min: 0 }),
|
|
11113
|
+
maxCells: expectNumber(budgets.maxCells, "manifest.budgets.maxCells", {
|
|
11114
|
+
min: 0,
|
|
11115
|
+
integer: true
|
|
11116
|
+
}),
|
|
11117
|
+
maxWallMs: expectNumber(budgets.maxWallMs, "manifest.budgets.maxWallMs", {
|
|
11118
|
+
min: 0,
|
|
11119
|
+
integer: true
|
|
11120
|
+
})
|
|
11121
|
+
},
|
|
11122
|
+
expectedArtifactDir: expectString(obj.expectedArtifactDir, "manifest.expectedArtifactDir")
|
|
11123
|
+
};
|
|
11124
|
+
}
|
|
11125
|
+
function validateProductBenchmarkRecord(value) {
|
|
11126
|
+
const obj = expectObject(value, "record");
|
|
11127
|
+
if (obj.schemaVersion !== 1) fail("record.schemaVersion", "must be 1");
|
|
11128
|
+
const agentProfile = expectObject(obj.agentProfile, "record.agentProfile");
|
|
11129
|
+
const model = expectObject(obj.model, "record.model");
|
|
11130
|
+
const backend = expectObject(obj.backend, "record.backend");
|
|
11131
|
+
const outcome = expectObject(obj.outcome, "record.outcome");
|
|
11132
|
+
const usage = expectObject(obj.usage, "record.usage");
|
|
11133
|
+
const integrity = expectObject(obj.integrity, "record.integrity");
|
|
11134
|
+
const artifacts = expectObject(obj.artifacts, "record.artifacts");
|
|
11135
|
+
const record = {
|
|
11136
|
+
schemaVersion: 1,
|
|
11137
|
+
projectId: expectString(obj.projectId, "record.projectId"),
|
|
11138
|
+
benchmarkId: expectString(obj.benchmarkId, "record.benchmarkId"),
|
|
11139
|
+
runId: expectString(obj.runId, "record.runId"),
|
|
11140
|
+
scenarioId: expectString(obj.scenarioId, "record.scenarioId"),
|
|
11141
|
+
split: expectSplit(obj.split, "record.split"),
|
|
11142
|
+
armId: expectString(obj.armId, "record.armId"),
|
|
11143
|
+
rep: expectNumber(obj.rep, "record.rep", { min: 1, integer: true }),
|
|
11144
|
+
agentProfile: {
|
|
11145
|
+
id: expectString(agentProfile.id, "record.agentProfile.id"),
|
|
11146
|
+
hash: expectString(agentProfile.hash, "record.agentProfile.hash"),
|
|
11147
|
+
path: expectString(agentProfile.path, "record.agentProfile.path"),
|
|
11148
|
+
declared: expectRuntimeResolution(agentProfile.declared, "record.agentProfile.declared"),
|
|
11149
|
+
resolved: expectRuntimeResolution(agentProfile.resolved, "record.agentProfile.resolved")
|
|
11150
|
+
},
|
|
11151
|
+
model: {
|
|
11152
|
+
provider: expectString(model.provider, "record.model.provider"),
|
|
11153
|
+
id: expectString(model.id, "record.model.id")
|
|
11154
|
+
},
|
|
11155
|
+
backend: {
|
|
11156
|
+
kind: expectString(backend.kind, "record.backend.kind"),
|
|
11157
|
+
version: expectString(backend.version, "record.backend.version")
|
|
11158
|
+
},
|
|
11159
|
+
outcome: {
|
|
11160
|
+
pass: expectBoolean(outcome.pass, "record.outcome.pass"),
|
|
11161
|
+
score: expectNumber(outcome.score, "record.outcome.score", { min: 0, max: 1 }),
|
|
11162
|
+
dimensions: expectDimensions(outcome.dimensions, "record.outcome.dimensions"),
|
|
11163
|
+
failureMode: outcome.failureMode === null ? null : expectString(outcome.failureMode, "record.outcome.failureMode")
|
|
11164
|
+
},
|
|
11165
|
+
usage: {
|
|
11166
|
+
inputTokens: expectNumber(usage.inputTokens, "record.usage.inputTokens", {
|
|
11167
|
+
min: 0,
|
|
11168
|
+
integer: true
|
|
11169
|
+
}),
|
|
11170
|
+
outputTokens: expectNumber(usage.outputTokens, "record.usage.outputTokens", {
|
|
11171
|
+
min: 0,
|
|
11172
|
+
integer: true
|
|
11173
|
+
}),
|
|
11174
|
+
costUsd: expectNumber(usage.costUsd, "record.usage.costUsd", { min: 0 }),
|
|
11175
|
+
wallMs: expectNumber(usage.wallMs, "record.usage.wallMs", { min: 0, integer: true }),
|
|
11176
|
+
toolCalls: expectNumber(usage.toolCalls, "record.usage.toolCalls", { min: 0, integer: true })
|
|
11177
|
+
},
|
|
11178
|
+
integrity: {
|
|
11179
|
+
realBackend: expectBoolean(integrity.realBackend, "record.integrity.realBackend"),
|
|
11180
|
+
rawCapture: expectBoolean(integrity.rawCapture, "record.integrity.rawCapture"),
|
|
11181
|
+
traceCapture: expectBoolean(integrity.traceCapture, "record.integrity.traceCapture"),
|
|
11182
|
+
noStubRows: expectBoolean(integrity.noStubRows, "record.integrity.noStubRows"),
|
|
11183
|
+
priced: expectBoolean(integrity.priced, "record.integrity.priced"),
|
|
11184
|
+
profileMaterialized: expectBoolean(
|
|
11185
|
+
integrity.profileMaterialized,
|
|
11186
|
+
"record.integrity.profileMaterialized"
|
|
11187
|
+
)
|
|
11188
|
+
},
|
|
11189
|
+
artifacts: {
|
|
11190
|
+
records: expectString(artifacts.records, "record.artifacts.records"),
|
|
11191
|
+
traces: expectString(artifacts.traces, "record.artifacts.traces"),
|
|
11192
|
+
raws: expectString(artifacts.raws, "record.artifacts.raws"),
|
|
11193
|
+
scores: expectString(artifacts.scores, "record.artifacts.scores"),
|
|
11194
|
+
workspace: expectString(artifacts.workspace, "record.artifacts.workspace")
|
|
11195
|
+
}
|
|
11196
|
+
};
|
|
11197
|
+
if (record.integrity.realBackend && record.usage.inputTokens + record.usage.outputTokens === 0) {
|
|
11198
|
+
fail("record.usage", "realBackend rows must carry non-zero token usage");
|
|
11199
|
+
}
|
|
11200
|
+
return record;
|
|
11201
|
+
}
|
|
11202
|
+
function productBenchmarkIntegrityFailures(record) {
|
|
11203
|
+
const failures = [];
|
|
11204
|
+
for (const [key, value] of Object.entries(record.integrity)) {
|
|
11205
|
+
if (!value) failures.push(`${record.runId}:${key}=false`);
|
|
11206
|
+
}
|
|
11207
|
+
return failures;
|
|
11208
|
+
}
|
|
11209
|
+
function readProductBenchmarkRecords(path) {
|
|
11210
|
+
const lines = readFileSync7(path, "utf8").split("\n").map((line) => line.trim()).filter(Boolean);
|
|
11211
|
+
const records = [];
|
|
11212
|
+
for (const [index, line] of lines.entries()) {
|
|
11213
|
+
try {
|
|
11214
|
+
records.push(validateProductBenchmarkRecord(JSON.parse(line)));
|
|
11215
|
+
} catch (err) {
|
|
11216
|
+
wrapValidationError(`${path}:${index + 1}`, err);
|
|
11217
|
+
}
|
|
11218
|
+
}
|
|
11219
|
+
return records;
|
|
11220
|
+
}
|
|
11221
|
+
function readProductBenchmarkManifest(path) {
|
|
11222
|
+
try {
|
|
11223
|
+
return validateProductBenchmarkManifest(JSON.parse(readFileSync7(path, "utf8")));
|
|
11224
|
+
} catch (err) {
|
|
11225
|
+
wrapValidationError(path, err);
|
|
11226
|
+
}
|
|
11227
|
+
}
|
|
11228
|
+
function validateProductBenchmarkRun(input) {
|
|
11229
|
+
const manifest = readProductBenchmarkManifest(input.manifestPath);
|
|
11230
|
+
const records = readProductBenchmarkRecords(input.recordsPath);
|
|
11231
|
+
const manifestProjectBench = `${manifest.projectId}/${manifest.benchmarkId}`;
|
|
11232
|
+
const integrityFailures = records.flatMap(productBenchmarkIntegrityFailures);
|
|
11233
|
+
const missingArtifacts = input.checkArtifacts === false ? [] : records.flatMap(
|
|
11234
|
+
(record) => missingArtifactsForRecord(record, input.artifactRoot ?? dirname4(input.recordsPath))
|
|
11235
|
+
);
|
|
11236
|
+
for (const [index, record] of records.entries()) {
|
|
11237
|
+
const recordProjectBench = `${record.projectId}/${record.benchmarkId}`;
|
|
11238
|
+
if (recordProjectBench !== manifestProjectBench) {
|
|
11239
|
+
fail(
|
|
11240
|
+
`records[${index}]`,
|
|
11241
|
+
`project/benchmark ${recordProjectBench} does not match manifest ${manifestProjectBench}`
|
|
11242
|
+
);
|
|
11243
|
+
}
|
|
11244
|
+
if (!manifest.arms.some((arm) => arm.id === record.armId))
|
|
11245
|
+
fail(`records[${index}].armId`, `unknown arm ${record.armId}`);
|
|
11246
|
+
if (!manifest.scenarios.some((scenario) => scenario.id === record.scenarioId)) {
|
|
11247
|
+
fail(`records[${index}].scenarioId`, `unknown scenario ${record.scenarioId}`);
|
|
11248
|
+
}
|
|
11249
|
+
}
|
|
11250
|
+
const repoFailures = ["url", "commit", "branch"].filter((key) => manifest.repo[key].trim().length === 0 || manifest.repo[key] === "unknown").map((key) => `manifest.repo.${key}`);
|
|
11251
|
+
const substrateFailures = ["agentEval", "agentRuntime", "agentInterface", "sandbox"].filter(
|
|
11252
|
+
(key) => manifest.substrate[key].trim().length === 0 || manifest.substrate[key] === "unknown"
|
|
11253
|
+
).map((key) => `manifest.substrate.${key}`);
|
|
11254
|
+
return {
|
|
11255
|
+
manifestPath: input.manifestPath,
|
|
11256
|
+
recordsPath: input.recordsPath,
|
|
11257
|
+
records: records.length,
|
|
11258
|
+
repoFailures,
|
|
11259
|
+
substrateFailures,
|
|
11260
|
+
projects: sortedUnique(records.map((record) => record.projectId)),
|
|
11261
|
+
benchmarks: sortedUnique(records.map((record) => record.benchmarkId)),
|
|
11262
|
+
arms: sortedUnique(records.map((record) => record.armId)),
|
|
11263
|
+
scenarios: sortedUnique(records.map((record) => record.scenarioId)),
|
|
11264
|
+
passed: records.filter((record) => record.outcome.pass).length,
|
|
11265
|
+
failed: records.filter((record) => !record.outcome.pass).length,
|
|
11266
|
+
inputTokens: sum3(records, (record) => record.usage.inputTokens),
|
|
11267
|
+
outputTokens: sum3(records, (record) => record.usage.outputTokens),
|
|
11268
|
+
costUsd: sum3(records, (record) => record.usage.costUsd),
|
|
11269
|
+
wallMs: sum3(records, (record) => record.usage.wallMs),
|
|
11270
|
+
integrityFailures,
|
|
11271
|
+
missingArtifacts
|
|
11272
|
+
};
|
|
11273
|
+
}
|
|
11274
|
+
function missingArtifactsForRecord(record, artifactRoot) {
|
|
11275
|
+
const missing = [];
|
|
11276
|
+
for (const [key, value] of Object.entries(record.artifacts)) {
|
|
11277
|
+
const path = resolveArtifactPath(value, artifactRoot);
|
|
11278
|
+
if (!existsSync6(path)) missing.push(`${record.runId}:${key}:${value}`);
|
|
11279
|
+
}
|
|
11280
|
+
return missing;
|
|
11281
|
+
}
|
|
11282
|
+
function resolveArtifactPath(value, artifactRoot) {
|
|
11283
|
+
return value.startsWith("/") ? value : join5(artifactRoot, value);
|
|
11284
|
+
}
|
|
11285
|
+
function assertUnique(values, path) {
|
|
11286
|
+
const seen = /* @__PURE__ */ new Set();
|
|
11287
|
+
for (const value of values) {
|
|
11288
|
+
if (seen.has(value)) fail(path, `duplicate ${value}`);
|
|
11289
|
+
seen.add(value);
|
|
11290
|
+
}
|
|
11291
|
+
}
|
|
11292
|
+
function sortedUnique(values) {
|
|
11293
|
+
return [...new Set(values)].sort();
|
|
11294
|
+
}
|
|
11295
|
+
function sum3(items, fn) {
|
|
11296
|
+
return items.reduce((total, item) => total + fn(item), 0);
|
|
11297
|
+
}
|
|
11298
|
+
function findProductBenchmarkArtifacts(runDir) {
|
|
11299
|
+
const manifestPath = join5(runDir, "product-benchmark-manifest.json");
|
|
11300
|
+
const recordsPath = join5(runDir, "product-benchmark-records.jsonl");
|
|
11301
|
+
if (existsSync6(manifestPath) && statSync3(manifestPath).isFile() && existsSync6(recordsPath) && statSync3(recordsPath).isFile()) {
|
|
11302
|
+
return { manifestPath, recordsPath };
|
|
11303
|
+
}
|
|
11304
|
+
return null;
|
|
11305
|
+
}
|
|
11306
|
+
function assertProductBenchmarkRun(runDir) {
|
|
11307
|
+
const artifacts = findProductBenchmarkArtifacts(runDir);
|
|
11308
|
+
if (!artifacts) {
|
|
11309
|
+
fail(runDir, "missing product-benchmark-manifest.json or product-benchmark-records.jsonl");
|
|
11310
|
+
}
|
|
11311
|
+
const report = validateProductBenchmarkRun({ ...artifacts, artifactRoot: runDir });
|
|
11312
|
+
const failures = [
|
|
11313
|
+
...report.repoFailures,
|
|
11314
|
+
...report.substrateFailures,
|
|
11315
|
+
...report.integrityFailures,
|
|
11316
|
+
...report.missingArtifacts
|
|
11317
|
+
];
|
|
11318
|
+
if (failures.length > 0) {
|
|
11319
|
+
fail(runDir, `product benchmark validation failed:
|
|
11320
|
+
${failures.join("\n")}`);
|
|
11321
|
+
}
|
|
11322
|
+
return report;
|
|
11323
|
+
}
|
|
10220
11324
|
export {
|
|
10221
11325
|
AGENT_PROFILE_KINDS,
|
|
10222
11326
|
ATTESTATION_ALGORITHM,
|
|
@@ -10375,7 +11479,6 @@ export {
|
|
|
10375
11479
|
assertNoHiddenLeak,
|
|
10376
11480
|
assertProductBenchmarkRun,
|
|
10377
11481
|
assertRealBackend,
|
|
10378
|
-
assertRecordIntegrity,
|
|
10379
11482
|
assertReleaseConfidence,
|
|
10380
11483
|
assertRunAgentProfileCell,
|
|
10381
11484
|
assertRunCaptured,
|
|
@@ -10421,11 +11524,9 @@ export {
|
|
|
10421
11524
|
causalAttribution,
|
|
10422
11525
|
checkBehavioralCanary,
|
|
10423
11526
|
checkCanaries,
|
|
10424
|
-
checkRecordIntegrity,
|
|
10425
11527
|
checkSlos,
|
|
10426
11528
|
checkTraceContracts,
|
|
10427
11529
|
clamp01,
|
|
10428
|
-
classifyEuAiRisk,
|
|
10429
11530
|
classifyFailure,
|
|
10430
11531
|
classifyTreatment,
|
|
10431
11532
|
cliffsDelta,
|
|
@@ -10504,7 +11605,6 @@ export {
|
|
|
10504
11605
|
errorStreakDetector,
|
|
10505
11606
|
estimateCost,
|
|
10506
11607
|
estimateTokens,
|
|
10507
|
-
euAiActReport,
|
|
10508
11608
|
evaluateActionPolicy,
|
|
10509
11609
|
evaluateContract,
|
|
10510
11610
|
evaluateHypothesis,
|
|
@@ -10513,7 +11613,6 @@ export {
|
|
|
10513
11613
|
evaluateReleaseConfidence,
|
|
10514
11614
|
evaluateTraceContract,
|
|
10515
11615
|
executeScenario,
|
|
10516
|
-
expandMatrix,
|
|
10517
11616
|
expandProfileAxes,
|
|
10518
11617
|
expectAgent,
|
|
10519
11618
|
exportProductBenchmark,
|
|
@@ -10552,7 +11651,6 @@ export {
|
|
|
10552
11651
|
formatFindings,
|
|
10553
11652
|
formatScorecardDiff,
|
|
10554
11653
|
gainHistogram,
|
|
10555
|
-
gatePerf,
|
|
10556
11654
|
gateTreatmentApplied,
|
|
10557
11655
|
gateTreatmentFromMetrics,
|
|
10558
11656
|
gateTreatmentFromSpans,
|
|
@@ -10632,7 +11730,6 @@ export {
|
|
|
10632
11730
|
modelPriceKey,
|
|
10633
11731
|
mulberry32,
|
|
10634
11732
|
multiToolchainLayer,
|
|
10635
|
-
nistAiRmfReport,
|
|
10636
11733
|
noProgressDetector,
|
|
10637
11734
|
normalizeScores,
|
|
10638
11735
|
notBlocked,
|
|
@@ -10697,7 +11794,6 @@ export {
|
|
|
10697
11794
|
referenceReplayScenarioToRunScore,
|
|
10698
11795
|
regexMatch,
|
|
10699
11796
|
regexMatches,
|
|
10700
|
-
renderMarkdown,
|
|
10701
11797
|
renderMarkdownReport,
|
|
10702
11798
|
renderPlaybookMarkdown,
|
|
10703
11799
|
renderPreferenceMemoryMarkdown,
|
|
@@ -10745,7 +11841,6 @@ export {
|
|
|
10745
11841
|
runsForScenario,
|
|
10746
11842
|
scalarScore,
|
|
10747
11843
|
scanForMuffledGates,
|
|
10748
|
-
scenarioKey,
|
|
10749
11844
|
scoreContinuity,
|
|
10750
11845
|
scoreFromEvals,
|
|
10751
11846
|
scoreKnowledgeReadiness,
|
|
@@ -10762,7 +11857,6 @@ export {
|
|
|
10762
11857
|
sentenceReorderMutator,
|
|
10763
11858
|
serializeFeedbackTrajectoriesJsonl,
|
|
10764
11859
|
signManifest,
|
|
10765
|
-
soc2Report,
|
|
10766
11860
|
spearmanR,
|
|
10767
11861
|
splitGold,
|
|
10768
11862
|
statusAdvanced,
|
|
@@ -10771,12 +11865,10 @@ export {
|
|
|
10771
11865
|
stringField,
|
|
10772
11866
|
stripFencedJson,
|
|
10773
11867
|
subjectiveEval,
|
|
10774
|
-
summarize,
|
|
10775
11868
|
summarizeBackendIntegrity,
|
|
10776
11869
|
summarizeHarnessResults,
|
|
10777
11870
|
summarizePrReviewBenchmark,
|
|
10778
11871
|
summarizePreferenceMemory,
|
|
10779
|
-
summarizeRecords,
|
|
10780
11872
|
summaryTable,
|
|
10781
11873
|
testJudge,
|
|
10782
11874
|
textInSnapshot,
|