@tangle-network/agent-eval 0.115.1 → 0.115.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (79) hide show
  1. package/CHANGELOG.md +20 -0
  2. package/dist/analyst/index.d.ts +7 -7
  3. package/dist/analyst/index.js +5 -5
  4. package/dist/{analyze-runs-Dmz6LA9e.d.ts → analyze-runs-BYHg6Irm.d.ts} +3 -3
  5. package/dist/belief-state/index.d.ts +3 -3
  6. package/dist/belief-state/index.js +1 -1
  7. package/dist/benchmarks/index.d.ts +3 -3
  8. package/dist/benchmarks/index.js +6 -6
  9. package/dist/campaign/index.d.ts +12 -12
  10. package/dist/campaign/index.js +6 -6
  11. package/dist/{chunk-DRPIZQIT.js → chunk-4D5RVB3W.js} +2 -2
  12. package/dist/{chunk-DWLIGZBX.js → chunk-5S5NJ63F.js} +2 -2
  13. package/dist/{chunk-VK6HBGAE.js → chunk-5UF54T55.js} +53 -1
  14. package/dist/chunk-5UF54T55.js.map +1 -0
  15. package/dist/{chunk-N6MTC3GK.js → chunk-ADYLPOSX.js} +2 -2
  16. package/dist/{chunk-IMWDSFUM.js → chunk-DXZRATT5.js} +2 -2
  17. package/dist/{chunk-S42AWHMP.js → chunk-E4BUPP7Z.js} +135 -17
  18. package/dist/chunk-E4BUPP7Z.js.map +1 -0
  19. package/dist/{chunk-FUCQVFMU.js → chunk-GY4SYVPJ.js} +12 -3
  20. package/dist/chunk-GY4SYVPJ.js.map +1 -0
  21. package/dist/{chunk-LVTGFSHF.js → chunk-I6LVHOV3.js} +2 -2
  22. package/dist/{chunk-WBOGKYM4.js → chunk-J6P6PK2R.js} +48 -7
  23. package/dist/chunk-J6P6PK2R.js.map +1 -0
  24. package/dist/{chunk-KDCMEZDI.js → chunk-KG4TD7EQ.js} +6 -6
  25. package/dist/{chunk-LOBMT6SB.js → chunk-ONM6PEAE.js} +3 -3
  26. package/dist/{chunk-AN5UYSVD.js → chunk-QMXXSNC4.js} +2 -2
  27. package/dist/{chunk-RSVSSZKF.js → chunk-TLDB7WRY.js} +2 -2
  28. package/dist/{chunk-I2HNIE6N.js → chunk-WSBUZMBU.js} +4 -4
  29. package/dist/cli.js +2 -2
  30. package/dist/{code-agent-session-yitf9I-F.d.ts → code-agent-session-D-g04tcy.d.ts} +8 -1
  31. package/dist/contract/index.d.ts +15 -15
  32. package/dist/contract/index.js +7 -7
  33. package/dist/{control-U8LBKUES.d.ts → control-CcBiAEnn.d.ts} +1 -1
  34. package/dist/control.d.ts +2 -2
  35. package/dist/control.js +2 -2
  36. package/dist/{default-registry-Bcf1uKVI.d.ts → default-registry-DltpYR5u.d.ts} +1 -1
  37. package/dist/{gepa-DolL_Fko.d.ts → gepa-dne9JDPL.d.ts} +1 -1
  38. package/dist/hosted/index.d.ts +4 -4
  39. package/dist/{index-CWr5SIG-.d.ts → index-BTEpx9He.d.ts} +2 -2
  40. package/dist/index.d.ts +22 -22
  41. package/dist/index.js +14 -12
  42. package/dist/index.js.map +1 -1
  43. package/dist/{insight-report-D4cXFsLt.d.ts → insight-report-IwwvqZZv.d.ts} +20 -2
  44. package/dist/{kind-factory-20hcaYpf.d.ts → kind-factory-DcNg13sZ.d.ts} +1 -1
  45. package/dist/meta-eval/index.d.ts +2 -2
  46. package/dist/multishot/index.d.ts +2 -2
  47. package/dist/openapi.json +1 -1
  48. package/dist/{policy-edit-az2qRmvN.d.ts → policy-edit-RLn8GWof.d.ts} +2 -2
  49. package/dist/{pre-registration-oNItiRBb.d.ts → pre-registration-D8h7ZxNL.d.ts} +3 -3
  50. package/dist/{provenance-BZmpWmn4.d.ts → provenance-Bibyg1U9.d.ts} +3 -3
  51. package/dist/{release-report-oBfOz8ku.d.ts → release-report-CCtzajxP.d.ts} +2 -2
  52. package/dist/reporting.d.ts +4 -4
  53. package/dist/{researcher-CaH0CwFC.d.ts → researcher-Dq-EtpbE.d.ts} +2 -2
  54. package/dist/rl.d.ts +6 -6
  55. package/dist/rl.js +3 -3
  56. package/dist/{rubric-predictive-validity-C-fMteAW.d.ts → rubric-predictive-validity-DYTLjGWu.d.ts} +1 -1
  57. package/dist/{run-record-DksGsfgv.d.ts → run-record-B7RTi_ix.d.ts} +34 -2
  58. package/dist/{runtime-trajectory-h5i0SZUj.d.ts → runtime-trajectory-Dws7Kpgi.d.ts} +1 -1
  59. package/dist/{semantic-concept-judge-CpzbtwD0.d.ts → semantic-concept-judge-DxJmRkyJ.d.ts} +1 -1
  60. package/dist/{summary-report-Bz-0-t8v.d.ts → summary-report-BJ5aNwZ1.d.ts} +1 -1
  61. package/dist/traces.d.ts +1 -1
  62. package/dist/traces.js +2 -2
  63. package/dist/{types-CgSlO6wT.d.ts → types-C5gJrOVT.d.ts} +1 -1
  64. package/dist/wire/index.js +2 -2
  65. package/package.json +1 -1
  66. package/dist/chunk-FUCQVFMU.js.map +0 -1
  67. package/dist/chunk-S42AWHMP.js.map +0 -1
  68. package/dist/chunk-VK6HBGAE.js.map +0 -1
  69. package/dist/chunk-WBOGKYM4.js.map +0 -1
  70. /package/dist/{chunk-DRPIZQIT.js.map → chunk-4D5RVB3W.js.map} +0 -0
  71. /package/dist/{chunk-DWLIGZBX.js.map → chunk-5S5NJ63F.js.map} +0 -0
  72. /package/dist/{chunk-N6MTC3GK.js.map → chunk-ADYLPOSX.js.map} +0 -0
  73. /package/dist/{chunk-IMWDSFUM.js.map → chunk-DXZRATT5.js.map} +0 -0
  74. /package/dist/{chunk-LVTGFSHF.js.map → chunk-I6LVHOV3.js.map} +0 -0
  75. /package/dist/{chunk-KDCMEZDI.js.map → chunk-KG4TD7EQ.js.map} +0 -0
  76. /package/dist/{chunk-LOBMT6SB.js.map → chunk-ONM6PEAE.js.map} +0 -0
  77. /package/dist/{chunk-AN5UYSVD.js.map → chunk-QMXXSNC4.js.map} +0 -0
  78. /package/dist/{chunk-RSVSSZKF.js.map → chunk-TLDB7WRY.js.map} +0 -0
  79. /package/dist/{chunk-I2HNIE6N.js.map → chunk-WSBUZMBU.js.map} +0 -0
package/CHANGELOG.md CHANGED
@@ -4,6 +4,26 @@ All notable changes to `@tangle-network/agent-eval` and its sibling `agent-eval-
4
4
 
5
5
  ---
6
6
 
7
+ ## [0.115.3] — 2026-07-12 — fail-closed structured output parsing
8
+
9
+ ### Fixed
10
+
11
+ - `callLlmJson()` now rejects responses terminated with finish reason `length`, even when the returned prefix happens to parse as JSON.
12
+ - JSON extraction no longer descends into a valid nested object or array when a response declares an incomplete top-level JSON root.
13
+
14
+ This patch prevents truncated structured responses from being silently accepted under the wrong response shape.
15
+ Consumers of `callLlmJson()` should update.
16
+
17
+ ## [0.115.2] — 2026-07-12 — truthful code-agent session accounting
18
+
19
+ ### Fixed
20
+
21
+ - fix(contract): ingest direct Codex 0.144.x exec JSONL lifecycle, tool, patch, terminal, and token events without double-counting transitions or reasoning tokens.
22
+ - fix(contract): preserve observed, estimated, and uncaptured USD provenance through code-agent session intake and analyzeRuns.
23
+
24
+ This patch corrects imported trace and cost semantics while retaining backward-compatible serialized RunRecords.
25
+ Consumers importing code-agent execution traces should update.
26
+
7
27
  ## [0.115.1] — 2026-07-11 — fair cross-surface baseline selection
8
28
 
9
29
  ### Fixed
@@ -1,20 +1,20 @@
1
1
  import { M as MultiLayerVerifier, V as VerifyOptions, S as Severity } from '../multi-layer-verifier-BsqKuLyN.js';
2
- import { R as RunCritic, S as SemanticConceptJudgeOptions, a as RunTrace, b as SemanticConceptJudgeInput, B as BehavioralMetrics } from '../semantic-concept-judge-CpzbtwD0.js';
3
- export { C as CreateAnalystAiConfig, D as DEFAULT_TRACE_ANALYST_KINDS, c as DiffPolicy, F as FAILURE_MODE_KIND_SPEC, d as FINDING_SUBJECT_GRAMMAR_PROMPT, e as FINDING_SUBJECT_KINDS, f as FINDING_SUBJECT_SYNTAX, g as FindingSubject, h as FindingSubjectKind, i as FindingSubjectStringSchema, j as FindingsDiff, k as FindingsStore, I as IMPROVEMENT_KIND_SPEC, K as KIND_EXPECTED_SUBJECTS, l as KNOWLEDGE_GAP_KIND_SPEC, m as KNOWLEDGE_POISONING_KIND_SPEC, P as PersistedFinding, n as SKILL_USAGE_ANALYST, o as SkillUsageAnalyst, p as SkillUsageRecord, q as SkillUsageReport, r as SkillUsageScanConfig, s as buildSkillUsageReport, t as createAnalystAi, u as defaultIsMaterial, v as diffFindings, w as emitSkillUsageFindings, x as findingSubjectGrammarPromptFor, y as parseFindingSubject, z as renderFindingSubject } from '../semantic-concept-judge-CpzbtwD0.js';
2
+ import { R as RunCritic, S as SemanticConceptJudgeOptions, a as RunTrace, b as SemanticConceptJudgeInput, B as BehavioralMetrics } from '../semantic-concept-judge-DxJmRkyJ.js';
3
+ export { C as CreateAnalystAiConfig, D as DEFAULT_TRACE_ANALYST_KINDS, c as DiffPolicy, F as FAILURE_MODE_KIND_SPEC, d as FINDING_SUBJECT_GRAMMAR_PROMPT, e as FINDING_SUBJECT_KINDS, f as FINDING_SUBJECT_SYNTAX, g as FindingSubject, h as FindingSubjectKind, i as FindingSubjectStringSchema, j as FindingsDiff, k as FindingsStore, I as IMPROVEMENT_KIND_SPEC, K as KIND_EXPECTED_SUBJECTS, l as KNOWLEDGE_GAP_KIND_SPEC, m as KNOWLEDGE_POISONING_KIND_SPEC, P as PersistedFinding, n as SKILL_USAGE_ANALYST, o as SkillUsageAnalyst, p as SkillUsageRecord, q as SkillUsageReport, r as SkillUsageScanConfig, s as buildSkillUsageReport, t as createAnalystAi, u as defaultIsMaterial, v as diffFindings, w as emitSkillUsageFindings, x as findingSubjectGrammarPromptFor, y as parseFindingSubject, z as renderFindingSubject } from '../semantic-concept-judge-DxJmRkyJ.js';
4
4
  import { b as JudgeFn, a as JudgeInput } from '../types-C7DGg5ex.js';
5
- import { a as Analyst, g as AnalystSeverity, A as AnalystFinding } from '../kind-factory-20hcaYpf.js';
6
- export { h as ANALYST_SEVERITIES, b as AnalystContext, i as AnalystCost, j as AnalystInputKind, k as AnalystRequirements, f as AnalystRunEvent, e as AnalystRunInputs, d as AnalystRunResult, c as AnalystRunSummary, l as ChatCallOpts, C as ChatClient, m as ChatRequest, n as ChatResponse, o as ChatTransport, p as CliBridgeTransportOpts, q as CreateChatClientOpts, r as CreateTraceAnalystKindOpts, D as DirectProviderTransportOpts, E as EvidenceRef, M as MockTransportOpts, R as RAW_FINDING_SCHEMA_PROMPT, s as RawAnalystFinding, t as RawAnalystFindingSchema, u as RouterTransportOpts, S as SandboxSdkTransportOpts, v as TraceAnalystGolden, T as TraceAnalystKindSpec, w as computeFindingId, x as createChatClient, y as createTraceAnalystKind, z as makeFinding, B as parseRawFinding, F as renderPriorFindings } from '../kind-factory-20hcaYpf.js';
5
+ import { a as Analyst, g as AnalystSeverity, A as AnalystFinding } from '../kind-factory-DcNg13sZ.js';
6
+ export { h as ANALYST_SEVERITIES, b as AnalystContext, i as AnalystCost, j as AnalystInputKind, k as AnalystRequirements, f as AnalystRunEvent, e as AnalystRunInputs, d as AnalystRunResult, c as AnalystRunSummary, l as ChatCallOpts, C as ChatClient, m as ChatRequest, n as ChatResponse, o as ChatTransport, p as CliBridgeTransportOpts, q as CreateChatClientOpts, r as CreateTraceAnalystKindOpts, D as DirectProviderTransportOpts, E as EvidenceRef, M as MockTransportOpts, R as RAW_FINDING_SCHEMA_PROMPT, s as RawAnalystFinding, t as RawAnalystFindingSchema, u as RouterTransportOpts, S as SandboxSdkTransportOpts, v as TraceAnalystGolden, T as TraceAnalystKindSpec, w as computeFindingId, x as createChatClient, y as createTraceAnalystKind, z as makeFinding, B as parseRawFinding, F as renderPriorFindings } from '../kind-factory-DcNg13sZ.js';
7
7
  import { TCloud } from '@tangle-network/tcloud';
8
8
  import { T as TraceAnalysisStore } from '../store-C1YxJDEK.js';
9
- export { a as AnalystHooks, A as AnalystRegistry, b as AnalystRegistryOptions, B as BudgetPolicy, D as DefaultAnalystRegistryOptions, R as RegistryRunOpts, c as buildDefaultAnalystRegistry } from '../default-registry-Bcf1uKVI.js';
10
- export { F as FindingToPolicyEditOptions, P as POLICY_EDIT_AXES, a as POLICY_EDIT_TARGET_SURFACES, b as PolicyEdit, c as PolicyEditAdmission, d as PolicyEditAdmissionOptions, e as PolicyEditAxis, f as PolicyEditChange, g as PolicyEditExpectedGain, h as PolicyEditGainDirection, i as PolicyEditGainUnit, j as PolicyEditInit, k as PolicyEditRisk, l as PolicyEditSchemaVersion, m as PolicyEditSource, n as PolicyEditTarget, o as PolicyEditTargetSurface, p as PolicyEditValidationError, q as admitPolicyEdit, r as applyPolicyEditToSurface, s as computePolicyEditId, t as isPolicyEdit, u as makePolicyEdit, v as policyEditFromFinding, w as policyEditsFromFindings, x as scorePolicyEditReadiness, y as validatePolicyEdit } from '../policy-edit-az2qRmvN.js';
9
+ export { a as AnalystHooks, A as AnalystRegistry, b as AnalystRegistryOptions, B as BudgetPolicy, D as DefaultAnalystRegistryOptions, R as RegistryRunOpts, c as buildDefaultAnalystRegistry } from '../default-registry-DltpYR5u.js';
10
+ export { F as FindingToPolicyEditOptions, P as POLICY_EDIT_AXES, a as POLICY_EDIT_TARGET_SURFACES, b as PolicyEdit, c as PolicyEditAdmission, d as PolicyEditAdmissionOptions, e as PolicyEditAxis, f as PolicyEditChange, g as PolicyEditExpectedGain, h as PolicyEditGainDirection, i as PolicyEditGainUnit, j as PolicyEditInit, k as PolicyEditRisk, l as PolicyEditSchemaVersion, m as PolicyEditSource, n as PolicyEditTarget, o as PolicyEditTargetSurface, p as PolicyEditValidationError, q as admitPolicyEdit, r as applyPolicyEditToSurface, s as computePolicyEditId, t as isPolicyEdit, u as makePolicyEdit, v as policyEditFromFinding, w as policyEditsFromFindings, x as scorePolicyEditReadiness, y as validatePolicyEdit } from '../policy-edit-RLn8GWof.js';
11
11
  import { L as LlmClientOptions } from '../llm-client-DyqEH4jH.js';
12
12
  import { AxFunction } from '@ax-llm/ax';
13
13
  import '../verdict-C9MlYujm.js';
14
14
  import 'zod';
15
15
  import '../schema-SGWcK9wa.js';
16
16
  import '../store-BsVi7ncX.js';
17
- import '../run-record-DksGsfgv.js';
17
+ import '../run-record-B7RTi_ix.js';
18
18
  import '@tangle-network/agent-interface';
19
19
  import '../errors-oeQrLqXC.js';
20
20
  import '../raw-provider-sink-C46HDghv.js';
@@ -11,12 +11,12 @@ import {
11
11
  diffFindings,
12
12
  emitSkillUsageFindings,
13
13
  runSemanticConceptJudge
14
- } from "../chunk-I2HNIE6N.js";
14
+ } from "../chunk-WSBUZMBU.js";
15
15
  import {
16
16
  behavioralAnalyst,
17
17
  buildDefaultAnalystRegistry,
18
18
  deriveEfficiencyFindings
19
- } from "../chunk-LVTGFSHF.js";
19
+ } from "../chunk-I6LVHOV3.js";
20
20
  import {
21
21
  POLICY_EDIT_AXES,
22
22
  POLICY_EDIT_TARGET_SURFACES,
@@ -33,7 +33,7 @@ import {
33
33
  policyEditsFromFindings,
34
34
  scorePolicyEditReadiness,
35
35
  validatePolicyEdit
36
- } from "../chunk-AN5UYSVD.js";
36
+ } from "../chunk-QMXXSNC4.js";
37
37
  import {
38
38
  ANALYST_SEVERITIES,
39
39
  AnalystRegistry,
@@ -62,11 +62,11 @@ import {
62
62
  renderPriorFindings,
63
63
  stripCodeFences,
64
64
  structureFindings
65
- } from "../chunk-DWLIGZBX.js";
65
+ } from "../chunk-5S5NJ63F.js";
66
66
  import "../chunk-LNQEP766.js";
67
67
  import "../chunk-XJYR7XFV.js";
68
68
  import "../chunk-VSMTAMNK.js";
69
- import "../chunk-FUCQVFMU.js";
69
+ import "../chunk-GY4SYVPJ.js";
70
70
  import "../chunk-PC4UYEBM.js";
71
71
  import "../chunk-ONWEPEDO.js";
72
72
  import "../chunk-PZ5AY32C.js";
@@ -1,7 +1,7 @@
1
- import { A as AnalystRegistry } from './default-registry-Bcf1uKVI.js';
1
+ import { A as AnalystRegistry } from './default-registry-DltpYR5u.js';
2
2
  import { D as DatasetScenario } from './dataset-NENEzRgk.js';
3
- import { R as RunRecord } from './run-record-DksGsfgv.js';
4
- import { I as InsightReport } from './insight-report-D4cXFsLt.js';
3
+ import { R as RunRecord } from './run-record-B7RTi_ix.js';
4
+ import { I as InsightReport } from './insight-report-IwwvqZZv.js';
5
5
 
6
6
  /**
7
7
  * # `analyzeRuns()` — turn a set of agent runs into an actionable decision packet.
@@ -1,9 +1,9 @@
1
1
  import { c as CalibrationReport } from '../calibration-Dz8TQV4y.js';
2
2
  import { O as OffPolicyEstimate, a as OffPolicyOptions, b as OffPolicyTrajectory } from '../off-policy-DiwuKKg7.js';
3
- import { d as CodeAgentSessionSource, a as CodeAgentSessionIntakeOptions, c as CodeAgentSessionMetrics, C as CodeAgentSessionDiagnostic } from '../code-agent-session-yitf9I-F.js';
4
- import { R as RunRecord, b as RunSplitTag } from '../run-record-DksGsfgv.js';
3
+ import { d as CodeAgentSessionSource, a as CodeAgentSessionIntakeOptions, c as CodeAgentSessionMetrics, C as CodeAgentSessionDiagnostic } from '../code-agent-session-D-g04tcy.js';
4
+ import { R as RunRecord, b as RunSplitTag } from '../run-record-B7RTi_ix.js';
5
5
  import { T as TraceStore } from '../store-BsVi7ncX.js';
6
- import { R as RuntimeTrajectoryRecord, P as ProjectRuntimeTrajectoryEvidenceOptions, a as RuntimeTrajectoryEvidenceProjection } from '../runtime-trajectory-h5i0SZUj.js';
6
+ import { R as RuntimeTrajectoryRecord, P as ProjectRuntimeTrajectoryEvidenceOptions, a as RuntimeTrajectoryEvidenceProjection } from '../runtime-trajectory-Dws7Kpgi.js';
7
7
  import '../schema-SGWcK9wa.js';
8
8
  import '../outcome-store-rnXLEqSn.js';
9
9
  import '@tangle-network/agent-interface';
@@ -4,7 +4,7 @@ import {
4
4
  fromKimiCodeSession,
5
5
  fromOpenCodeSession,
6
6
  fromPiSession
7
- } from "../chunk-S42AWHMP.js";
7
+ } from "../chunk-E4BUPP7Z.js";
8
8
  import {
9
9
  calibrationFromPairs
10
10
  } from "../chunk-NPCTHQIO.js";
@@ -1,6 +1,6 @@
1
- export { B as BENCHMARK_SPLIT_SEED, a as BenchmarkAdapter, b as BenchmarkDatasetItem, k as BenchmarkDistribution, c as BenchmarkEvaluation, d as BenchmarkFamily, l as BenchmarkMetricCalibrationOptions, m as BenchmarkMetricCalibrationResult, n as BenchmarkReport, e as BenchmarkResponder, o as BenchmarkRunOptions, p as BenchmarkRunResult, f as BenchmarkScenario, q as BenchmarkSliceSummary, g as BenchmarkSource, h as BenchmarkTaskKind, r as BuildStandardRetrievalItemsOptions, R as RetrievalIdAdapterOptions, S as StandardRetrievalArtifact, s as StandardRetrievalDocument, t as StandardRetrievalEvaluationOptions, u as StandardRetrievalPayload, v as StandardRetrievalQrel, w as StandardRetrievalQuery, x as StandardRetrievalResult, y as buildStandardRetrievalItems, z as calibrateBenchmarkMetric, A as createRetrievalIdBenchmarkAdapter, i as deterministicSplit, C as evaluateStandardRetrieval, D as normalizeRetrievedDocumentIds, E as parseBeirCorpusJsonl, F as parseBeirQueriesJsonl, G as parseJsonlRows, H as parseQrels, I as parseTsvRows, J as renderBenchmarkReportMarkdown, K as retrievalMetricsAtCutoff, L as routing, M as runBenchmarkAdapter, N as summarizeBenchmarkCampaign } from '../index-CWr5SIG-.js';
2
- import '../types-CgSlO6wT.js';
3
- import '../run-record-DksGsfgv.js';
1
+ export { B as BENCHMARK_SPLIT_SEED, a as BenchmarkAdapter, b as BenchmarkDatasetItem, k as BenchmarkDistribution, c as BenchmarkEvaluation, d as BenchmarkFamily, l as BenchmarkMetricCalibrationOptions, m as BenchmarkMetricCalibrationResult, n as BenchmarkReport, e as BenchmarkResponder, o as BenchmarkRunOptions, p as BenchmarkRunResult, f as BenchmarkScenario, q as BenchmarkSliceSummary, g as BenchmarkSource, h as BenchmarkTaskKind, r as BuildStandardRetrievalItemsOptions, R as RetrievalIdAdapterOptions, S as StandardRetrievalArtifact, s as StandardRetrievalDocument, t as StandardRetrievalEvaluationOptions, u as StandardRetrievalPayload, v as StandardRetrievalQrel, w as StandardRetrievalQuery, x as StandardRetrievalResult, y as buildStandardRetrievalItems, z as calibrateBenchmarkMetric, A as createRetrievalIdBenchmarkAdapter, i as deterministicSplit, C as evaluateStandardRetrieval, D as normalizeRetrievedDocumentIds, E as parseBeirCorpusJsonl, F as parseBeirQueriesJsonl, G as parseJsonlRows, H as parseQrels, I as parseTsvRows, J as renderBenchmarkReportMarkdown, K as retrievalMetricsAtCutoff, L as routing, M as runBenchmarkAdapter, N as summarizeBenchmarkCampaign } from '../index-BTEpx9He.js';
2
+ import '../types-C5gJrOVT.js';
3
+ import '../run-record-B7RTi_ix.js';
4
4
  import '@tangle-network/agent-interface';
5
5
  import '../errors-oeQrLqXC.js';
6
6
  import '../schema-SGWcK9wa.js';
@@ -17,21 +17,21 @@ import {
17
17
  runBenchmarkAdapter,
18
18
  summarizeBenchmarkCampaign
19
19
  } from "../chunk-3LXTCTWL.js";
20
- import "../chunk-KDCMEZDI.js";
21
- import "../chunk-N6MTC3GK.js";
20
+ import "../chunk-KG4TD7EQ.js";
21
+ import "../chunk-ADYLPOSX.js";
22
22
  import "../chunk-VI2UW6B6.js";
23
23
  import "../chunk-FAOEFFRT.js";
24
- import "../chunk-AN5UYSVD.js";
25
- import "../chunk-DWLIGZBX.js";
24
+ import "../chunk-QMXXSNC4.js";
25
+ import "../chunk-5S5NJ63F.js";
26
26
  import "../chunk-ARU2PZFM.js";
27
27
  import "../chunk-PJQFMIOX.js";
28
28
  import "../chunk-RPDDVKI7.js";
29
29
  import "../chunk-GGE4NNQT.js";
30
30
  import "../chunk-LNQEP766.js";
31
- import "../chunk-VK6HBGAE.js";
31
+ import "../chunk-5UF54T55.js";
32
32
  import "../chunk-XJYR7XFV.js";
33
33
  import "../chunk-VSMTAMNK.js";
34
- import "../chunk-FUCQVFMU.js";
34
+ import "../chunk-GY4SYVPJ.js";
35
35
  import "../chunk-PC4UYEBM.js";
36
36
  import "../chunk-ONWEPEDO.js";
37
37
  import "../chunk-PZ5AY32C.js";
@@ -1,20 +1,20 @@
1
- import { P as PairedArmsComparison, S as SignedManifest, B as BackendIntegrityReport, C as CompletionRequirement, R as RuntimeEventLike, a as CompletionVerdict, b as ProducedState, c as CorrectnessChecker } from '../pre-registration-oNItiRBb.js';
2
- export { L as LlmJudgeDimension, d as LlmJudgeOptions, l as llmJudge } from '../pre-registration-oNItiRBb.js';
1
+ import { P as PairedArmsComparison, S as SignedManifest, B as BackendIntegrityReport, C as CompletionRequirement, R as RuntimeEventLike, a as CompletionVerdict, b as ProducedState, c as CorrectnessChecker } from '../pre-registration-D8h7ZxNL.js';
2
+ export { L as LlmJudgeDimension, d as LlmJudgeOptions, l as llmJudge } from '../pre-registration-D8h7ZxNL.js';
3
3
  import { A as AnalyzeTracesOptions, a as AnalyzeTracesInput, b as AnalyzeTracesResult } from '../analyst-C8HHvfJp.js';
4
- import { S as Scenario, M as MutableSurface, D as DispatchContext, b as JudgeConfig, g as Gate, e as GenerationRecord, J as JudgeScore, L as LabeledScenarioStore, s as LabeledScenarioWrite, t as LabeledScenarioSampleArgs, u as LabeledScenarioRecord, v as LabelTrust, f as SurfaceProposer, w as ProposedCandidate, x as ProposeContext, m as CodeSurface, y as LabeledScenarioSource, C as CampaignResult } from '../types-CgSlO6wT.js';
5
- export { i as CampaignAggregates, j as CampaignArtifactWriter, k as CampaignCellResult, l as CampaignCostMeter, z as CampaignTokenUsage, d as CampaignTraceWriter, c as DispatchFn, n as GateContext, h as GateDecision, G as GateResult, o as GenerationCandidate, A as JudgeAggregate, a as JudgeDimension, p as Mutator, O as OptimizationProposer, q as OptimizerConfig, P as ParetoParent, R as RedactionStatus, B as ScenarioAggregate, r as SessionScript, T as TraceSpan, E as isProposedCandidate, F as labelTrustRank } from '../types-CgSlO6wT.js';
6
- import { C as CampaignRunPlan, P as PlanCampaignRunOptions, b as RunCampaignOptions, c as RunImprovementLoopOptions } from '../gepa-DolL_Fko.js';
7
- export { f as CampaignRunPlanCell, h as GepaProposerConstraints, G as GepaProposerOptions, O as OpenAutoPrOptions, i as OpenAutoPrResult, a as RunImprovementLoopResult, R as RunOptimizationOptions, j as RunOptimizationResult, k as countSentenceEdits, l as defaultRenderDiff, m as extractH2Sections, g as gepaProposer, o as openAutoPr, p as planCampaignRun, r as runCampaign, d as runImprovementLoop, n as runOptimization } from '../gepa-DolL_Fko.js';
4
+ import { S as Scenario, M as MutableSurface, D as DispatchContext, b as JudgeConfig, g as Gate, e as GenerationRecord, J as JudgeScore, L as LabeledScenarioStore, s as LabeledScenarioWrite, t as LabeledScenarioSampleArgs, u as LabeledScenarioRecord, v as LabelTrust, f as SurfaceProposer, w as ProposedCandidate, x as ProposeContext, m as CodeSurface, y as LabeledScenarioSource, C as CampaignResult } from '../types-C5gJrOVT.js';
5
+ export { i as CampaignAggregates, j as CampaignArtifactWriter, k as CampaignCellResult, l as CampaignCostMeter, z as CampaignTokenUsage, d as CampaignTraceWriter, c as DispatchFn, n as GateContext, h as GateDecision, G as GateResult, o as GenerationCandidate, A as JudgeAggregate, a as JudgeDimension, p as Mutator, O as OptimizationProposer, q as OptimizerConfig, P as ParetoParent, R as RedactionStatus, B as ScenarioAggregate, r as SessionScript, T as TraceSpan, E as isProposedCandidate, F as labelTrustRank } from '../types-C5gJrOVT.js';
6
+ import { C as CampaignRunPlan, P as PlanCampaignRunOptions, b as RunCampaignOptions, c as RunImprovementLoopOptions } from '../gepa-dne9JDPL.js';
7
+ export { f as CampaignRunPlanCell, h as GepaProposerConstraints, G as GepaProposerOptions, O as OpenAutoPrOptions, i as OpenAutoPrResult, a as RunImprovementLoopResult, R as RunOptimizationOptions, j as RunOptimizationResult, k as countSentenceEdits, l as defaultRenderDiff, m as extractH2Sections, g as gepaProposer, o as openAutoPr, p as planCampaignRun, r as runCampaign, d as runImprovementLoop, n as runOptimization } from '../gepa-dne9JDPL.js';
8
8
  import { a as PairedBootstrapResult, E as EProcessState } from '../statistics-oUbOJe-S.js';
9
9
  import { C as CampaignStorage } from '../storage-Dw_f7WMt.js';
10
10
  export { f as fsCampaignStorage, i as inMemoryCampaignStorage } from '../storage-Dw_f7WMt.js';
11
- export { A as AxisEvidence, a as AxisVerdict, B as BuildEvidenceVectorOptions, l as BuildLoopProvenanceArgs, D as DefaultProductionGateOptions, m as EmitLoopProvenanceArgs, n as EmitLoopProvenanceResult, E as EvidenceVector, b as EvolutionaryProposerOptions, H as HeldOutGateOptions, o as LoopProvenanceBackend, q as LoopProvenanceCandidate, L as LoopProvenanceRecord, O as ObjectiveSource, c as ParetoSignificanceGateOptions, P as PowerPreflight, s as PowerPreflightOptions, d as PromotionObjective, e as PromotionPolicy, R as RunEvalOptions, f as buildEvidenceVector, t as buildLoopProvenanceRecord, g as composeGate, h as defaultProductionGate, u as emitLoopProvenance, i as evolutionaryProposer, j as heldOutGate, v as loopProvenanceSpans, p as paretoPolicy, k as paretoSignificanceGate, w as powerPreflight, x as provenanceRecordPath, y as provenanceSpansPath, r as runEval } from '../provenance-BZmpWmn4.js';
11
+ export { A as AxisEvidence, a as AxisVerdict, B as BuildEvidenceVectorOptions, l as BuildLoopProvenanceArgs, D as DefaultProductionGateOptions, m as EmitLoopProvenanceArgs, n as EmitLoopProvenanceResult, E as EvidenceVector, b as EvolutionaryProposerOptions, H as HeldOutGateOptions, o as LoopProvenanceBackend, q as LoopProvenanceCandidate, L as LoopProvenanceRecord, O as ObjectiveSource, c as ParetoSignificanceGateOptions, P as PowerPreflight, s as PowerPreflightOptions, d as PromotionObjective, e as PromotionPolicy, R as RunEvalOptions, f as buildEvidenceVector, t as buildLoopProvenanceRecord, g as composeGate, h as defaultProductionGate, u as emitLoopProvenance, i as evolutionaryProposer, j as heldOutGate, v as loopProvenanceSpans, p as paretoPolicy, k as paretoSignificanceGate, w as powerPreflight, x as provenanceRecordPath, y as provenanceSpansPath, r as runEval } from '../provenance-Bibyg1U9.js';
12
12
  import { L as LlmClientOptions } from '../llm-client-DyqEH4jH.js';
13
13
  import { AgentProfile } from '@tangle-network/agent-interface';
14
14
  import { A as AgentEvalError, V as ValidationError } from '../errors-oeQrLqXC.js';
15
- import { b as RunSplitTag, R as RunRecord } from '../run-record-DksGsfgv.js';
16
- import { b as PolicyEdit, F as FindingToPolicyEditOptions, d as PolicyEditAdmissionOptions, c as PolicyEditAdmission } from '../policy-edit-az2qRmvN.js';
17
- import { T as TraceAnalystKindSpec, A as AnalystFinding } from '../kind-factory-20hcaYpf.js';
15
+ import { b as RunSplitTag, R as RunRecord } from '../run-record-B7RTi_ix.js';
16
+ import { b as PolicyEdit, F as FindingToPolicyEditOptions, d as PolicyEditAdmissionOptions, c as PolicyEditAdmission } from '../policy-edit-RLn8GWof.js';
17
+ import { T as TraceAnalystKindSpec, A as AnalystFinding } from '../kind-factory-DcNg13sZ.js';
18
18
  import '@tangle-network/tcloud';
19
19
  import '../raw-provider-sink-C46HDghv.js';
20
20
  import '../verdict-C9MlYujm.js';
@@ -26,8 +26,8 @@ import '../schema-SGWcK9wa.js';
26
26
  import '../judge-calibration-7C-IDmKr.js';
27
27
  import '../types-C7DGg5ex.js';
28
28
  import '../hosted/index.js';
29
- import '../insight-report-D4cXFsLt.js';
30
- import '../summary-report-Bz-0-t8v.js';
29
+ import '../insight-report-IwwvqZZv.js';
30
+ import '../summary-report-BJ5aNwZ1.js';
31
31
  import '../failure-cluster-C48PiReX.js';
32
32
  import 'zod';
33
33
 
@@ -65,7 +65,7 @@ import {
65
65
  userStoryScoreboard,
66
66
  validateSearchLedgerEvent,
67
67
  verifyCodeSurface
68
- } from "../chunk-KDCMEZDI.js";
68
+ } from "../chunk-KG4TD7EQ.js";
69
69
  import {
70
70
  assertCodeSurfaceIdentity,
71
71
  buildEvidenceVector,
@@ -100,7 +100,7 @@ import {
100
100
  runOptimization,
101
101
  surfaceContentHash,
102
102
  surfaceHash
103
- } from "../chunk-N6MTC3GK.js";
103
+ } from "../chunk-ADYLPOSX.js";
104
104
  import "../chunk-VI2UW6B6.js";
105
105
  import {
106
106
  fsCampaignStorage,
@@ -110,17 +110,17 @@ import {
110
110
  runCampaign,
111
111
  tangleTracesRoot
112
112
  } from "../chunk-FAOEFFRT.js";
113
- import "../chunk-AN5UYSVD.js";
114
- import "../chunk-DWLIGZBX.js";
113
+ import "../chunk-QMXXSNC4.js";
114
+ import "../chunk-5S5NJ63F.js";
115
115
  import "../chunk-ARU2PZFM.js";
116
116
  import "../chunk-PJQFMIOX.js";
117
117
  import "../chunk-RPDDVKI7.js";
118
118
  import "../chunk-GGE4NNQT.js";
119
119
  import "../chunk-LNQEP766.js";
120
- import "../chunk-VK6HBGAE.js";
120
+ import "../chunk-5UF54T55.js";
121
121
  import "../chunk-XJYR7XFV.js";
122
122
  import "../chunk-VSMTAMNK.js";
123
- import "../chunk-FUCQVFMU.js";
123
+ import "../chunk-GY4SYVPJ.js";
124
124
  import "../chunk-PC4UYEBM.js";
125
125
  import "../chunk-ONWEPEDO.js";
126
126
  import "../chunk-PZ5AY32C.js";
@@ -1,6 +1,6 @@
1
1
  import {
2
2
  callLlmJson
3
- } from "./chunk-FUCQVFMU.js";
3
+ } from "./chunk-GY4SYVPJ.js";
4
4
 
5
5
  // src/wire/schemas.ts
6
6
  import { extendZodWithOpenApi } from "@asteasolutions/zod-to-openapi";
@@ -1002,4 +1002,4 @@ export {
1002
1002
  startServer,
1003
1003
  startServerAsync
1004
1004
  };
1005
- //# sourceMappingURL=chunk-DRPIZQIT.js.map
1005
+ //# sourceMappingURL=chunk-4D5RVB3W.js.map
@@ -4,7 +4,7 @@ import {
4
4
  } from "./chunk-LNQEP766.js";
5
5
  import {
6
6
  callLlm
7
- } from "./chunk-FUCQVFMU.js";
7
+ } from "./chunk-GY4SYVPJ.js";
8
8
 
9
9
  // src/analyst/types.ts
10
10
  import { createHash } from "crypto";
@@ -1148,4 +1148,4 @@ export {
1148
1148
  DEFAULT_TRACE_ANALYST_KINDS,
1149
1149
  AnalystRegistry
1150
1150
  };
1151
- //# sourceMappingURL=chunk-DWLIGZBX.js.map
1151
+ //# sourceMappingURL=chunk-5S5NJ63F.js.map
@@ -50,6 +50,9 @@ function validateRunRecord(input) {
50
50
  expectFiniteNumber(obj.wallMs, "wallMs");
51
51
  if (obj.queueMs !== void 0) expectFiniteNumber(obj.queueMs, "queueMs");
52
52
  expectFiniteNumber(obj.costUsd, "costUsd");
53
+ if (obj.costProvenance !== void 0) {
54
+ validateCostProvenance(obj.costProvenance, obj.costUsd);
55
+ }
53
56
  if (!modelHasSnapshot(obj.model)) {
54
57
  throw new RunRecordValidationError(
55
58
  `model "${obj.model}" lacks a snapshot version (use 'name@YYYY-MM-DD' or 'name-YYYYMMDD')`,
@@ -151,6 +154,54 @@ function validateRunRecord(input) {
151
154
  }
152
155
  return input;
153
156
  }
157
+ function resolveRunCostProvenance(run) {
158
+ if (run.costProvenance) return run.costProvenance;
159
+ if (run.outcome.raw.cost_estimated === 1) {
160
+ return { kind: "estimated", usd: run.costUsd };
161
+ }
162
+ if (run.costUsd > 0) return { kind: "observed", usd: run.costUsd };
163
+ return { kind: "uncaptured", usd: null };
164
+ }
165
+ function validateCostProvenance(input, costUsd) {
166
+ if (input === null || typeof input !== "object") {
167
+ throw new RunRecordValidationError("costProvenance must be an object", "costProvenance");
168
+ }
169
+ const value = input;
170
+ if (value.kind !== "observed" && value.kind !== "estimated" && value.kind !== "uncaptured") {
171
+ throw new RunRecordValidationError(
172
+ "costProvenance.kind must be observed, estimated, or uncaptured",
173
+ "costProvenance.kind"
174
+ );
175
+ }
176
+ if (value.kind === "uncaptured") {
177
+ if (value.usd !== null) {
178
+ throw new RunRecordValidationError(
179
+ "uncaptured costProvenance.usd must be null",
180
+ "costProvenance.usd"
181
+ );
182
+ }
183
+ if (costUsd !== 0) {
184
+ throw new RunRecordValidationError(
185
+ "uncaptured costProvenance requires the compatibility costUsd sentinel 0",
186
+ "costUsd"
187
+ );
188
+ }
189
+ return;
190
+ }
191
+ expectFiniteNumber(value.usd, "costProvenance.usd");
192
+ if (value.usd < 0) {
193
+ throw new RunRecordValidationError(
194
+ "costProvenance.usd must be non-negative",
195
+ "costProvenance.usd"
196
+ );
197
+ }
198
+ if (value.usd !== costUsd) {
199
+ throw new RunRecordValidationError(
200
+ "costProvenance.usd must equal costUsd",
201
+ "costProvenance.usd"
202
+ );
203
+ }
204
+ }
154
205
  function isRunRecord(input) {
155
206
  try {
156
207
  validateRunRecord(input);
@@ -241,9 +292,10 @@ function modelHasSnapshot(model) {
241
292
  export {
242
293
  RunRecordValidationError,
243
294
  validateRunRecord,
295
+ resolveRunCostProvenance,
244
296
  isRunRecord,
245
297
  parseRunRecordSafe,
246
298
  roundTripRunRecord,
247
299
  modelHasSnapshot
248
300
  };
249
- //# sourceMappingURL=chunk-VK6HBGAE.js.map
301
+ //# sourceMappingURL=chunk-5UF54T55.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"sources":["../src/run-record.ts"],"sourcesContent":["/**\n * Paper-grade RunRecord schema + runtime validator.\n *\n * Every run that participates in a promotion gate, paper table, or\n * researcher loop SHOULD be recorded as a `RunRecord`. The mandatory\n * fields are exactly those the paper \"Two Loops, Three Roles\" requires\n * for reproducibility: who/what/when/cost/seed/hash, plus the search vs\n * holdout split tag and either a `searchScore` or a `holdoutScore`.\n *\n * This is intentionally NOT a replacement for the rich `Run` /\n * `ProposeReviewReport` / `ScenarioResult` types already in the\n * package. Those are runtime structures with full provenance. A\n * `RunRecord` is the analysis-time projection — the JSON-friendly\n * row you'd put in a parquet file or paste into a notebook.\n *\n * Validate at the boundary:\n *\n * const rec = validateRunRecord(rawJson) // throws on missing\n * const ok = isRunRecord(rawJson) // boolean check\n * const rec = parseRunRecordSafe(rawJson) // { ok, value | error }\n *\n * The validator runs in pure TS — zod is intentionally NOT a\n * dependency. Round-trip tested in `tests/run-record.test.ts`.\n */\n\nimport type { AgentProfileCell } from './agent-profile-cell'\nimport { validateAgentProfileCell } from './agent-profile-cell'\nimport { ValidationError } from './errors'\nimport type { FailureClass } from './trace/schema'\n\n/** Search/dev/holdout split tag. 'search' is the paper-grade alias for the\n * combined train+test pool that the optimizer is allowed to read. */\nexport type RunSplitTag = 'search' | 'dev' | 'holdout'\n\nexport interface RunTokenUsage {\n input: number\n output: number\n cached?: number\n}\n\n/**\n * How a run's USD amount was obtained.\n *\n * `costUsd` remains mandatory for wire compatibility. New producers should\n * always populate this discriminated union so a missing bill is never\n * mistaken for an observed zero-dollar run. For `uncaptured`, `costUsd` uses\n * the legacy `0` sentinel while this field carries the truthful null.\n */\nexport type RunCostProvenance =\n | { kind: 'observed'; usd: number }\n | { kind: 'estimated'; usd: number }\n | { kind: 'uncaptured'; usd: null }\n\nexport interface RunJudgeMetadata {\n model: string\n promptVersion: string\n /** [0,1] confidence the judge declared. Constant judge confidence\n * across many runs is a fallback signal (see `canary.ts`). */\n confidence: number\n /** True if the judge degraded to a fallback path (rules-only,\n * prior-call cache, etc.). The canary uses this to alert. */\n fallback: boolean\n}\n\n/**\n * Per-judge / per-dimension breakdown for runs scored by an ensemble of\n * judges over a multi-dimensional rubric.\n *\n * The collapsed `outcome.searchScore` / `holdoutScore` carries the\n * composite the gate uses. The full breakdown belongs here so consumers\n * can answer \"which judge disagreed?\", \"which dimension dragged the\n * composite down?\", and \"did half the panel fail?\" without re-running.\n *\n * `perJudge[judgeId][dim]` is the canonical source; `perDimMean` and\n * `composite` are convenience projections — derivable but precomputed so\n * downstream IRR primitives (`interRaterReliability`,\n * `corpusInterRaterAgreement`) and reporters don't pay the same\n * aggregation twice.\n *\n * Fail-loud discipline: judges that errored out land in `failedJudges`\n * by id. A missing key in `perJudge` is ambiguous (silent zero vs not\n * run); the explicit list makes a partial-failure recorded as such.\n */\nexport interface JudgeScoresRecord {\n /** Per-judge per-dimension scores. `{ \"kimi-k2.6\": { helpfulness: 0.8, clarity: 0.7 }, ... }`. */\n perJudge: Record<string, Record<string, number>>\n /** Per-dim mean across judges. Convenience — derivable from `perJudge`. */\n perDimMean: Record<string, number>\n /** Composite mean across all dims and judges. Mirrors the score\n * the gate sees on `outcome.searchScore` / `holdoutScore`. */\n composite: number\n /** Judges that errored or returned an unparseable verdict. Recorded\n * by id (e.g. `['glm-5.1']`) so a partial-failure case is explicit,\n * not inferred from missing keys in `perJudge`. */\n failedJudges?: string[]\n /** Free-form notes the judges emitted (joined across judges or\n * first-judge only — consumer's choice). */\n notes?: string\n}\n\nexport interface RunOutcome {\n /** Score on the search/optimization split. Optional because a\n * holdout-only evaluation only fills `holdoutScore`. */\n searchScore?: number\n /** Score on the held-out split. Optional because a search-only run\n * only fills `searchScore`. At least one must be present. */\n holdoutScore?: number\n /** Bag of any other metric the run produced — judge dimensions,\n * pass/fail counters, latency stats, etc. Numeric only — keeps\n * reporters honest. */\n raw: Record<string, number>\n /** Per-judge / per-dim breakdown. Consumers writing ensemble\n * judgements populate this; substrate primitives like\n * `interRaterReliability` and `corpusInterRaterAgreement` accept\n * these records as input. Optional — single-judge or scalar-only\n * runs leave it unset. */\n judgeScores?: JudgeScoresRecord\n /** Authenticity / realness verdict — did the run build the REAL thing on the\n * intended infra, or fake it (see `./authenticity`)? Optional: only domains\n * with an authenticity config populate it. Carried in the corpus so the\n * flywheel / off-policy learning can optimize for real completion, not gamed\n * pass-rate. `score` is 0-1; `gated` is the anti-Goodhart flag — a gated run\n * must not count as a real success regardless of `score`. */\n realness?: { score: number; gated: boolean; reason?: string }\n}\n\n/**\n * Mandatory paper-grade fields for a single evaluation run. Optional\n * fields are extension points; mandatory fields throw if missing.\n *\n * Hash discipline:\n * - `promptHash` is the sha256 of the EFFECTIVE prompt sent to the\n * model (after any steering bundle merge).\n * - `configHash` is the sha256 of the effective run config (model,\n * temperature, tools, judges, splits). The pair (promptHash,\n * configHash) uniquely identifies an experiment cell.\n *\n * Model snapshot discipline:\n * - `model` MUST encode a snapshot version. Bare aliases like\n * `claude-sonnet-4` or `gpt-4o` are banned — they remap silently.\n * Use `claude-sonnet-4-6@2025-04-15` or `gpt-4o-2024-11-20`.\n */\nexport interface RunRecord {\n /** UUID for the run. */\n runId: string\n /** Logical experiment grouping (a treatment vs a baseline within\n * the same sweep should share `experimentId`). */\n experimentId: string\n /** Stable identifier for the candidate (variant) being run. The\n * promotion gate compares two `candidateId`s on matched items. */\n candidateId: string\n /** RNG seed for the run. Always recorded — silent re-seeding is\n * the most common cause of non-reproducible numbers. */\n seed: number\n /** Model identifier WITH snapshot version. */\n model: string\n /** sha256 of the effective prompt (post-steering). */\n promptHash: string\n /** sha256 of the effective config. */\n configHash: string\n /** Git SHA the harness was run from. */\n commitSha: string\n /** End-to-end wall-clock duration in milliseconds. */\n wallMs: number\n /** Time spent queued before execution started, if known. */\n queueMs?: number\n /** Total USD cost. Mandatory — runs without a cost number are\n * unbounded by definition and must not be admitted into the gate.\n * `0` is retained as the compatibility sentinel for an uncaptured amount;\n * inspect `costProvenance` before treating it as observed. */\n costUsd: number\n /** Observed, model-priced estimate, or genuinely uncaptured USD amount.\n * Optional only so existing serialized RunRecords remain valid. */\n costProvenance?: RunCostProvenance\n /** Token usage breakdown. */\n tokenUsage: RunTokenUsage\n /** Judge-side metadata, if a judge was used. */\n judgeMetadata?: RunJudgeMetadata\n /** Per-split scores + raw bag. */\n outcome: RunOutcome\n /** Canonical, cross-agent failure class drawn from the shared\n * `FAILURE_CLASSES` taxonomy. This is the aggregation key that makes\n * \"which failure dominates across the whole fleet\" answerable in ONE\n * vocabulary — every agent classifies against the same enum. Producers\n * set it via the substrate classifier; leave unset only when the failure\n * genuinely can't be classified. */\n failureClass?: FailureClass\n /** Free-form domain-specific failure detail, scoped UNDER `failureClass`\n * (e.g. failureClass='tool_recovery_failure', failureMode='forge_build_unsatisfied').\n * The within-agent drill-down; `failureClass` is the cross-agent key. */\n failureMode?: string\n /** Which split this run was drawn from. */\n splitTag: RunSplitTag\n /**\n * Stable scenario identifier the run was scored against. Optional for\n * backwards compatibility, but **strongly recommended**: every primitive\n * that pairs runs by scenario (preferences, paired stats, BT tournament)\n * keys on this. The campaign artifact populates it canonically; legacy\n * runs without it fall back to inference from `outcome.raw.scenario_id`\n * or `experimentId`.\n */\n scenarioId?: string\n /**\n * Canonical identity for the agent profile cell that produced this row:\n * profile artifact hash plus optional harness/model/prompt/reporting\n * dimensions. Use `agentProfile.cellId` to group persona sweeps and\n * longitudinal reports by the complete source profile, not by a loose\n * candidate label or opaque config hash.\n */\n agentProfile?: AgentProfileCell\n}\n\n// ── Validation ───────────────────────────────────────────────────────\n\nconst MANDATORY_TOP_LEVEL = [\n 'runId',\n 'experimentId',\n 'candidateId',\n 'seed',\n 'model',\n 'promptHash',\n 'configHash',\n 'commitSha',\n 'wallMs',\n 'costUsd',\n 'tokenUsage',\n 'outcome',\n 'splitTag',\n] as const\n\nconst SPLIT_TAGS: ReadonlyArray<RunSplitTag> = ['search', 'dev', 'holdout']\n\nexport class RunRecordValidationError extends ValidationError {\n readonly path: string\n constructor(message: string, path = '') {\n super(path ? `${message} (at ${path})` : message)\n this.path = path\n }\n}\n\n/**\n * Strict validator. Throws `RunRecordValidationError` on the first\n * missing or wrongly-typed field. Returns the input cast to\n * `RunRecord` on success — the validator does not coerce.\n */\nexport function validateRunRecord(input: unknown): RunRecord {\n if (input === null || typeof input !== 'object') {\n throw new RunRecordValidationError('expected object')\n }\n const obj = input as Record<string, unknown>\n\n for (const key of MANDATORY_TOP_LEVEL) {\n if (!(key in obj)) {\n throw new RunRecordValidationError(`missing mandatory field \"${key}\"`)\n }\n }\n\n expectString(obj.runId, 'runId')\n expectString(obj.experimentId, 'experimentId')\n expectString(obj.candidateId, 'candidateId')\n expectFiniteNumber(obj.seed, 'seed')\n expectString(obj.model, 'model')\n expectString(obj.promptHash, 'promptHash')\n expectString(obj.configHash, 'configHash')\n expectString(obj.commitSha, 'commitSha')\n expectFiniteNumber(obj.wallMs, 'wallMs')\n if (obj.queueMs !== undefined) expectFiniteNumber(obj.queueMs, 'queueMs')\n expectFiniteNumber(obj.costUsd, 'costUsd')\n if (obj.costProvenance !== undefined) {\n validateCostProvenance(obj.costProvenance, obj.costUsd as number)\n }\n\n // Snapshot discipline: bare model aliases are not paper-grade.\n if (!modelHasSnapshot(obj.model as string)) {\n throw new RunRecordValidationError(\n `model \"${obj.model}\" lacks a snapshot version (use 'name@YYYY-MM-DD' or 'name-YYYYMMDD')`,\n 'model',\n )\n }\n\n // Token usage.\n const tu = obj.tokenUsage\n if (tu === null || typeof tu !== 'object') {\n throw new RunRecordValidationError('tokenUsage must be an object', 'tokenUsage')\n }\n const tuRec = tu as Record<string, unknown>\n expectFiniteNumber(tuRec.input, 'tokenUsage.input')\n expectFiniteNumber(tuRec.output, 'tokenUsage.output')\n if (tuRec.cached !== undefined) expectFiniteNumber(tuRec.cached, 'tokenUsage.cached')\n\n // Judge metadata, optional.\n if (obj.judgeMetadata !== undefined) {\n const jm = obj.judgeMetadata\n if (jm === null || typeof jm !== 'object') {\n throw new RunRecordValidationError('judgeMetadata must be an object', 'judgeMetadata')\n }\n const jmRec = jm as Record<string, unknown>\n expectString(jmRec.model, 'judgeMetadata.model')\n expectString(jmRec.promptVersion, 'judgeMetadata.promptVersion')\n expectFiniteNumber(jmRec.confidence, 'judgeMetadata.confidence')\n if (typeof jmRec.fallback !== 'boolean') {\n throw new RunRecordValidationError(\n 'judgeMetadata.fallback must be boolean',\n 'judgeMetadata.fallback',\n )\n }\n }\n\n // Outcome.\n const out = obj.outcome\n if (out === null || typeof out !== 'object') {\n throw new RunRecordValidationError('outcome must be an object', 'outcome')\n }\n const outRec = out as Record<string, unknown>\n if (outRec.searchScore !== undefined)\n expectFiniteNumber(outRec.searchScore, 'outcome.searchScore')\n if (outRec.holdoutScore !== undefined)\n expectFiniteNumber(outRec.holdoutScore, 'outcome.holdoutScore')\n if (outRec.searchScore === undefined && outRec.holdoutScore === undefined) {\n throw new RunRecordValidationError(\n 'outcome must define searchScore or holdoutScore (or both)',\n 'outcome',\n )\n }\n const raw = outRec.raw\n if (raw === null || typeof raw !== 'object') {\n throw new RunRecordValidationError('outcome.raw must be an object', 'outcome.raw')\n }\n for (const [k, v] of Object.entries(raw as Record<string, unknown>)) {\n expectFiniteNumber(v, `outcome.raw.${k}`)\n }\n // Realness verdict, optional.\n if (outRec.realness !== undefined) {\n const r = outRec.realness\n if (r === null || typeof r !== 'object') {\n throw new RunRecordValidationError('outcome.realness must be an object', 'outcome.realness')\n }\n const rr = r as Record<string, unknown>\n expectFiniteNumber(rr.score, 'outcome.realness.score')\n if (typeof rr.gated !== 'boolean') {\n throw new RunRecordValidationError(\n 'outcome.realness.gated must be a boolean',\n 'outcome.realness.gated',\n )\n }\n }\n\n // Per-judge / per-dim breakdown, optional.\n if (outRec.judgeScores !== undefined) {\n validateJudgeScores(outRec.judgeScores, 'outcome.judgeScores')\n }\n\n // Failure mode optional.\n if (obj.failureMode !== undefined) expectString(obj.failureMode, 'failureMode')\n\n if (obj.agentProfile !== undefined) {\n try {\n const profile = validateAgentProfileCell(obj.agentProfile)\n if (profile.model !== undefined && profile.model !== obj.model) {\n throw new RunRecordValidationError(\n `agentProfile.model \"${profile.model}\" does not match model \"${obj.model}\"`,\n 'agentProfile.model',\n )\n }\n if (profile.promptHash !== undefined && profile.promptHash !== obj.promptHash) {\n throw new RunRecordValidationError(\n `agentProfile.promptHash \"${profile.promptHash}\" does not match promptHash \"${obj.promptHash}\"`,\n 'agentProfile.promptHash',\n )\n }\n } catch (error) {\n if (error instanceof RunRecordValidationError) throw error\n if (error instanceof Error) {\n throw new RunRecordValidationError(error.message, 'agentProfile')\n }\n throw error\n }\n }\n\n // Split tag.\n if (typeof obj.splitTag !== 'string' || !SPLIT_TAGS.includes(obj.splitTag as RunSplitTag)) {\n throw new RunRecordValidationError(\n `splitTag must be one of ${SPLIT_TAGS.join(', ')}, got ${String(obj.splitTag)}`,\n 'splitTag',\n )\n }\n\n return input as RunRecord\n}\n\n/**\n * Resolve provenance for both new and legacy records.\n *\n * Legacy producers sometimes set `outcome.raw.cost_estimated = 1`. A positive\n * unlabeled amount is treated as observed, matching the historical contract.\n * Zero without an explicit label is conservatively uncaptured: claiming an\n * observed $0 would be stronger than the serialized evidence supports.\n */\nexport function resolveRunCostProvenance(\n run: Pick<RunRecord, 'costUsd' | 'costProvenance' | 'outcome'>,\n): RunCostProvenance {\n if (run.costProvenance) return run.costProvenance\n if (run.outcome.raw.cost_estimated === 1) {\n return { kind: 'estimated', usd: run.costUsd }\n }\n if (run.costUsd > 0) return { kind: 'observed', usd: run.costUsd }\n return { kind: 'uncaptured', usd: null }\n}\n\nfunction validateCostProvenance(input: unknown, costUsd: number): void {\n if (input === null || typeof input !== 'object') {\n throw new RunRecordValidationError('costProvenance must be an object', 'costProvenance')\n }\n const value = input as Record<string, unknown>\n if (value.kind !== 'observed' && value.kind !== 'estimated' && value.kind !== 'uncaptured') {\n throw new RunRecordValidationError(\n 'costProvenance.kind must be observed, estimated, or uncaptured',\n 'costProvenance.kind',\n )\n }\n if (value.kind === 'uncaptured') {\n if (value.usd !== null) {\n throw new RunRecordValidationError(\n 'uncaptured costProvenance.usd must be null',\n 'costProvenance.usd',\n )\n }\n if (costUsd !== 0) {\n throw new RunRecordValidationError(\n 'uncaptured costProvenance requires the compatibility costUsd sentinel 0',\n 'costUsd',\n )\n }\n return\n }\n expectFiniteNumber(value.usd, 'costProvenance.usd')\n if ((value.usd as number) < 0) {\n throw new RunRecordValidationError(\n 'costProvenance.usd must be non-negative',\n 'costProvenance.usd',\n )\n }\n if (value.usd !== costUsd) {\n throw new RunRecordValidationError(\n 'costProvenance.usd must equal costUsd',\n 'costProvenance.usd',\n )\n }\n}\n\n/** Boolean validator — convenience for filtering arrays. */\nexport function isRunRecord(input: unknown): input is RunRecord {\n try {\n validateRunRecord(input)\n return true\n } catch {\n return false\n }\n}\n\n/** Non-throwing validator — returns a discriminated union. */\nexport function parseRunRecordSafe(\n input: unknown,\n): { ok: true; value: RunRecord } | { ok: false; error: RunRecordValidationError } {\n try {\n return { ok: true, value: validateRunRecord(input) }\n } catch (e) {\n if (e instanceof RunRecordValidationError) return { ok: false, error: e }\n throw e\n }\n}\n\n/** Round-trip helper — `JSON.parse(JSON.stringify(record))` then validate. */\nexport function roundTripRunRecord(record: RunRecord): RunRecord {\n const json = JSON.stringify(record)\n return validateRunRecord(JSON.parse(json))\n}\n\n// ── Internals ────────────────────────────────────────────────────────\n\nfunction expectString(value: unknown, path: string): void {\n if (typeof value !== 'string' || value.length === 0) {\n throw new RunRecordValidationError(`expected non-empty string`, path)\n }\n}\n\nfunction expectFiniteNumber(value: unknown, path: string): void {\n if (typeof value !== 'number' || !Number.isFinite(value)) {\n throw new RunRecordValidationError(`expected finite number`, path)\n }\n}\n\nfunction validateJudgeScores(value: unknown, path: string): void {\n if (value === null || typeof value !== 'object') {\n throw new RunRecordValidationError('judgeScores must be an object', path)\n }\n const rec = value as Record<string, unknown>\n\n const perJudge = rec.perJudge\n if (perJudge === null || typeof perJudge !== 'object') {\n throw new RunRecordValidationError('perJudge must be an object', `${path}.perJudge`)\n }\n for (const [judgeId, dims] of Object.entries(perJudge as Record<string, unknown>)) {\n if (dims === null || typeof dims !== 'object') {\n throw new RunRecordValidationError(\n 'per-judge entry must be an object of dimension scores',\n `${path}.perJudge.${judgeId}`,\n )\n }\n for (const [dim, score] of Object.entries(dims as Record<string, unknown>)) {\n expectFiniteNumber(score, `${path}.perJudge.${judgeId}.${dim}`)\n }\n }\n\n const perDimMean = rec.perDimMean\n if (perDimMean === null || typeof perDimMean !== 'object') {\n throw new RunRecordValidationError('perDimMean must be an object', `${path}.perDimMean`)\n }\n for (const [dim, mean] of Object.entries(perDimMean as Record<string, unknown>)) {\n expectFiniteNumber(mean, `${path}.perDimMean.${dim}`)\n }\n\n expectFiniteNumber(rec.composite, `${path}.composite`)\n\n if (rec.failedJudges !== undefined) {\n if (!Array.isArray(rec.failedJudges)) {\n throw new RunRecordValidationError(\n 'failedJudges must be an array of strings',\n `${path}.failedJudges`,\n )\n }\n for (let i = 0; i < rec.failedJudges.length; i++) {\n const id = rec.failedJudges[i]\n if (typeof id !== 'string' || id.length === 0) {\n throw new RunRecordValidationError(\n 'failedJudges entry must be a non-empty string',\n `${path}.failedJudges[${i}]`,\n )\n }\n }\n }\n\n if (rec.notes !== undefined && typeof rec.notes !== 'string') {\n throw new RunRecordValidationError('notes must be a string', `${path}.notes`)\n }\n}\n\n/**\n * Heuristic snapshot check. Accepts:\n * - `name@YYYY-MM-DD` (Anthropic style: `claude-sonnet-4-6@2025-04-15`)\n * - `name-YYYYMMDD` (OpenAI style: `gpt-4o-2024-11-20`)\n * - `name@<arbitrary-token>` (allow opaque snapshots like `@v3`)\n * - explicit `:date-...` Vertex-style tags\n *\n * Rejects bare aliases like `claude-sonnet-4` or `gpt-4o` that remap\n * silently as providers ship new snapshots.\n */\nexport function modelHasSnapshot(model: string): boolean {\n if (model.includes('@')) return true\n if (/-\\d{8}$/.test(model)) return true\n if (/-\\d{4}-\\d{2}-\\d{2}$/.test(model)) return true\n if (/:date-/.test(model)) return true\n return false\n}\n"],"mappings":";;;;;;;;AAsNA,IAAM,sBAAsB;AAAA,EAC1B;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AACF;AAEA,IAAM,aAAyC,CAAC,UAAU,OAAO,SAAS;AAEnE,IAAM,2BAAN,cAAuC,gBAAgB;AAAA,EACnD;AAAA,EACT,YAAY,SAAiB,OAAO,IAAI;AACtC,UAAM,OAAO,GAAG,OAAO,QAAQ,IAAI,MAAM,OAAO;AAChD,SAAK,OAAO;AAAA,EACd;AACF;AAOO,SAAS,kBAAkB,OAA2B;AAC3D,MAAI,UAAU,QAAQ,OAAO,UAAU,UAAU;AAC/C,UAAM,IAAI,yBAAyB,iBAAiB;AAAA,EACtD;AACA,QAAM,MAAM;AAEZ,aAAW,OAAO,qBAAqB;AACrC,QAAI,EAAE,OAAO,MAAM;AACjB,YAAM,IAAI,yBAAyB,4BAA4B,GAAG,GAAG;AAAA,IACvE;AAAA,EACF;AAEA,eAAa,IAAI,OAAO,OAAO;AAC/B,eAAa,IAAI,cAAc,cAAc;AAC7C,eAAa,IAAI,aAAa,aAAa;AAC3C,qBAAmB,IAAI,MAAM,MAAM;AACnC,eAAa,IAAI,OAAO,OAAO;AAC/B,eAAa,IAAI,YAAY,YAAY;AACzC,eAAa,IAAI,YAAY,YAAY;AACzC,eAAa,IAAI,WAAW,WAAW;AACvC,qBAAmB,IAAI,QAAQ,QAAQ;AACvC,MAAI,IAAI,YAAY,OAAW,oBAAmB,IAAI,SAAS,SAAS;AACxE,qBAAmB,IAAI,SAAS,SAAS;AACzC,MAAI,IAAI,mBAAmB,QAAW;AACpC,2BAAuB,IAAI,gBAAgB,IAAI,OAAiB;AAAA,EAClE;AAGA,MAAI,CAAC,iBAAiB,IAAI,KAAe,GAAG;AAC1C,UAAM,IAAI;AAAA,MACR,UAAU,IAAI,KAAK;AAAA,MACnB;AAAA,IACF;AAAA,EACF;AAGA,QAAM,KAAK,IAAI;AACf,MAAI,OAAO,QAAQ,OAAO,OAAO,UAAU;AACzC,UAAM,IAAI,yBAAyB,gCAAgC,YAAY;AAAA,EACjF;AACA,QAAM,QAAQ;AACd,qBAAmB,MAAM,OAAO,kBAAkB;AAClD,qBAAmB,MAAM,QAAQ,mBAAmB;AACpD,MAAI,MAAM,WAAW,OAAW,oBAAmB,MAAM,QAAQ,mBAAmB;AAGpF,MAAI,IAAI,kBAAkB,QAAW;AACnC,UAAM,KAAK,IAAI;AACf,QAAI,OAAO,QAAQ,OAAO,OAAO,UAAU;AACzC,YAAM,IAAI,yBAAyB,mCAAmC,eAAe;AAAA,IACvF;AACA,UAAM,QAAQ;AACd,iBAAa,MAAM,OAAO,qBAAqB;AAC/C,iBAAa,MAAM,eAAe,6BAA6B;AAC/D,uBAAmB,MAAM,YAAY,0BAA0B;AAC/D,QAAI,OAAO,MAAM,aAAa,WAAW;AACvC,YAAM,IAAI;AAAA,QACR;AAAA,QACA;AAAA,MACF;AAAA,IACF;AAAA,EACF;AAGA,QAAM,MAAM,IAAI;AAChB,MAAI,QAAQ,QAAQ,OAAO,QAAQ,UAAU;AAC3C,UAAM,IAAI,yBAAyB,6BAA6B,SAAS;AAAA,EAC3E;AACA,QAAM,SAAS;AACf,MAAI,OAAO,gBAAgB;AACzB,uBAAmB,OAAO,aAAa,qBAAqB;AAC9D,MAAI,OAAO,iBAAiB;AAC1B,uBAAmB,OAAO,cAAc,sBAAsB;AAChE,MAAI,OAAO,gBAAgB,UAAa,OAAO,iBAAiB,QAAW;AACzE,UAAM,IAAI;AAAA,MACR;AAAA,MACA;AAAA,IACF;AAAA,EACF;AACA,QAAM,MAAM,OAAO;AACnB,MAAI,QAAQ,QAAQ,OAAO,QAAQ,UAAU;AAC3C,UAAM,IAAI,yBAAyB,iCAAiC,aAAa;AAAA,EACnF;AACA,aAAW,CAAC,GAAG,CAAC,KAAK,OAAO,QAAQ,GAA8B,GAAG;AACnE,uBAAmB,GAAG,eAAe,CAAC,EAAE;AAAA,EAC1C;AAEA,MAAI,OAAO,aAAa,QAAW;AACjC,UAAM,IAAI,OAAO;AACjB,QAAI,MAAM,QAAQ,OAAO,MAAM,UAAU;AACvC,YAAM,IAAI,yBAAyB,sCAAsC,kBAAkB;AAAA,IAC7F;AACA,UAAM,KAAK;AACX,uBAAmB,GAAG,OAAO,wBAAwB;AACrD,QAAI,OAAO,GAAG,UAAU,WAAW;AACjC,YAAM,IAAI;AAAA,QACR;AAAA,QACA;AAAA,MACF;AAAA,IACF;AAAA,EACF;AAGA,MAAI,OAAO,gBAAgB,QAAW;AACpC,wBAAoB,OAAO,aAAa,qBAAqB;AAAA,EAC/D;AAGA,MAAI,IAAI,gBAAgB,OAAW,cAAa,IAAI,aAAa,aAAa;AAE9E,MAAI,IAAI,iBAAiB,QAAW;AAClC,QAAI;AACF,YAAM,UAAU,yBAAyB,IAAI,YAAY;AACzD,UAAI,QAAQ,UAAU,UAAa,QAAQ,UAAU,IAAI,OAAO;AAC9D,cAAM,IAAI;AAAA,UACR,uBAAuB,QAAQ,KAAK,2BAA2B,IAAI,KAAK;AAAA,UACxE;AAAA,QACF;AAAA,MACF;AACA,UAAI,QAAQ,eAAe,UAAa,QAAQ,eAAe,IAAI,YAAY;AAC7E,cAAM,IAAI;AAAA,UACR,4BAA4B,QAAQ,UAAU,gCAAgC,IAAI,UAAU;AAAA,UAC5F;AAAA,QACF;AAAA,MACF;AAAA,IACF,SAAS,OAAO;AACd,UAAI,iBAAiB,yBAA0B,OAAM;AACrD,UAAI,iBAAiB,OAAO;AAC1B,cAAM,IAAI,yBAAyB,MAAM,SAAS,cAAc;AAAA,MAClE;AACA,YAAM;AAAA,IACR;AAAA,EACF;AAGA,MAAI,OAAO,IAAI,aAAa,YAAY,CAAC,WAAW,SAAS,IAAI,QAAuB,GAAG;AACzF,UAAM,IAAI;AAAA,MACR,2BAA2B,WAAW,KAAK,IAAI,CAAC,SAAS,OAAO,IAAI,QAAQ,CAAC;AAAA,MAC7E;AAAA,IACF;AAAA,EACF;AAEA,SAAO;AACT;AAUO,SAAS,yBACd,KACmB;AACnB,MAAI,IAAI,eAAgB,QAAO,IAAI;AACnC,MAAI,IAAI,QAAQ,IAAI,mBAAmB,GAAG;AACxC,WAAO,EAAE,MAAM,aAAa,KAAK,IAAI,QAAQ;AAAA,EAC/C;AACA,MAAI,IAAI,UAAU,EAAG,QAAO,EAAE,MAAM,YAAY,KAAK,IAAI,QAAQ;AACjE,SAAO,EAAE,MAAM,cAAc,KAAK,KAAK;AACzC;AAEA,SAAS,uBAAuB,OAAgB,SAAuB;AACrE,MAAI,UAAU,QAAQ,OAAO,UAAU,UAAU;AAC/C,UAAM,IAAI,yBAAyB,oCAAoC,gBAAgB;AAAA,EACzF;AACA,QAAM,QAAQ;AACd,MAAI,MAAM,SAAS,cAAc,MAAM,SAAS,eAAe,MAAM,SAAS,cAAc;AAC1F,UAAM,IAAI;AAAA,MACR;AAAA,MACA;AAAA,IACF;AAAA,EACF;AACA,MAAI,MAAM,SAAS,cAAc;AAC/B,QAAI,MAAM,QAAQ,MAAM;AACtB,YAAM,IAAI;AAAA,QACR;AAAA,QACA;AAAA,MACF;AAAA,IACF;AACA,QAAI,YAAY,GAAG;AACjB,YAAM,IAAI;AAAA,QACR;AAAA,QACA;AAAA,MACF;AAAA,IACF;AACA;AAAA,EACF;AACA,qBAAmB,MAAM,KAAK,oBAAoB;AAClD,MAAK,MAAM,MAAiB,GAAG;AAC7B,UAAM,IAAI;AAAA,MACR;AAAA,MACA;AAAA,IACF;AAAA,EACF;AACA,MAAI,MAAM,QAAQ,SAAS;AACzB,UAAM,IAAI;AAAA,MACR;AAAA,MACA;AAAA,IACF;AAAA,EACF;AACF;AAGO,SAAS,YAAY,OAAoC;AAC9D,MAAI;AACF,sBAAkB,KAAK;AACvB,WAAO;AAAA,EACT,QAAQ;AACN,WAAO;AAAA,EACT;AACF;AAGO,SAAS,mBACd,OACiF;AACjF,MAAI;AACF,WAAO,EAAE,IAAI,MAAM,OAAO,kBAAkB,KAAK,EAAE;AAAA,EACrD,SAAS,GAAG;AACV,QAAI,aAAa,yBAA0B,QAAO,EAAE,IAAI,OAAO,OAAO,EAAE;AACxE,UAAM;AAAA,EACR;AACF;AAGO,SAAS,mBAAmB,QAA8B;AAC/D,QAAM,OAAO,KAAK,UAAU,MAAM;AAClC,SAAO,kBAAkB,KAAK,MAAM,IAAI,CAAC;AAC3C;AAIA,SAAS,aAAa,OAAgB,MAAoB;AACxD,MAAI,OAAO,UAAU,YAAY,MAAM,WAAW,GAAG;AACnD,UAAM,IAAI,yBAAyB,6BAA6B,IAAI;AAAA,EACtE;AACF;AAEA,SAAS,mBAAmB,OAAgB,MAAoB;AAC9D,MAAI,OAAO,UAAU,YAAY,CAAC,OAAO,SAAS,KAAK,GAAG;AACxD,UAAM,IAAI,yBAAyB,0BAA0B,IAAI;AAAA,EACnE;AACF;AAEA,SAAS,oBAAoB,OAAgB,MAAoB;AAC/D,MAAI,UAAU,QAAQ,OAAO,UAAU,UAAU;AAC/C,UAAM,IAAI,yBAAyB,iCAAiC,IAAI;AAAA,EAC1E;AACA,QAAM,MAAM;AAEZ,QAAM,WAAW,IAAI;AACrB,MAAI,aAAa,QAAQ,OAAO,aAAa,UAAU;AACrD,UAAM,IAAI,yBAAyB,8BAA8B,GAAG,IAAI,WAAW;AAAA,EACrF;AACA,aAAW,CAAC,SAAS,IAAI,KAAK,OAAO,QAAQ,QAAmC,GAAG;AACjF,QAAI,SAAS,QAAQ,OAAO,SAAS,UAAU;AAC7C,YAAM,IAAI;AAAA,QACR;AAAA,QACA,GAAG,IAAI,aAAa,OAAO;AAAA,MAC7B;AAAA,IACF;AACA,eAAW,CAAC,KAAK,KAAK,KAAK,OAAO,QAAQ,IAA+B,GAAG;AAC1E,yBAAmB,OAAO,GAAG,IAAI,aAAa,OAAO,IAAI,GAAG,EAAE;AAAA,IAChE;AAAA,EACF;AAEA,QAAM,aAAa,IAAI;AACvB,MAAI,eAAe,QAAQ,OAAO,eAAe,UAAU;AACzD,UAAM,IAAI,yBAAyB,gCAAgC,GAAG,IAAI,aAAa;AAAA,EACzF;AACA,aAAW,CAAC,KAAK,IAAI,KAAK,OAAO,QAAQ,UAAqC,GAAG;AAC/E,uBAAmB,MAAM,GAAG,IAAI,eAAe,GAAG,EAAE;AAAA,EACtD;AAEA,qBAAmB,IAAI,WAAW,GAAG,IAAI,YAAY;AAErD,MAAI,IAAI,iBAAiB,QAAW;AAClC,QAAI,CAAC,MAAM,QAAQ,IAAI,YAAY,GAAG;AACpC,YAAM,IAAI;AAAA,QACR;AAAA,QACA,GAAG,IAAI;AAAA,MACT;AAAA,IACF;AACA,aAAS,IAAI,GAAG,IAAI,IAAI,aAAa,QAAQ,KAAK;AAChD,YAAM,KAAK,IAAI,aAAa,CAAC;AAC7B,UAAI,OAAO,OAAO,YAAY,GAAG,WAAW,GAAG;AAC7C,cAAM,IAAI;AAAA,UACR;AAAA,UACA,GAAG,IAAI,iBAAiB,CAAC;AAAA,QAC3B;AAAA,MACF;AAAA,IACF;AAAA,EACF;AAEA,MAAI,IAAI,UAAU,UAAa,OAAO,IAAI,UAAU,UAAU;AAC5D,UAAM,IAAI,yBAAyB,0BAA0B,GAAG,IAAI,QAAQ;AAAA,EAC9E;AACF;AAYO,SAAS,iBAAiB,OAAwB;AACvD,MAAI,MAAM,SAAS,GAAG,EAAG,QAAO;AAChC,MAAI,UAAU,KAAK,KAAK,EAAG,QAAO;AAClC,MAAI,sBAAsB,KAAK,KAAK,EAAG,QAAO;AAC9C,MAAI,SAAS,KAAK,KAAK,EAAG,QAAO;AACjC,SAAO;AACT;","names":[]}
@@ -13,7 +13,7 @@ import {
13
13
  } from "./chunk-GGE4NNQT.js";
14
14
  import {
15
15
  callLlm
16
- } from "./chunk-FUCQVFMU.js";
16
+ } from "./chunk-GY4SYVPJ.js";
17
17
  import {
18
18
  ValidationError
19
19
  } from "./chunk-ONWEPEDO.js";
@@ -2633,4 +2633,4 @@ export {
2633
2633
  provenanceSpansPath,
2634
2634
  emitLoopProvenance
2635
2635
  };
2636
- //# sourceMappingURL=chunk-N6MTC3GK.js.map
2636
+ //# sourceMappingURL=chunk-ADYLPOSX.js.map
@@ -3,7 +3,7 @@ import {
3
3
  } from "./chunk-TVVP3ZZQ.js";
4
4
  import {
5
5
  validateRunRecord
6
- } from "./chunk-VK6HBGAE.js";
6
+ } from "./chunk-5UF54T55.js";
7
7
 
8
8
  // src/action-policy.ts
9
9
  function evaluateActionPolicy(action, policy = {}, options = {}) {
@@ -1523,4 +1523,4 @@ export {
1523
1523
  runProposeReviewAsControlLoop,
1524
1524
  controlFailureClassFromVerification
1525
1525
  };
1526
- //# sourceMappingURL=chunk-IMWDSFUM.js.map
1526
+ //# sourceMappingURL=chunk-DXZRATT5.js.map