@tangle-network/agent-eval 0.115.0 → 0.115.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +20 -0
- package/dist/analyst/index.d.ts +7 -7
- package/dist/{analyze-runs-Dmz6LA9e.d.ts → analyze-runs-BYHg6Irm.d.ts} +3 -3
- package/dist/belief-state/index.d.ts +3 -3
- package/dist/belief-state/index.js +1 -1
- package/dist/benchmarks/index.d.ts +3 -3
- package/dist/benchmarks/index.js +2 -2
- package/dist/campaign/index.d.ts +13 -12
- package/dist/campaign/index.js +2 -2
- package/dist/{chunk-MNR6ZW4P.js → chunk-5NVBGKPH.js} +3 -3
- package/dist/chunk-5NVBGKPH.js.map +1 -0
- package/dist/{chunk-VK6HBGAE.js → chunk-5UF54T55.js} +53 -1
- package/dist/chunk-5UF54T55.js.map +1 -0
- package/dist/{chunk-IMWDSFUM.js → chunk-DXZRATT5.js} +2 -2
- package/dist/{chunk-S42AWHMP.js → chunk-E4BUPP7Z.js} +135 -17
- package/dist/chunk-E4BUPP7Z.js.map +1 -0
- package/dist/{chunk-WBOGKYM4.js → chunk-J6P6PK2R.js} +48 -7
- package/dist/chunk-J6P6PK2R.js.map +1 -0
- package/dist/{chunk-LOBMT6SB.js → chunk-QG5F6463.js} +2 -2
- package/dist/{chunk-RSVSSZKF.js → chunk-TLDB7WRY.js} +2 -2
- package/dist/{code-agent-session-yitf9I-F.d.ts → code-agent-session-D-g04tcy.d.ts} +8 -1
- package/dist/contract/index.d.ts +15 -15
- package/dist/contract/index.js +3 -3
- package/dist/{control-U8LBKUES.d.ts → control-CcBiAEnn.d.ts} +1 -1
- package/dist/control.d.ts +2 -2
- package/dist/control.js +2 -2
- package/dist/{default-registry-Bcf1uKVI.d.ts → default-registry-DltpYR5u.d.ts} +1 -1
- package/dist/{gepa-DolL_Fko.d.ts → gepa-dne9JDPL.d.ts} +1 -1
- package/dist/hosted/index.d.ts +4 -4
- package/dist/{index-CWr5SIG-.d.ts → index-BTEpx9He.d.ts} +2 -2
- package/dist/index.d.ts +22 -22
- package/dist/index.js +8 -6
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-D4cXFsLt.d.ts → insight-report-IwwvqZZv.d.ts} +20 -2
- package/dist/{kind-factory-20hcaYpf.d.ts → kind-factory-DcNg13sZ.d.ts} +1 -1
- package/dist/meta-eval/index.d.ts +2 -2
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/{policy-edit-az2qRmvN.d.ts → policy-edit-RLn8GWof.d.ts} +2 -2
- package/dist/{pre-registration-oNItiRBb.d.ts → pre-registration-D8h7ZxNL.d.ts} +3 -3
- package/dist/{provenance-BZmpWmn4.d.ts → provenance-Bibyg1U9.d.ts} +3 -3
- package/dist/{release-report-oBfOz8ku.d.ts → release-report-CCtzajxP.d.ts} +2 -2
- package/dist/reporting.d.ts +4 -4
- package/dist/{researcher-CaH0CwFC.d.ts → researcher-Dq-EtpbE.d.ts} +2 -2
- package/dist/rl.d.ts +6 -6
- package/dist/rl.js +2 -2
- package/dist/{rubric-predictive-validity-C-fMteAW.d.ts → rubric-predictive-validity-DYTLjGWu.d.ts} +1 -1
- package/dist/{run-record-DksGsfgv.d.ts → run-record-B7RTi_ix.d.ts} +34 -2
- package/dist/{runtime-trajectory-h5i0SZUj.d.ts → runtime-trajectory-Dws7Kpgi.d.ts} +1 -1
- package/dist/{semantic-concept-judge-CpzbtwD0.d.ts → semantic-concept-judge-DxJmRkyJ.d.ts} +1 -1
- package/dist/{summary-report-Bz-0-t8v.d.ts → summary-report-BJ5aNwZ1.d.ts} +1 -1
- package/dist/traces.d.ts +1 -1
- package/dist/traces.js +2 -2
- package/dist/{types-CgSlO6wT.d.ts → types-C5gJrOVT.d.ts} +1 -1
- package/package.json +1 -1
- package/dist/chunk-MNR6ZW4P.js.map +0 -1
- package/dist/chunk-S42AWHMP.js.map +0 -1
- package/dist/chunk-VK6HBGAE.js.map +0 -1
- package/dist/chunk-WBOGKYM4.js.map +0 -1
- /package/dist/{chunk-IMWDSFUM.js.map → chunk-DXZRATT5.js.map} +0 -0
- /package/dist/{chunk-LOBMT6SB.js.map → chunk-QG5F6463.js.map} +0 -0
- /package/dist/{chunk-RSVSSZKF.js.map → chunk-TLDB7WRY.js.map} +0 -0
package/CHANGELOG.md
CHANGED
|
@@ -4,6 +4,26 @@ All notable changes to `@tangle-network/agent-eval` and its sibling `agent-eval-
|
|
|
4
4
|
|
|
5
5
|
---
|
|
6
6
|
|
|
7
|
+
## [0.115.2] — 2026-07-12 — truthful code-agent session accounting
|
|
8
|
+
|
|
9
|
+
### Fixed
|
|
10
|
+
|
|
11
|
+
- fix(contract): ingest direct Codex 0.144.x exec JSONL lifecycle, tool, patch, terminal, and token events without double-counting transitions or reasoning tokens.
|
|
12
|
+
- fix(contract): preserve observed, estimated, and uncaptured USD provenance through code-agent session intake and analyzeRuns.
|
|
13
|
+
|
|
14
|
+
This patch corrects imported trace and cost semantics while retaining backward-compatible serialized RunRecords.
|
|
15
|
+
Consumers importing code-agent execution traces should update.
|
|
16
|
+
|
|
17
|
+
## [0.115.1] — 2026-07-11 — fair cross-surface baseline selection
|
|
18
|
+
|
|
19
|
+
### Fixed
|
|
20
|
+
|
|
21
|
+
- `analyzeCrossSurfaceInteractions()` now builds the naive stack only from single-surface candidates that satisfy individual eligibility.
|
|
22
|
+
Complete, non-regressing neutral constituents remain available exclusively to interaction-aware search, preserving pure-synergy discovery without weakening the naive comparison.
|
|
23
|
+
|
|
24
|
+
This patch corrects selection semantics without changing the report schema.
|
|
25
|
+
Consumers comparing naive and interaction-aware compositions should update.
|
|
26
|
+
|
|
7
27
|
## [0.115.0] — 2026-07-11 — auditable cross-surface improvement search
|
|
8
28
|
|
|
9
29
|
### Added
|
package/dist/analyst/index.d.ts
CHANGED
|
@@ -1,20 +1,20 @@
|
|
|
1
1
|
import { M as MultiLayerVerifier, V as VerifyOptions, S as Severity } from '../multi-layer-verifier-BsqKuLyN.js';
|
|
2
|
-
import { R as RunCritic, S as SemanticConceptJudgeOptions, a as RunTrace, b as SemanticConceptJudgeInput, B as BehavioralMetrics } from '../semantic-concept-judge-
|
|
3
|
-
export { C as CreateAnalystAiConfig, D as DEFAULT_TRACE_ANALYST_KINDS, c as DiffPolicy, F as FAILURE_MODE_KIND_SPEC, d as FINDING_SUBJECT_GRAMMAR_PROMPT, e as FINDING_SUBJECT_KINDS, f as FINDING_SUBJECT_SYNTAX, g as FindingSubject, h as FindingSubjectKind, i as FindingSubjectStringSchema, j as FindingsDiff, k as FindingsStore, I as IMPROVEMENT_KIND_SPEC, K as KIND_EXPECTED_SUBJECTS, l as KNOWLEDGE_GAP_KIND_SPEC, m as KNOWLEDGE_POISONING_KIND_SPEC, P as PersistedFinding, n as SKILL_USAGE_ANALYST, o as SkillUsageAnalyst, p as SkillUsageRecord, q as SkillUsageReport, r as SkillUsageScanConfig, s as buildSkillUsageReport, t as createAnalystAi, u as defaultIsMaterial, v as diffFindings, w as emitSkillUsageFindings, x as findingSubjectGrammarPromptFor, y as parseFindingSubject, z as renderFindingSubject } from '../semantic-concept-judge-
|
|
2
|
+
import { R as RunCritic, S as SemanticConceptJudgeOptions, a as RunTrace, b as SemanticConceptJudgeInput, B as BehavioralMetrics } from '../semantic-concept-judge-DxJmRkyJ.js';
|
|
3
|
+
export { C as CreateAnalystAiConfig, D as DEFAULT_TRACE_ANALYST_KINDS, c as DiffPolicy, F as FAILURE_MODE_KIND_SPEC, d as FINDING_SUBJECT_GRAMMAR_PROMPT, e as FINDING_SUBJECT_KINDS, f as FINDING_SUBJECT_SYNTAX, g as FindingSubject, h as FindingSubjectKind, i as FindingSubjectStringSchema, j as FindingsDiff, k as FindingsStore, I as IMPROVEMENT_KIND_SPEC, K as KIND_EXPECTED_SUBJECTS, l as KNOWLEDGE_GAP_KIND_SPEC, m as KNOWLEDGE_POISONING_KIND_SPEC, P as PersistedFinding, n as SKILL_USAGE_ANALYST, o as SkillUsageAnalyst, p as SkillUsageRecord, q as SkillUsageReport, r as SkillUsageScanConfig, s as buildSkillUsageReport, t as createAnalystAi, u as defaultIsMaterial, v as diffFindings, w as emitSkillUsageFindings, x as findingSubjectGrammarPromptFor, y as parseFindingSubject, z as renderFindingSubject } from '../semantic-concept-judge-DxJmRkyJ.js';
|
|
4
4
|
import { b as JudgeFn, a as JudgeInput } from '../types-C7DGg5ex.js';
|
|
5
|
-
import { a as Analyst, g as AnalystSeverity, A as AnalystFinding } from '../kind-factory-
|
|
6
|
-
export { h as ANALYST_SEVERITIES, b as AnalystContext, i as AnalystCost, j as AnalystInputKind, k as AnalystRequirements, f as AnalystRunEvent, e as AnalystRunInputs, d as AnalystRunResult, c as AnalystRunSummary, l as ChatCallOpts, C as ChatClient, m as ChatRequest, n as ChatResponse, o as ChatTransport, p as CliBridgeTransportOpts, q as CreateChatClientOpts, r as CreateTraceAnalystKindOpts, D as DirectProviderTransportOpts, E as EvidenceRef, M as MockTransportOpts, R as RAW_FINDING_SCHEMA_PROMPT, s as RawAnalystFinding, t as RawAnalystFindingSchema, u as RouterTransportOpts, S as SandboxSdkTransportOpts, v as TraceAnalystGolden, T as TraceAnalystKindSpec, w as computeFindingId, x as createChatClient, y as createTraceAnalystKind, z as makeFinding, B as parseRawFinding, F as renderPriorFindings } from '../kind-factory-
|
|
5
|
+
import { a as Analyst, g as AnalystSeverity, A as AnalystFinding } from '../kind-factory-DcNg13sZ.js';
|
|
6
|
+
export { h as ANALYST_SEVERITIES, b as AnalystContext, i as AnalystCost, j as AnalystInputKind, k as AnalystRequirements, f as AnalystRunEvent, e as AnalystRunInputs, d as AnalystRunResult, c as AnalystRunSummary, l as ChatCallOpts, C as ChatClient, m as ChatRequest, n as ChatResponse, o as ChatTransport, p as CliBridgeTransportOpts, q as CreateChatClientOpts, r as CreateTraceAnalystKindOpts, D as DirectProviderTransportOpts, E as EvidenceRef, M as MockTransportOpts, R as RAW_FINDING_SCHEMA_PROMPT, s as RawAnalystFinding, t as RawAnalystFindingSchema, u as RouterTransportOpts, S as SandboxSdkTransportOpts, v as TraceAnalystGolden, T as TraceAnalystKindSpec, w as computeFindingId, x as createChatClient, y as createTraceAnalystKind, z as makeFinding, B as parseRawFinding, F as renderPriorFindings } from '../kind-factory-DcNg13sZ.js';
|
|
7
7
|
import { TCloud } from '@tangle-network/tcloud';
|
|
8
8
|
import { T as TraceAnalysisStore } from '../store-C1YxJDEK.js';
|
|
9
|
-
export { a as AnalystHooks, A as AnalystRegistry, b as AnalystRegistryOptions, B as BudgetPolicy, D as DefaultAnalystRegistryOptions, R as RegistryRunOpts, c as buildDefaultAnalystRegistry } from '../default-registry-
|
|
10
|
-
export { F as FindingToPolicyEditOptions, P as POLICY_EDIT_AXES, a as POLICY_EDIT_TARGET_SURFACES, b as PolicyEdit, c as PolicyEditAdmission, d as PolicyEditAdmissionOptions, e as PolicyEditAxis, f as PolicyEditChange, g as PolicyEditExpectedGain, h as PolicyEditGainDirection, i as PolicyEditGainUnit, j as PolicyEditInit, k as PolicyEditRisk, l as PolicyEditSchemaVersion, m as PolicyEditSource, n as PolicyEditTarget, o as PolicyEditTargetSurface, p as PolicyEditValidationError, q as admitPolicyEdit, r as applyPolicyEditToSurface, s as computePolicyEditId, t as isPolicyEdit, u as makePolicyEdit, v as policyEditFromFinding, w as policyEditsFromFindings, x as scorePolicyEditReadiness, y as validatePolicyEdit } from '../policy-edit-
|
|
9
|
+
export { a as AnalystHooks, A as AnalystRegistry, b as AnalystRegistryOptions, B as BudgetPolicy, D as DefaultAnalystRegistryOptions, R as RegistryRunOpts, c as buildDefaultAnalystRegistry } from '../default-registry-DltpYR5u.js';
|
|
10
|
+
export { F as FindingToPolicyEditOptions, P as POLICY_EDIT_AXES, a as POLICY_EDIT_TARGET_SURFACES, b as PolicyEdit, c as PolicyEditAdmission, d as PolicyEditAdmissionOptions, e as PolicyEditAxis, f as PolicyEditChange, g as PolicyEditExpectedGain, h as PolicyEditGainDirection, i as PolicyEditGainUnit, j as PolicyEditInit, k as PolicyEditRisk, l as PolicyEditSchemaVersion, m as PolicyEditSource, n as PolicyEditTarget, o as PolicyEditTargetSurface, p as PolicyEditValidationError, q as admitPolicyEdit, r as applyPolicyEditToSurface, s as computePolicyEditId, t as isPolicyEdit, u as makePolicyEdit, v as policyEditFromFinding, w as policyEditsFromFindings, x as scorePolicyEditReadiness, y as validatePolicyEdit } from '../policy-edit-RLn8GWof.js';
|
|
11
11
|
import { L as LlmClientOptions } from '../llm-client-DyqEH4jH.js';
|
|
12
12
|
import { AxFunction } from '@ax-llm/ax';
|
|
13
13
|
import '../verdict-C9MlYujm.js';
|
|
14
14
|
import 'zod';
|
|
15
15
|
import '../schema-SGWcK9wa.js';
|
|
16
16
|
import '../store-BsVi7ncX.js';
|
|
17
|
-
import '../run-record-
|
|
17
|
+
import '../run-record-B7RTi_ix.js';
|
|
18
18
|
import '@tangle-network/agent-interface';
|
|
19
19
|
import '../errors-oeQrLqXC.js';
|
|
20
20
|
import '../raw-provider-sink-C46HDghv.js';
|
|
@@ -1,7 +1,7 @@
|
|
|
1
|
-
import { A as AnalystRegistry } from './default-registry-
|
|
1
|
+
import { A as AnalystRegistry } from './default-registry-DltpYR5u.js';
|
|
2
2
|
import { D as DatasetScenario } from './dataset-NENEzRgk.js';
|
|
3
|
-
import { R as RunRecord } from './run-record-
|
|
4
|
-
import { I as InsightReport } from './insight-report-
|
|
3
|
+
import { R as RunRecord } from './run-record-B7RTi_ix.js';
|
|
4
|
+
import { I as InsightReport } from './insight-report-IwwvqZZv.js';
|
|
5
5
|
|
|
6
6
|
/**
|
|
7
7
|
* # `analyzeRuns()` — turn a set of agent runs into an actionable decision packet.
|
|
@@ -1,9 +1,9 @@
|
|
|
1
1
|
import { c as CalibrationReport } from '../calibration-Dz8TQV4y.js';
|
|
2
2
|
import { O as OffPolicyEstimate, a as OffPolicyOptions, b as OffPolicyTrajectory } from '../off-policy-DiwuKKg7.js';
|
|
3
|
-
import { d as CodeAgentSessionSource, a as CodeAgentSessionIntakeOptions, c as CodeAgentSessionMetrics, C as CodeAgentSessionDiagnostic } from '../code-agent-session-
|
|
4
|
-
import { R as RunRecord, b as RunSplitTag } from '../run-record-
|
|
3
|
+
import { d as CodeAgentSessionSource, a as CodeAgentSessionIntakeOptions, c as CodeAgentSessionMetrics, C as CodeAgentSessionDiagnostic } from '../code-agent-session-D-g04tcy.js';
|
|
4
|
+
import { R as RunRecord, b as RunSplitTag } from '../run-record-B7RTi_ix.js';
|
|
5
5
|
import { T as TraceStore } from '../store-BsVi7ncX.js';
|
|
6
|
-
import { R as RuntimeTrajectoryRecord, P as ProjectRuntimeTrajectoryEvidenceOptions, a as RuntimeTrajectoryEvidenceProjection } from '../runtime-trajectory-
|
|
6
|
+
import { R as RuntimeTrajectoryRecord, P as ProjectRuntimeTrajectoryEvidenceOptions, a as RuntimeTrajectoryEvidenceProjection } from '../runtime-trajectory-Dws7Kpgi.js';
|
|
7
7
|
import '../schema-SGWcK9wa.js';
|
|
8
8
|
import '../outcome-store-rnXLEqSn.js';
|
|
9
9
|
import '@tangle-network/agent-interface';
|
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
export { B as BENCHMARK_SPLIT_SEED, a as BenchmarkAdapter, b as BenchmarkDatasetItem, k as BenchmarkDistribution, c as BenchmarkEvaluation, d as BenchmarkFamily, l as BenchmarkMetricCalibrationOptions, m as BenchmarkMetricCalibrationResult, n as BenchmarkReport, e as BenchmarkResponder, o as BenchmarkRunOptions, p as BenchmarkRunResult, f as BenchmarkScenario, q as BenchmarkSliceSummary, g as BenchmarkSource, h as BenchmarkTaskKind, r as BuildStandardRetrievalItemsOptions, R as RetrievalIdAdapterOptions, S as StandardRetrievalArtifact, s as StandardRetrievalDocument, t as StandardRetrievalEvaluationOptions, u as StandardRetrievalPayload, v as StandardRetrievalQrel, w as StandardRetrievalQuery, x as StandardRetrievalResult, y as buildStandardRetrievalItems, z as calibrateBenchmarkMetric, A as createRetrievalIdBenchmarkAdapter, i as deterministicSplit, C as evaluateStandardRetrieval, D as normalizeRetrievedDocumentIds, E as parseBeirCorpusJsonl, F as parseBeirQueriesJsonl, G as parseJsonlRows, H as parseQrels, I as parseTsvRows, J as renderBenchmarkReportMarkdown, K as retrievalMetricsAtCutoff, L as routing, M as runBenchmarkAdapter, N as summarizeBenchmarkCampaign } from '../index-
|
|
2
|
-
import '../types-
|
|
3
|
-
import '../run-record-
|
|
1
|
+
export { B as BENCHMARK_SPLIT_SEED, a as BenchmarkAdapter, b as BenchmarkDatasetItem, k as BenchmarkDistribution, c as BenchmarkEvaluation, d as BenchmarkFamily, l as BenchmarkMetricCalibrationOptions, m as BenchmarkMetricCalibrationResult, n as BenchmarkReport, e as BenchmarkResponder, o as BenchmarkRunOptions, p as BenchmarkRunResult, f as BenchmarkScenario, q as BenchmarkSliceSummary, g as BenchmarkSource, h as BenchmarkTaskKind, r as BuildStandardRetrievalItemsOptions, R as RetrievalIdAdapterOptions, S as StandardRetrievalArtifact, s as StandardRetrievalDocument, t as StandardRetrievalEvaluationOptions, u as StandardRetrievalPayload, v as StandardRetrievalQrel, w as StandardRetrievalQuery, x as StandardRetrievalResult, y as buildStandardRetrievalItems, z as calibrateBenchmarkMetric, A as createRetrievalIdBenchmarkAdapter, i as deterministicSplit, C as evaluateStandardRetrieval, D as normalizeRetrievedDocumentIds, E as parseBeirCorpusJsonl, F as parseBeirQueriesJsonl, G as parseJsonlRows, H as parseQrels, I as parseTsvRows, J as renderBenchmarkReportMarkdown, K as retrievalMetricsAtCutoff, L as routing, M as runBenchmarkAdapter, N as summarizeBenchmarkCampaign } from '../index-BTEpx9He.js';
|
|
2
|
+
import '../types-C5gJrOVT.js';
|
|
3
|
+
import '../run-record-B7RTi_ix.js';
|
|
4
4
|
import '@tangle-network/agent-interface';
|
|
5
5
|
import '../errors-oeQrLqXC.js';
|
|
6
6
|
import '../schema-SGWcK9wa.js';
|
package/dist/benchmarks/index.js
CHANGED
|
@@ -17,7 +17,7 @@ import {
|
|
|
17
17
|
runBenchmarkAdapter,
|
|
18
18
|
summarizeBenchmarkCampaign
|
|
19
19
|
} from "../chunk-3LXTCTWL.js";
|
|
20
|
-
import "../chunk-
|
|
20
|
+
import "../chunk-5NVBGKPH.js";
|
|
21
21
|
import "../chunk-N6MTC3GK.js";
|
|
22
22
|
import "../chunk-VI2UW6B6.js";
|
|
23
23
|
import "../chunk-FAOEFFRT.js";
|
|
@@ -28,7 +28,7 @@ import "../chunk-PJQFMIOX.js";
|
|
|
28
28
|
import "../chunk-RPDDVKI7.js";
|
|
29
29
|
import "../chunk-GGE4NNQT.js";
|
|
30
30
|
import "../chunk-LNQEP766.js";
|
|
31
|
-
import "../chunk-
|
|
31
|
+
import "../chunk-5UF54T55.js";
|
|
32
32
|
import "../chunk-XJYR7XFV.js";
|
|
33
33
|
import "../chunk-VSMTAMNK.js";
|
|
34
34
|
import "../chunk-FUCQVFMU.js";
|
package/dist/campaign/index.d.ts
CHANGED
|
@@ -1,20 +1,20 @@
|
|
|
1
|
-
import { P as PairedArmsComparison, S as SignedManifest, B as BackendIntegrityReport, C as CompletionRequirement, R as RuntimeEventLike, a as CompletionVerdict, b as ProducedState, c as CorrectnessChecker } from '../pre-registration-
|
|
2
|
-
export { L as LlmJudgeDimension, d as LlmJudgeOptions, l as llmJudge } from '../pre-registration-
|
|
1
|
+
import { P as PairedArmsComparison, S as SignedManifest, B as BackendIntegrityReport, C as CompletionRequirement, R as RuntimeEventLike, a as CompletionVerdict, b as ProducedState, c as CorrectnessChecker } from '../pre-registration-D8h7ZxNL.js';
|
|
2
|
+
export { L as LlmJudgeDimension, d as LlmJudgeOptions, l as llmJudge } from '../pre-registration-D8h7ZxNL.js';
|
|
3
3
|
import { A as AnalyzeTracesOptions, a as AnalyzeTracesInput, b as AnalyzeTracesResult } from '../analyst-C8HHvfJp.js';
|
|
4
|
-
import { S as Scenario, M as MutableSurface, D as DispatchContext, b as JudgeConfig, g as Gate, e as GenerationRecord, J as JudgeScore, L as LabeledScenarioStore, s as LabeledScenarioWrite, t as LabeledScenarioSampleArgs, u as LabeledScenarioRecord, v as LabelTrust, f as SurfaceProposer, w as ProposedCandidate, x as ProposeContext, m as CodeSurface, y as LabeledScenarioSource, C as CampaignResult } from '../types-
|
|
5
|
-
export { i as CampaignAggregates, j as CampaignArtifactWriter, k as CampaignCellResult, l as CampaignCostMeter, z as CampaignTokenUsage, d as CampaignTraceWriter, c as DispatchFn, n as GateContext, h as GateDecision, G as GateResult, o as GenerationCandidate, A as JudgeAggregate, a as JudgeDimension, p as Mutator, O as OptimizationProposer, q as OptimizerConfig, P as ParetoParent, R as RedactionStatus, B as ScenarioAggregate, r as SessionScript, T as TraceSpan, E as isProposedCandidate, F as labelTrustRank } from '../types-
|
|
6
|
-
import { C as CampaignRunPlan, P as PlanCampaignRunOptions, b as RunCampaignOptions, c as RunImprovementLoopOptions } from '../gepa-
|
|
7
|
-
export { f as CampaignRunPlanCell, h as GepaProposerConstraints, G as GepaProposerOptions, O as OpenAutoPrOptions, i as OpenAutoPrResult, a as RunImprovementLoopResult, R as RunOptimizationOptions, j as RunOptimizationResult, k as countSentenceEdits, l as defaultRenderDiff, m as extractH2Sections, g as gepaProposer, o as openAutoPr, p as planCampaignRun, r as runCampaign, d as runImprovementLoop, n as runOptimization } from '../gepa-
|
|
4
|
+
import { S as Scenario, M as MutableSurface, D as DispatchContext, b as JudgeConfig, g as Gate, e as GenerationRecord, J as JudgeScore, L as LabeledScenarioStore, s as LabeledScenarioWrite, t as LabeledScenarioSampleArgs, u as LabeledScenarioRecord, v as LabelTrust, f as SurfaceProposer, w as ProposedCandidate, x as ProposeContext, m as CodeSurface, y as LabeledScenarioSource, C as CampaignResult } from '../types-C5gJrOVT.js';
|
|
5
|
+
export { i as CampaignAggregates, j as CampaignArtifactWriter, k as CampaignCellResult, l as CampaignCostMeter, z as CampaignTokenUsage, d as CampaignTraceWriter, c as DispatchFn, n as GateContext, h as GateDecision, G as GateResult, o as GenerationCandidate, A as JudgeAggregate, a as JudgeDimension, p as Mutator, O as OptimizationProposer, q as OptimizerConfig, P as ParetoParent, R as RedactionStatus, B as ScenarioAggregate, r as SessionScript, T as TraceSpan, E as isProposedCandidate, F as labelTrustRank } from '../types-C5gJrOVT.js';
|
|
6
|
+
import { C as CampaignRunPlan, P as PlanCampaignRunOptions, b as RunCampaignOptions, c as RunImprovementLoopOptions } from '../gepa-dne9JDPL.js';
|
|
7
|
+
export { f as CampaignRunPlanCell, h as GepaProposerConstraints, G as GepaProposerOptions, O as OpenAutoPrOptions, i as OpenAutoPrResult, a as RunImprovementLoopResult, R as RunOptimizationOptions, j as RunOptimizationResult, k as countSentenceEdits, l as defaultRenderDiff, m as extractH2Sections, g as gepaProposer, o as openAutoPr, p as planCampaignRun, r as runCampaign, d as runImprovementLoop, n as runOptimization } from '../gepa-dne9JDPL.js';
|
|
8
8
|
import { a as PairedBootstrapResult, E as EProcessState } from '../statistics-oUbOJe-S.js';
|
|
9
9
|
import { C as CampaignStorage } from '../storage-Dw_f7WMt.js';
|
|
10
10
|
export { f as fsCampaignStorage, i as inMemoryCampaignStorage } from '../storage-Dw_f7WMt.js';
|
|
11
|
-
export { A as AxisEvidence, a as AxisVerdict, B as BuildEvidenceVectorOptions, l as BuildLoopProvenanceArgs, D as DefaultProductionGateOptions, m as EmitLoopProvenanceArgs, n as EmitLoopProvenanceResult, E as EvidenceVector, b as EvolutionaryProposerOptions, H as HeldOutGateOptions, o as LoopProvenanceBackend, q as LoopProvenanceCandidate, L as LoopProvenanceRecord, O as ObjectiveSource, c as ParetoSignificanceGateOptions, P as PowerPreflight, s as PowerPreflightOptions, d as PromotionObjective, e as PromotionPolicy, R as RunEvalOptions, f as buildEvidenceVector, t as buildLoopProvenanceRecord, g as composeGate, h as defaultProductionGate, u as emitLoopProvenance, i as evolutionaryProposer, j as heldOutGate, v as loopProvenanceSpans, p as paretoPolicy, k as paretoSignificanceGate, w as powerPreflight, x as provenanceRecordPath, y as provenanceSpansPath, r as runEval } from '../provenance-
|
|
11
|
+
export { A as AxisEvidence, a as AxisVerdict, B as BuildEvidenceVectorOptions, l as BuildLoopProvenanceArgs, D as DefaultProductionGateOptions, m as EmitLoopProvenanceArgs, n as EmitLoopProvenanceResult, E as EvidenceVector, b as EvolutionaryProposerOptions, H as HeldOutGateOptions, o as LoopProvenanceBackend, q as LoopProvenanceCandidate, L as LoopProvenanceRecord, O as ObjectiveSource, c as ParetoSignificanceGateOptions, P as PowerPreflight, s as PowerPreflightOptions, d as PromotionObjective, e as PromotionPolicy, R as RunEvalOptions, f as buildEvidenceVector, t as buildLoopProvenanceRecord, g as composeGate, h as defaultProductionGate, u as emitLoopProvenance, i as evolutionaryProposer, j as heldOutGate, v as loopProvenanceSpans, p as paretoPolicy, k as paretoSignificanceGate, w as powerPreflight, x as provenanceRecordPath, y as provenanceSpansPath, r as runEval } from '../provenance-Bibyg1U9.js';
|
|
12
12
|
import { L as LlmClientOptions } from '../llm-client-DyqEH4jH.js';
|
|
13
13
|
import { AgentProfile } from '@tangle-network/agent-interface';
|
|
14
14
|
import { A as AgentEvalError, V as ValidationError } from '../errors-oeQrLqXC.js';
|
|
15
|
-
import { b as RunSplitTag, R as RunRecord } from '../run-record-
|
|
16
|
-
import { b as PolicyEdit, F as FindingToPolicyEditOptions, d as PolicyEditAdmissionOptions, c as PolicyEditAdmission } from '../policy-edit-
|
|
17
|
-
import { T as TraceAnalystKindSpec, A as AnalystFinding } from '../kind-factory-
|
|
15
|
+
import { b as RunSplitTag, R as RunRecord } from '../run-record-B7RTi_ix.js';
|
|
16
|
+
import { b as PolicyEdit, F as FindingToPolicyEditOptions, d as PolicyEditAdmissionOptions, c as PolicyEditAdmission } from '../policy-edit-RLn8GWof.js';
|
|
17
|
+
import { T as TraceAnalystKindSpec, A as AnalystFinding } from '../kind-factory-DcNg13sZ.js';
|
|
18
18
|
import '@tangle-network/tcloud';
|
|
19
19
|
import '../raw-provider-sink-C46HDghv.js';
|
|
20
20
|
import '../verdict-C9MlYujm.js';
|
|
@@ -26,8 +26,8 @@ import '../schema-SGWcK9wa.js';
|
|
|
26
26
|
import '../judge-calibration-7C-IDmKr.js';
|
|
27
27
|
import '../types-C7DGg5ex.js';
|
|
28
28
|
import '../hosted/index.js';
|
|
29
|
-
import '../insight-report-
|
|
30
|
-
import '../summary-report-
|
|
29
|
+
import '../insight-report-IwwvqZZv.js';
|
|
30
|
+
import '../summary-report-BJ5aNwZ1.js';
|
|
31
31
|
import '../failure-cluster-C48PiReX.js';
|
|
32
32
|
import 'zod';
|
|
33
33
|
|
|
@@ -572,6 +572,7 @@ interface CrossSurfaceBestSingleSelection {
|
|
|
572
572
|
ranking: CrossSurfaceRankedSingle[];
|
|
573
573
|
}
|
|
574
574
|
interface CrossSurfaceNaiveStackSelection {
|
|
575
|
+
/** Every individually eligible single, stacked in canonical component order. */
|
|
575
576
|
candidateId: string;
|
|
576
577
|
componentIds: string[];
|
|
577
578
|
}
|
package/dist/campaign/index.js
CHANGED
|
@@ -65,7 +65,7 @@ import {
|
|
|
65
65
|
userStoryScoreboard,
|
|
66
66
|
validateSearchLedgerEvent,
|
|
67
67
|
verifyCodeSurface
|
|
68
|
-
} from "../chunk-
|
|
68
|
+
} from "../chunk-5NVBGKPH.js";
|
|
69
69
|
import {
|
|
70
70
|
assertCodeSurfaceIdentity,
|
|
71
71
|
buildEvidenceVector,
|
|
@@ -117,7 +117,7 @@ import "../chunk-PJQFMIOX.js";
|
|
|
117
117
|
import "../chunk-RPDDVKI7.js";
|
|
118
118
|
import "../chunk-GGE4NNQT.js";
|
|
119
119
|
import "../chunk-LNQEP766.js";
|
|
120
|
-
import "../chunk-
|
|
120
|
+
import "../chunk-5UF54T55.js";
|
|
121
121
|
import "../chunk-XJYR7XFV.js";
|
|
122
122
|
import "../chunk-VSMTAMNK.js";
|
|
123
123
|
import "../chunk-FUCQVFMU.js";
|
|
@@ -55,7 +55,7 @@ import {
|
|
|
55
55
|
import {
|
|
56
56
|
modelHasSnapshot,
|
|
57
57
|
validateRunRecord
|
|
58
|
-
} from "./chunk-
|
|
58
|
+
} from "./chunk-5UF54T55.js";
|
|
59
59
|
import {
|
|
60
60
|
buildAgentProfileCell
|
|
61
61
|
} from "./chunk-XJYR7XFV.js";
|
|
@@ -1769,7 +1769,7 @@ function selectCandidates(context, summaryById, eligibleSingles, interactionRead
|
|
|
1769
1769
|
componentId: best.candidate.componentIds[0],
|
|
1770
1770
|
ranking
|
|
1771
1771
|
} : null;
|
|
1772
|
-
const naiveStack = selectNaiveStack(context,
|
|
1772
|
+
const naiveStack = selectNaiveStack(context, eligibleSingles);
|
|
1773
1773
|
const interactionAware = selectInteractionAware(
|
|
1774
1774
|
context,
|
|
1775
1775
|
summaryById,
|
|
@@ -7316,4 +7316,4 @@ export {
|
|
|
7316
7316
|
verifyCodeSurface,
|
|
7317
7317
|
resolveWorktreePath
|
|
7318
7318
|
};
|
|
7319
|
-
//# sourceMappingURL=chunk-
|
|
7319
|
+
//# sourceMappingURL=chunk-5NVBGKPH.js.map
|