@tangle-network/agent-eval 0.80.0 → 0.82.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +27 -0
- package/README.md +53 -152
- package/dist/adapters/http.d.ts +2 -2
- package/dist/adapters/langchain.d.ts +2 -2
- package/dist/adapters/otel.d.ts +4 -4
- package/dist/{agent-profile-aSEaJ9Pl.d.ts → agent-profile-D0PBIWlV.d.ts} +14 -2
- package/dist/analyst/index.d.ts +10 -10
- package/dist/analyst/index.js +3 -3
- package/dist/{analyst-t7zZS3TV.d.ts → analyst-C8HHvfJp.d.ts} +1 -1
- package/dist/belief-state/index.d.ts +344 -8
- package/dist/belief-state/index.js +1518 -142
- package/dist/belief-state/index.js.map +1 -1
- package/dist/benchmarks/index.d.ts +2 -2
- package/dist/campaign/index.d.ts +40 -120
- package/dist/campaign/index.js +129 -238
- package/dist/campaign/index.js.map +1 -1
- package/dist/{chunk-RPLZ4OIB.js → chunk-BABOZOSN.js} +7 -4
- package/dist/{chunk-RPLZ4OIB.js.map → chunk-BABOZOSN.js.map} +1 -1
- package/dist/{chunk-IHDHUN2X.js → chunk-CVVHBFGN.js} +3 -3
- package/dist/chunk-CVVHBFGN.js.map +1 -0
- package/dist/{chunk-B26KI423.js → chunk-FZWAFVAA.js} +2 -2
- package/dist/{chunk-ITBRCT73.js → chunk-IDVBLYCY.js} +2 -10
- package/dist/chunk-IDVBLYCY.js.map +1 -0
- package/dist/{chunk-5LVWPNS5.js → chunk-L5G7OUKD.js} +4 -4
- package/dist/{chunk-5LVWPNS5.js.map → chunk-L5G7OUKD.js.map} +1 -1
- package/dist/{chunk-GXHLRXDI.js → chunk-OTYQPHPL.js} +4 -4
- package/dist/{chunk-6REHLN5J.js → chunk-QS3RBQPI.js} +2 -2
- package/dist/{chunk-GWGO2K6Y.js → chunk-RBNA5AZT.js} +2 -2
- package/dist/chunk-S42AWHMP.js +697 -0
- package/dist/chunk-S42AWHMP.js.map +1 -0
- package/dist/chunk-VI2UW6B6.js +162 -0
- package/dist/chunk-VI2UW6B6.js.map +1 -0
- package/dist/{chunk-CF67I6QY.js → chunk-VIDQF3F5.js} +2 -2
- package/dist/{chunk-XXNIODOM.js → chunk-WJL2NJXN.js} +3 -3
- package/dist/{chunk-LB2UOI5F.js → chunk-YGYXHNAQ.js} +47 -160
- package/dist/chunk-YGYXHNAQ.js.map +1 -0
- package/dist/{chunk-KX6F6NCG.js → chunk-Z7VFTS2J.js} +2 -2
- package/dist/{chunk-ZPSKPT3V.js → chunk-ZZ2HOPME.js} +5 -2
- package/dist/chunk-ZZ2HOPME.js.map +1 -0
- package/dist/cli.js +2 -2
- package/dist/code-agent-session-BRXmavYv.d.ts +80 -0
- package/dist/contract/index.d.ts +45 -15
- package/dist/contract/index.js +32 -7
- package/dist/contract/index.js.map +1 -1
- package/dist/{control-CehLtoET.d.ts → control-GeE8OhpN.d.ts} +1 -1
- package/dist/control.d.ts +2 -2
- package/dist/hosted/index.d.ts +4 -4
- package/dist/{index-B1RKber3.d.ts → index-DE3RXAXD.d.ts} +1 -1
- package/dist/index.d.ts +78 -287
- package/dist/index.js +87 -410
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-dlpEzQDi.d.ts → insight-report-3ADTfClO.d.ts} +1 -1
- package/dist/{kind-factory-DqV2t1Xk.d.ts → kind-factory-CVecZZG_.d.ts} +2 -2
- package/dist/{llm-client-DbjLfz-K.d.ts → llm-client-CuUg2Mn3.d.ts} +1 -1
- package/dist/meta-eval/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/pipelines/index.js +2 -2
- package/dist/{provenance-jG-Gngg8.d.ts → provenance-DPpNIOJD.d.ts} +4 -4
- package/dist/{registry-BK0Zee01.d.ts → registry-DrEQ3Luj.d.ts} +1 -1
- package/dist/{release-report-CXXZlR8g.d.ts → release-report-hlNtD12q.d.ts} +2 -2
- package/dist/reporting.d.ts +5 -5
- package/dist/reporting.js +3 -3
- package/dist/{researcher-rInLj9De.d.ts → researcher-BLPHBbNV.d.ts} +3 -3
- package/dist/rl.d.ts +7 -7
- package/dist/rl.js +4 -4
- package/dist/{rubric-predictive-validity-CLPuwiUw.d.ts → rubric-predictive-validity-CnEl9Jc8.d.ts} +1 -1
- package/dist/{run-campaign-OVEZF24D.js → run-campaign-4Y5V5CN3.js} +3 -3
- package/dist/{run-improvement-loop-BAl_aVOZ.d.ts → run-improvement-loop-CNqQckTj.d.ts} +3 -3
- package/dist/{run-record-sItO5ftF.d.ts → run-record-De9VarXR.d.ts} +1 -1
- package/dist/{semantic-concept-judge-qXEUV2w7.d.ts → semantic-concept-judge-DIEgr_6v.d.ts} +6 -6
- package/dist/{statistics-B7yCbi9i.d.ts → statistics-CnC1FMbx.d.ts} +3 -7
- package/dist/{store-GmBE2pZZ.d.ts → store-C1YxJDEK.d.ts} +1 -1
- package/dist/{summary-report-BTaXq1TS.d.ts → summary-report-Db0dDSWP.d.ts} +1 -1
- package/dist/traces.d.ts +5 -5
- package/dist/{types-DRvV0zRo.d.ts → types-Cu3u_x59.d.ts} +3 -3
- package/dist/{types-4mm2msnR.d.ts → types-D7lLRYe9.d.ts} +1 -1
- package/dist/wire/index.js +2 -2
- package/dist/workflow/index.d.ts +7 -7
- package/dist/workflow/index.js +1 -1
- package/docs/concepts.md +1 -0
- package/docs/research/belief-state-agent-eval-roadmap.md +39 -7
- package/docs/self-improvement-map.md +111 -0
- package/package.json +2 -2
- package/dist/chunk-IHDHUN2X.js.map +0 -1
- package/dist/chunk-ITBRCT73.js.map +0 -1
- package/dist/chunk-LB2UOI5F.js.map +0 -1
- package/dist/chunk-ZPSKPT3V.js.map +0 -1
- /package/dist/{chunk-B26KI423.js.map → chunk-FZWAFVAA.js.map} +0 -0
- /package/dist/{chunk-GXHLRXDI.js.map → chunk-OTYQPHPL.js.map} +0 -0
- /package/dist/{chunk-6REHLN5J.js.map → chunk-QS3RBQPI.js.map} +0 -0
- /package/dist/{chunk-GWGO2K6Y.js.map → chunk-RBNA5AZT.js.map} +0 -0
- /package/dist/{chunk-CF67I6QY.js.map → chunk-VIDQF3F5.js.map} +0 -0
- /package/dist/{chunk-XXNIODOM.js.map → chunk-WJL2NJXN.js.map} +0 -0
- /package/dist/{chunk-KX6F6NCG.js.map → chunk-Z7VFTS2J.js.map} +0 -0
- /package/dist/{run-campaign-OVEZF24D.js.map → run-campaign-4Y5V5CN3.js.map} +0 -0
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { G as GainDistributionBin, P as ParetoFigureSpec } from './summary-report-
|
|
1
|
+
import { G as GainDistributionBin, P as ParetoFigureSpec } from './summary-report-Db0dDSWP.js';
|
|
2
2
|
import { a as ContinuousAgreement } from './judge-calibration-DilmB3Ml.js';
|
|
3
3
|
|
|
4
4
|
/**
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { AxAIService, AxFunction } from '@ax-llm/ax';
|
|
2
|
-
import { T as TraceAnalysisStore } from './store-
|
|
2
|
+
import { T as TraceAnalysisStore } from './store-C1YxJDEK.js';
|
|
3
3
|
import { z } from 'zod';
|
|
4
|
-
import { g as AnalystCost, a as AnalystContext, A as Analyst } from './types-
|
|
4
|
+
import { g as AnalystCost, a as AnalystContext, A as Analyst } from './types-Cu3u_x59.js';
|
|
5
5
|
|
|
6
6
|
/**
|
|
7
7
|
* Typed Ax output for analyst findings.
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
export { C as CalibrationBin, a as CalibrationOptions, b as CalibrationPair, c as CalibrationReport, d as CorrelationResult, e as CorrelationStudyOptions, f as CorrelationStudyResult, E as EvalMetricSpec, O as OutcomePair, g as calibrationCurve, h as calibrationFromPairs, i as correlationStudy } from '../calibration-Cpr3WaX3.js';
|
|
2
2
|
export { D as DeploymentOutcome, F as FileSystemOutcomeStore, a as FileSystemOutcomeStoreOptions, I as InMemoryOutcomeStore, O as OutcomeFilter, b as OutcomeStore } from '../outcome-store-rnXLEqSn.js';
|
|
3
|
-
export { R as RubricOutcomePair, a as RubricPredictiveValidityInput, b as RubricPredictiveValidityReport, c as RubricRanking, r as rubricPredictiveValidity } from '../rubric-predictive-validity-
|
|
3
|
+
export { R as RubricOutcomePair, a as RubricPredictiveValidityInput, b as RubricPredictiveValidityReport, c as RubricRanking, r as rubricPredictiveValidity } from '../rubric-predictive-validity-CnEl9Jc8.js';
|
|
4
4
|
import '../store-CKUAgsJz.js';
|
|
5
5
|
import '../schema-m0gsnbt3.js';
|
|
6
|
-
import '../run-record-
|
|
6
|
+
import '../run-record-De9VarXR.js';
|
|
7
7
|
import '../errors-Dwqw-T_m.js';
|
package/dist/openapi.json
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
"openapi": "3.1.0",
|
|
3
3
|
"info": {
|
|
4
4
|
"title": "@tangle-network/agent-eval — wire protocol",
|
|
5
|
-
"version": "0.
|
|
5
|
+
"version": "0.82.0",
|
|
6
6
|
"description": "HTTP and stdio RPC interface to agent-eval. The TypeScript runtime is the source of truth; this spec is the contract that cross-language clients (Python, Rust, Go) generate from.\n\nWire-protocol version: 1.0.0. Bumps on breaking changes to request/response schemas.",
|
|
7
7
|
"contact": {
|
|
8
8
|
"name": "Tangle Network",
|
package/dist/pipelines/index.js
CHANGED
|
@@ -3,13 +3,13 @@ import {
|
|
|
3
3
|
classifyFailure,
|
|
4
4
|
compareToBaseline,
|
|
5
5
|
computeToolUseMetrics
|
|
6
|
-
} from "../chunk-
|
|
6
|
+
} from "../chunk-RBNA5AZT.js";
|
|
7
7
|
import {
|
|
8
8
|
buildTrajectory
|
|
9
9
|
} from "../chunk-RZTMDUO7.js";
|
|
10
10
|
import {
|
|
11
11
|
interRaterReliability
|
|
12
|
-
} from "../chunk-
|
|
12
|
+
} from "../chunk-IDVBLYCY.js";
|
|
13
13
|
import {
|
|
14
14
|
aggregateLlm,
|
|
15
15
|
argHash,
|
|
@@ -1,9 +1,9 @@
|
|
|
1
|
-
import { o as Mutator, I as ImprovementDriver, S as Scenario, G as Gate, k as GateResult, j as GateContext, g as CampaignResult, M as MutableSurface, c as GateDecision } from './types-
|
|
1
|
+
import { o as Mutator, I as ImprovementDriver, S as Scenario, G as Gate, k as GateResult, j as GateContext, g as CampaignResult, M as MutableSurface, c as GateDecision } from './types-D7lLRYe9.js';
|
|
2
2
|
import { R as RedTeamCase } from './red-team-DW9Ca_tj.js';
|
|
3
|
-
import { R as RunRecord } from './run-record-
|
|
3
|
+
import { R as RunRecord } from './run-record-De9VarXR.js';
|
|
4
4
|
import { D as Direction } from './pareto-E-pembql.js';
|
|
5
|
-
import { a as PairedBootstrapResult } from './statistics-
|
|
6
|
-
import {
|
|
5
|
+
import { a as PairedBootstrapResult } from './statistics-CnC1FMbx.js';
|
|
6
|
+
import { b as RunCampaignOptions, C as CampaignStorage } from './run-improvement-loop-CNqQckTj.js';
|
|
7
7
|
import { HostedClient, TraceSpanEvent } from './hosted/index.js';
|
|
8
8
|
|
|
9
9
|
/**
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { A as Analyst, a as AnalystContext, b as AnalystRunSummary, c as AnalystFinding, d as AnalystRunResult, C as ChatClient, e as AnalystRunInputs, f as AnalystRunEvent } from './types-
|
|
1
|
+
import { A as Analyst, a as AnalystContext, b as AnalystRunSummary, c as AnalystFinding, d as AnalystRunResult, C as ChatClient, e as AnalystRunInputs, f as AnalystRunEvent } from './types-Cu3u_x59.js';
|
|
2
2
|
|
|
3
3
|
/**
|
|
4
4
|
* AnalystRegistry — orchestrate N analysts against one run.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { D as DatasetSplit, c as DatasetManifest, a as DatasetScenario } from './dataset-B2kL-fSM.js';
|
|
2
|
-
import { m as GateDecision } from './summary-report-
|
|
3
|
-
import { R as RunRecord,
|
|
2
|
+
import { m as GateDecision } from './summary-report-Db0dDSWP.js';
|
|
3
|
+
import { R as RunRecord, a as RunSplitTag } from './run-record-De9VarXR.js';
|
|
4
4
|
|
|
5
5
|
/**
|
|
6
6
|
* Release confidence gate.
|
package/dist/reporting.d.ts
CHANGED
|
@@ -1,9 +1,9 @@
|
|
|
1
|
-
export { R as RubricOutcomePair, a as RubricPredictiveValidityInput, b as RubricPredictiveValidityReport, c as RubricRanking, r as rubricPredictiveValidity } from './rubric-predictive-validity-
|
|
2
|
-
export { B as BootstrapOptions, a as BootstrapResult, J as JudgeReplayGateArgs, R as ReleaseConfidenceAxis, b as ReleaseConfidenceAxisName, c as ReleaseConfidenceInput, d as ReleaseConfidenceIssue, e as ReleaseConfidenceMetrics, f as ReleaseConfidenceScorecard, g as ReleaseConfidenceStatus, h as ReleaseConfidenceThresholds, i as ReleaseTraceEvidence, j as RenderReleaseReportOptions, V as Verdict, k as assertReleaseConfidence, l as bootstrapCi, m as evaluateReleaseConfidence, n as judgeReplayGate, r as renderReleaseReport } from './release-report-
|
|
1
|
+
export { R as RubricOutcomePair, a as RubricPredictiveValidityInput, b as RubricPredictiveValidityReport, c as RubricRanking, r as rubricPredictiveValidity } from './rubric-predictive-validity-CnEl9Jc8.js';
|
|
2
|
+
export { B as BootstrapOptions, a as BootstrapResult, J as JudgeReplayGateArgs, R as ReleaseConfidenceAxis, b as ReleaseConfidenceAxisName, c as ReleaseConfidenceInput, d as ReleaseConfidenceIssue, e as ReleaseConfidenceMetrics, f as ReleaseConfidenceScorecard, g as ReleaseConfidenceStatus, h as ReleaseConfidenceThresholds, i as ReleaseTraceEvidence, j as RenderReleaseReportOptions, V as Verdict, k as assertReleaseConfidence, l as bootstrapCi, m as evaluateReleaseConfidence, n as judgeReplayGate, r as renderReleaseReport } from './release-report-hlNtD12q.js';
|
|
3
3
|
export { I as InterimReleaseConfidence, a as InterimReleaseConfidenceInput, P as PairedEvalueOptions, b as PairedEvalueSequence, c as PairedEvalueStep, S as SequentialDecision, e as evaluateInterimReleaseConfidence, p as pairedEvalueSequence } from './sequential-5iSVfzl2.js';
|
|
4
|
-
export { P as PairedBootstrapOptions, a as PairedBootstrapResult, b as benjaminiHochberg, p as pairedBootstrap, w as wilcoxonSignedRank } from './statistics-
|
|
5
|
-
export { G as GainDistributionBin, a as GainDistributionFigureSpec, b as GainDistributionOptions, P as ParetoFigureSpec, c as ParetoPoint, R as RESEARCH_REPORT_HARD_PAIR_FLOOR, d as ResearchReport, e as ResearchReportCandidate, f as ResearchReportDecision, g as ResearchReportMethodology, h as ResearchReportOptions, i as ResearchReportRecommendation, S as SummaryTable, j as SummaryTableOptions, k as SummaryTableRow, l as gainHistogram, p as paretoChart, r as researchReport, s as summaryTable } from './summary-report-
|
|
6
|
-
import './run-record-
|
|
4
|
+
export { P as PairedBootstrapOptions, a as PairedBootstrapResult, b as benjaminiHochberg, p as pairedBootstrap, w as wilcoxonSignedRank } from './statistics-CnC1FMbx.js';
|
|
5
|
+
export { G as GainDistributionBin, a as GainDistributionFigureSpec, b as GainDistributionOptions, P as ParetoFigureSpec, c as ParetoPoint, R as RESEARCH_REPORT_HARD_PAIR_FLOOR, d as ResearchReport, e as ResearchReportCandidate, f as ResearchReportDecision, g as ResearchReportMethodology, h as ResearchReportOptions, i as ResearchReportRecommendation, S as SummaryTable, j as SummaryTableOptions, k as SummaryTableRow, l as gainHistogram, p as paretoChart, r as researchReport, s as summaryTable } from './summary-report-Db0dDSWP.js';
|
|
6
|
+
import './run-record-De9VarXR.js';
|
|
7
7
|
import './errors-Dwqw-T_m.js';
|
|
8
8
|
import './schema-m0gsnbt3.js';
|
|
9
9
|
import './outcome-store-rnXLEqSn.js';
|
package/dist/reporting.js
CHANGED
|
@@ -4,7 +4,7 @@ import {
|
|
|
4
4
|
evaluateReleaseConfidence,
|
|
5
5
|
judgeReplayGate,
|
|
6
6
|
renderReleaseReport
|
|
7
|
-
} from "./chunk-
|
|
7
|
+
} from "./chunk-FZWAFVAA.js";
|
|
8
8
|
import {
|
|
9
9
|
rubricPredictiveValidity
|
|
10
10
|
} from "./chunk-YRZ4M5GS.js";
|
|
@@ -18,12 +18,12 @@ import {
|
|
|
18
18
|
paretoChart,
|
|
19
19
|
researchReport,
|
|
20
20
|
summaryTable
|
|
21
|
-
} from "./chunk-
|
|
21
|
+
} from "./chunk-Z7VFTS2J.js";
|
|
22
22
|
import {
|
|
23
23
|
benjaminiHochberg,
|
|
24
24
|
pairedBootstrap,
|
|
25
25
|
wilcoxonSignedRank
|
|
26
|
-
} from "./chunk-
|
|
26
|
+
} from "./chunk-IDVBLYCY.js";
|
|
27
27
|
import "./chunk-VSMTAMNK.js";
|
|
28
28
|
import "./chunk-3BFEG2F6.js";
|
|
29
29
|
import "./chunk-PZ5AY32C.js";
|
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
import {
|
|
2
|
-
import { L as LlmClientOptions, a as LlmRouteRequirements } from './llm-client-
|
|
3
|
-
import { h as ResearchReportOptions, d as ResearchReport, m as GateDecision } from './summary-report-
|
|
1
|
+
import { a as RunSplitTag, b as RunTokenUsage, c as RunJudgeMetadata, J as JudgeScoresRecord, A as AgentProfileCell, d as AgentProfileCellInput, R as RunRecord } from './run-record-De9VarXR.js';
|
|
2
|
+
import { L as LlmClientOptions, a as LlmRouteRequirements } from './llm-client-CuUg2Mn3.js';
|
|
3
|
+
import { h as ResearchReportOptions, d as ResearchReport, m as GateDecision } from './summary-report-Db0dDSWP.js';
|
|
4
4
|
import { T as TraceEmitter, R as RunCompleteHook } from './emitter-DEZwY14K.js';
|
|
5
5
|
import { R as RunIntegrityExpectations, a as RunIntegrityReport } from './integrity-CJzrpUua.js';
|
|
6
6
|
import { R as RawProviderSink } from './raw-provider-sink-C46HDghv.js';
|
package/dist/rl.d.ts
CHANGED
|
@@ -1,19 +1,19 @@
|
|
|
1
1
|
export { O as OffPolicyEstimate, a as OffPolicyOptions, b as OffPolicyTrajectory, d as doublyRobust, i as inverseProbabilityWeighting, o as offPolicyEstimateAll, s as selfNormalizedImportanceWeighting } from './off-policy-DiwuKKg7.js';
|
|
2
|
-
import { R as RunRecord,
|
|
3
|
-
import { g as CampaignResult } from './types-
|
|
2
|
+
import { R as RunRecord, a as RunSplitTag } from './run-record-De9VarXR.js';
|
|
3
|
+
import { g as CampaignResult } from './types-D7lLRYe9.js';
|
|
4
4
|
import { a as VerificationReport } from './multi-layer-verifier-DlWCXuxL.js';
|
|
5
5
|
import { S as Span } from './schema-m0gsnbt3.js';
|
|
6
6
|
import { T as TraceStore } from './store-CKUAgsJz.js';
|
|
7
7
|
import { b as OutcomeStore } from './outcome-store-rnXLEqSn.js';
|
|
8
8
|
export { D as DeploymentOutcome, F as FileSystemOutcomeStore, a as FileSystemOutcomeStoreOptions, I as InMemoryOutcomeStore } from './outcome-store-rnXLEqSn.js';
|
|
9
|
-
import { b as RubricPredictiveValidityReport } from './rubric-predictive-validity-
|
|
10
|
-
import { R as Researcher, F as FailureMode, S as SteeringChange, E as ExperimentPlan, a as ExperimentResult, b as EvalCampaignResult, c as EvalCampaignOptions } from './researcher-
|
|
11
|
-
export { r as runEvalCampaign } from './researcher-
|
|
9
|
+
import { b as RubricPredictiveValidityReport } from './rubric-predictive-validity-CnEl9Jc8.js';
|
|
10
|
+
import { R as Researcher, F as FailureMode, S as SteeringChange, E as ExperimentPlan, a as ExperimentResult, b as EvalCampaignResult, c as EvalCampaignOptions } from './researcher-BLPHBbNV.js';
|
|
11
|
+
export { r as runEvalCampaign } from './researcher-BLPHBbNV.js';
|
|
12
12
|
import { I as InterimReleaseConfidence } from './sequential-5iSVfzl2.js';
|
|
13
13
|
import './errors-Dwqw-T_m.js';
|
|
14
|
-
import './llm-client-
|
|
14
|
+
import './llm-client-CuUg2Mn3.js';
|
|
15
15
|
import './raw-provider-sink-C46HDghv.js';
|
|
16
|
-
import './summary-report-
|
|
16
|
+
import './summary-report-Db0dDSWP.js';
|
|
17
17
|
import './failure-cluster-CL7IVgkJ.js';
|
|
18
18
|
import './emitter-DEZwY14K.js';
|
|
19
19
|
import './integrity-CJzrpUua.js';
|
package/dist/rl.js
CHANGED
|
@@ -16,19 +16,19 @@ import {
|
|
|
16
16
|
} from "./chunk-3RF76KTD.js";
|
|
17
17
|
import {
|
|
18
18
|
runEvalCampaign
|
|
19
|
-
} from "./chunk-
|
|
20
|
-
import "./chunk-
|
|
19
|
+
} from "./chunk-WJL2NJXN.js";
|
|
20
|
+
import "./chunk-CVVHBFGN.js";
|
|
21
21
|
import {
|
|
22
22
|
rubricPredictiveValidity
|
|
23
23
|
} from "./chunk-YRZ4M5GS.js";
|
|
24
24
|
import {
|
|
25
25
|
evaluateInterimReleaseConfidence
|
|
26
26
|
} from "./chunk-MAZ26DC7.js";
|
|
27
|
-
import "./chunk-
|
|
27
|
+
import "./chunk-Z7VFTS2J.js";
|
|
28
28
|
import {
|
|
29
29
|
benjaminiHochberg,
|
|
30
30
|
wilcoxonSignedRank
|
|
31
|
-
} from "./chunk-
|
|
31
|
+
} from "./chunk-IDVBLYCY.js";
|
|
32
32
|
import "./chunk-SBCB6VZY.js";
|
|
33
33
|
import "./chunk-PC4UYEBM.js";
|
|
34
34
|
import "./chunk-KWRRMR3J.js";
|
|
@@ -1,10 +1,10 @@
|
|
|
1
1
|
import {
|
|
2
2
|
runCampaign
|
|
3
|
-
} from "./chunk-
|
|
4
|
-
import "./chunk-
|
|
3
|
+
} from "./chunk-ZZ2HOPME.js";
|
|
4
|
+
import "./chunk-IDVBLYCY.js";
|
|
5
5
|
import "./chunk-3BFEG2F6.js";
|
|
6
6
|
import "./chunk-PZ5AY32C.js";
|
|
7
7
|
export {
|
|
8
8
|
runCampaign
|
|
9
9
|
};
|
|
10
|
-
//# sourceMappingURL=run-campaign-
|
|
10
|
+
//# sourceMappingURL=run-campaign-4Y5V5CN3.js.map
|
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
import { L as LlmClientOptions } from './llm-client-
|
|
2
|
-
import { I as ImprovementDriver, S as Scenario, g as CampaignResult, k as GateResult, D as DispatchFn, a as JudgeConfig, L as LabeledScenarioStore, h as CampaignTraceWriter, m as GenerationRecord, M as MutableSurface, P as ParetoParent, G as Gate } from './types-
|
|
1
|
+
import { L as LlmClientOptions } from './llm-client-CuUg2Mn3.js';
|
|
2
|
+
import { I as ImprovementDriver, S as Scenario, g as CampaignResult, k as GateResult, D as DispatchFn, a as JudgeConfig, L as LabeledScenarioStore, h as CampaignTraceWriter, m as GenerationRecord, M as MutableSurface, P as ParetoParent, G as Gate } from './types-D7lLRYe9.js';
|
|
3
3
|
|
|
4
4
|
/**
|
|
5
5
|
* @experimental
|
|
@@ -424,4 +424,4 @@ interface RunImprovementLoopResult<TArtifact, TScenario extends Scenario> extend
|
|
|
424
424
|
declare function runImprovementLoop<TScenario extends Scenario, TArtifact>(opts: RunImprovementLoopOptions<TScenario, TArtifact>): Promise<RunImprovementLoopResult<TArtifact, TScenario>>;
|
|
425
425
|
declare function defaultRenderDiff(winnerSurface: MutableSurface, baselineSurface: MutableSurface): string;
|
|
426
426
|
|
|
427
|
-
export { type CampaignStorage as C, type GepaDriverOptions as G, type OpenAutoPrOptions as O, type
|
|
427
|
+
export { type CampaignStorage as C, type GepaDriverOptions as G, type OpenAutoPrOptions as O, type RunOptimizationOptions as R, type RunImprovementLoopResult as a, type RunCampaignOptions as b, type RunImprovementLoopOptions as c, runImprovementLoop as d, type GepaDriverConstraints as e, fsCampaignStorage as f, gepaDriver as g, type OpenAutoPrResult as h, inMemoryCampaignStorage as i, type RunOptimizationResult as j, countSentenceEdits as k, defaultRenderDiff as l, extractH2Sections as m, runOptimization as n, openAutoPr as o, runCampaign as r, surfaceHash as s };
|
|
@@ -315,4 +315,4 @@ declare function parseRunRecordSafe(input: unknown): {
|
|
|
315
315
|
/** Round-trip helper — `JSON.parse(JSON.stringify(record))` then validate. */
|
|
316
316
|
declare function roundTripRunRecord(record: RunRecord): RunRecord;
|
|
317
317
|
|
|
318
|
-
export { type AgentProfileCell as A, validateAgentProfileCell as B, validateRunRecord as C, verifyAgentProfileCell as D, type JudgeScoresRecord as J, type RunRecord as R, type SandboxAgentProfileLike as S, type
|
|
318
|
+
export { type AgentProfileCell as A, validateAgentProfileCell as B, validateRunRecord as C, verifyAgentProfileCell as D, type JudgeScoresRecord as J, type RunRecord as R, type SandboxAgentProfileLike as S, type RunSplitTag as a, type RunTokenUsage as b, type RunJudgeMetadata as c, type AgentProfileCellInput as d, AGENT_PROFILE_KINDS as e, type AgentProfileCellSchemaVersion as f, AgentProfileCellValidationError as g, type AgentProfileDimensionValue as h, type AgentProfileHarness as i, type AgentProfileJson as j, type AgentProfileKind as k, type AgentProfileSource as l, type AgentProfileSourceInput as m, type RunOutcome as n, RunRecordValidationError as o, agentProfileCellHashMaterial as p, agentProfileCellKey as q, assertRunAgentProfileCell as r, buildAgentProfileCell as s, buildSandboxAgentProfileCell as t, groupRunsByAgentProfileCell as u, isRunRecord as v, parseRunRecordSafe as w, requireAgentProfileCell as x, roundTripRunRecord as y, toAgentProfileJson as z };
|
|
@@ -1,10 +1,10 @@
|
|
|
1
1
|
import { AxAIService } from '@ax-llm/ax';
|
|
2
|
-
import { c as TraceAnalystKindSpec } from './kind-factory-
|
|
3
|
-
import { b as AnalystRegistryOptions, a as AnalystRegistry } from './registry-
|
|
2
|
+
import { c as TraceAnalystKindSpec } from './kind-factory-CVecZZG_.js';
|
|
3
|
+
import { b as AnalystRegistryOptions, a as AnalystRegistry } from './registry-DrEQ3Luj.js';
|
|
4
4
|
import { z } from 'zod';
|
|
5
|
-
import { c as AnalystFinding, A as Analyst, a as AnalystContext } from './types-
|
|
6
|
-
import { a as TraceAnalystSpan } from './store-
|
|
7
|
-
import { L as LlmClientOptions } from './llm-client-
|
|
5
|
+
import { c as AnalystFinding, A as Analyst, a as AnalystContext } from './types-Cu3u_x59.js';
|
|
6
|
+
import { a as TraceAnalystSpan } from './store-C1YxJDEK.js';
|
|
7
|
+
import { L as LlmClientOptions } from './llm-client-CuUg2Mn3.js';
|
|
8
8
|
import { S as Severity } from './multi-layer-verifier-DlWCXuxL.js';
|
|
9
9
|
|
|
10
10
|
interface CreateAnalystAiConfig {
|
|
@@ -615,7 +615,7 @@ declare const DEFAULT_COMPLEXITY_WEIGHTS: Record<ConceptComplexity, number>;
|
|
|
615
615
|
interface SemanticConceptJudgeOptions {
|
|
616
616
|
/** Model id to call. Default 'claude-sonnet-4-6' via agent-eval defaults. */
|
|
617
617
|
model?: string;
|
|
618
|
-
/** Per-call timeout. Default
|
|
618
|
+
/** Per-call timeout. Default 300s. */
|
|
619
619
|
timeoutMs?: number;
|
|
620
620
|
/** Pipeline budget for the prompt (source blob truncation). Default 45000. */
|
|
621
621
|
maxSourceChars?: number;
|
|
@@ -1,13 +1,9 @@
|
|
|
1
1
|
import { C as ContinuousAgreementOptions, a as ContinuousAgreement } from './judge-calibration-DilmB3Ml.js';
|
|
2
2
|
import { J as JudgeScore } from './types-Croy5h7V.js';
|
|
3
3
|
|
|
4
|
-
/**
|
|
5
|
-
*
|
|
6
|
-
|
|
7
|
-
* already use inverted scoring in the prompt (10 = no hallucination),
|
|
8
|
-
* but this function ensures consistency if raw scores leak through.
|
|
9
|
-
*/
|
|
10
|
-
declare function normalizeScores(scores: JudgeScore[]): JudgeScore[];
|
|
4
|
+
/** Identity: dimensions already follow "higher = better" by prompt convention
|
|
5
|
+
* (inverted dims like hallucination are scored 10 = best at the source). */
|
|
6
|
+
declare const normalizeScores: (scores: JudgeScore[]) => JudgeScore[];
|
|
11
7
|
/** Weighted mean — falls back to uniform weights when omitted */
|
|
12
8
|
declare function weightedMean(scores: {
|
|
13
9
|
score: number;
|
|
@@ -245,4 +245,4 @@ interface TraceAnalysisStore {
|
|
|
245
245
|
}): Promise<SearchSpanResult>;
|
|
246
246
|
}
|
|
247
247
|
|
|
248
|
-
export { DEFAULT_TRACE_ANALYST_BUDGETS as D, type QueryTracesPage as Q, type SearchSpanResult as S, type TraceAnalysisStore as T, type ViewSpansResult as V, type TraceAnalystSpan as a, type DatasetOverview as b, type SearchTraceResult as c, type SpanMatchRecord as d, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX as e, type TraceAnalystByteBudgets as f, type TraceAnalystFilters as g, type TraceAnalystSpanKind as h, type TraceAnalystSpanStatus as i, type TraceAnalystTraceSummary as j, type ViewTraceOversized as k, type ViewTraceResult as l };
|
|
248
|
+
export { DEFAULT_TRACE_ANALYST_BUDGETS as D, type ErrorCluster as E, type QueryTracesPage as Q, type SearchSpanResult as S, type TraceAnalysisStore as T, type ViewSpansResult as V, type TraceAnalystSpan as a, type DatasetOverview as b, type SearchTraceResult as c, type SpanMatchRecord as d, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX as e, type TraceAnalystByteBudgets as f, type TraceAnalystFilters as g, type TraceAnalystSpanKind as h, type TraceAnalystSpanStatus as i, type TraceAnalystTraceSummary as j, type ViewTraceOversized as k, type ViewTraceResult as l };
|
package/dist/traces.d.ts
CHANGED
|
@@ -10,11 +10,11 @@ export { a as aggregateLlm, b as argHash, g as groupBy, j as judgeSpans, l as ll
|
|
|
10
10
|
export { D as DEFAULT_REDACTION_RULES, b as REDACTION_VERSION, a as RedactionReport, R as RedactionRule, r as redactString, c as redactValue } from './redact-B40YG2M_.js';
|
|
11
11
|
import { R as Run } from './schema-m0gsnbt3.js';
|
|
12
12
|
export { A as Artifact, B as BudgetLedgerEntry, h as BudgetSpec, E as EventKind, i as FAILURE_CLASSES, F as FailureClass, G as GenericSpan, J as JudgeSpan, L as LlmSpan, M as Message, d as RetrievalSpan, g as RunLayer, b as RunOutcome, f as RunStatus, e as SandboxSpan, S as Span, j as SpanBase, c as SpanKind, k as SpanStatus, l as TRACE_SCHEMA_VERSION, T as ToolSpan, a as TraceEvent, m as isJudgeSpan, n as isLlmSpan, o as isRetrievalSpan, p as isSandboxSpan, q as isToolSpan } from './schema-m0gsnbt3.js';
|
|
13
|
-
import { A as AnalyzeTracesOptions, b as AnalyzeTracesResult } from './analyst-
|
|
14
|
-
export { a as AnalyzeTracesInput, c as AnalyzeTracesTurnSnapshot, d as analyzeTraces } from './analyst-
|
|
15
|
-
import { h as TraceAnalystSpanKind, i as TraceAnalystSpanStatus, T as TraceAnalysisStore, g as TraceAnalystFilters, b as DatasetOverview, Q as QueryTracesPage, l as ViewTraceResult, V as ViewSpansResult, c as SearchTraceResult, S as SearchSpanResult } from './store-
|
|
16
|
-
export { D as DEFAULT_TRACE_ANALYST_BUDGETS, d as SpanMatchRecord, e as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, f as TraceAnalystByteBudgets, a as TraceAnalystSpan, j as TraceAnalystTraceSummary, k as ViewTraceOversized } from './store-
|
|
17
|
-
import {
|
|
13
|
+
import { A as AnalyzeTracesOptions, b as AnalyzeTracesResult } from './analyst-C8HHvfJp.js';
|
|
14
|
+
export { a as AnalyzeTracesInput, c as AnalyzeTracesTurnSnapshot, d as analyzeTraces } from './analyst-C8HHvfJp.js';
|
|
15
|
+
import { h as TraceAnalystSpanKind, i as TraceAnalystSpanStatus, T as TraceAnalysisStore, g as TraceAnalystFilters, b as DatasetOverview, Q as QueryTracesPage, l as ViewTraceResult, V as ViewSpansResult, c as SearchTraceResult, S as SearchSpanResult } from './store-C1YxJDEK.js';
|
|
16
|
+
export { D as DEFAULT_TRACE_ANALYST_BUDGETS, E as ErrorCluster, d as SpanMatchRecord, e as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, f as TraceAnalystByteBudgets, a as TraceAnalystSpan, j as TraceAnalystTraceSummary, k as ViewTraceOversized } from './store-C1YxJDEK.js';
|
|
17
|
+
import { a as RunSplitTag, b as RunTokenUsage, R as RunRecord } from './run-record-De9VarXR.js';
|
|
18
18
|
import { AxFunction } from '@ax-llm/ax';
|
|
19
19
|
|
|
20
20
|
/**
|
|
@@ -1,7 +1,7 @@
|
|
|
1
|
-
import { R as RunRecord } from './run-record-
|
|
2
|
-
import { T as TraceAnalysisStore } from './store-
|
|
1
|
+
import { R as RunRecord } from './run-record-De9VarXR.js';
|
|
2
|
+
import { T as TraceAnalysisStore } from './store-C1YxJDEK.js';
|
|
3
3
|
import { a as JudgeInput } from './types-Croy5h7V.js';
|
|
4
|
-
import { b as LlmCallRequest, c as LlmCallResult } from './llm-client-
|
|
4
|
+
import { b as LlmCallRequest, c as LlmCallResult } from './llm-client-CuUg2Mn3.js';
|
|
5
5
|
|
|
6
6
|
/**
|
|
7
7
|
* ChatClient — the single LLM abstraction analysts call.
|
package/dist/wire/index.js
CHANGED
|
@@ -34,8 +34,8 @@ import {
|
|
|
34
34
|
runRpcOnce,
|
|
35
35
|
startServer,
|
|
36
36
|
startServerAsync
|
|
37
|
-
} from "../chunk-
|
|
38
|
-
import "../chunk-
|
|
37
|
+
} from "../chunk-QS3RBQPI.js";
|
|
38
|
+
import "../chunk-CVVHBFGN.js";
|
|
39
39
|
import "../chunk-PC4UYEBM.js";
|
|
40
40
|
import "../chunk-3BFEG2F6.js";
|
|
41
41
|
import "../chunk-PZ5AY32C.js";
|
package/dist/workflow/index.d.ts
CHANGED
|
@@ -1,24 +1,24 @@
|
|
|
1
1
|
import { W as WorkflowTopology } from '../harness-optimizer-EnEnQPsr.js';
|
|
2
|
-
import {
|
|
3
|
-
import { c as AnalystFinding, h as AnalystSeverity, E as EvidenceRef } from '../types-
|
|
4
|
-
import { F as FailureClusterInsight } from '../insight-report-
|
|
2
|
+
import { a as RunSplitTag, b as RunTokenUsage, R as RunRecord } from '../run-record-De9VarXR.js';
|
|
3
|
+
import { c as AnalystFinding, h as AnalystSeverity, E as EvidenceRef } from '../types-Cu3u_x59.js';
|
|
4
|
+
import { F as FailureClusterInsight } from '../insight-report-3ADTfClO.js';
|
|
5
5
|
import { a as VerificationReport, L as LayerResult } from '../multi-layer-verifier-DlWCXuxL.js';
|
|
6
6
|
import { F as FailureClusterReport } from '../failure-cluster-CL7IVgkJ.js';
|
|
7
7
|
import { R as RedactionRule, a as RedactionReport } from '../redact-B40YG2M_.js';
|
|
8
8
|
import { D as DatasetSplit } from '../dataset-B2kL-fSM.js';
|
|
9
9
|
import { a as FeedbackTrajectory } from '../feedback-trajectory-B3rErRsh.js';
|
|
10
|
-
import { a as PairedBootstrapResult } from '../statistics-
|
|
10
|
+
import { a as PairedBootstrapResult } from '../statistics-CnC1FMbx.js';
|
|
11
11
|
import '../pareto-E-pembql.js';
|
|
12
12
|
import '../run-critic-BAIjX99r.js';
|
|
13
13
|
import '../schema-m0gsnbt3.js';
|
|
14
14
|
import '../store-CKUAgsJz.js';
|
|
15
15
|
import '../errors-Dwqw-T_m.js';
|
|
16
|
-
import '../store-
|
|
16
|
+
import '../store-C1YxJDEK.js';
|
|
17
17
|
import '../types-Croy5h7V.js';
|
|
18
18
|
import '@tangle-network/tcloud';
|
|
19
|
-
import '../llm-client-
|
|
19
|
+
import '../llm-client-CuUg2Mn3.js';
|
|
20
20
|
import '../raw-provider-sink-C46HDghv.js';
|
|
21
|
-
import '../summary-report-
|
|
21
|
+
import '../summary-report-Db0dDSWP.js';
|
|
22
22
|
import '../judge-calibration-DilmB3Ml.js';
|
|
23
23
|
import '../control-runtime-DuFBYg7A.js';
|
|
24
24
|
import '../emitter-DEZwY14K.js';
|
package/dist/workflow/index.js
CHANGED
package/docs/concepts.md
CHANGED
|
@@ -182,6 +182,7 @@ release decision.
|
|
|
182
182
|
|
|
183
183
|
## Where to go next
|
|
184
184
|
|
|
185
|
+
- **Confused by "GEPA / HALO / trace analysis / drivers everywhere"?** → [self-improvement-map.md](./self-improvement-map.md) — one loop, four roles, the seven-driver catalog (production vs bench-only), and why `gepa-refine` is the same loop on a test bench.
|
|
185
186
|
- **Need the layman feature map?** → [feature-guide.md](./feature-guide.md) — what each primitive does, when to use it, integration patterns, and guardrails.
|
|
186
187
|
- **Just want to score a string against a rubric?** → [wire-protocol.md](./wire-protocol.md) — HTTP/RPC interface, pluggable from any language.
|
|
187
188
|
- **Need a reusable driver/worker/evaluator loop?** → [control-runtime.md](./control-runtime.md) — generic runtime plus coding, browser, computer-use, and research integration patterns.
|
|
@@ -211,14 +211,16 @@ The API should report uncertainty and support problems, not hide them behind a s
|
|
|
211
211
|
|
|
212
212
|
### Q3 2026 - Phase 0: Tracking, Corpus, and Decision Inventory
|
|
213
213
|
|
|
214
|
-
- [
|
|
215
|
-
- [
|
|
216
|
-
- [
|
|
214
|
+
- [x] Keep this document current as the research tracker.
|
|
215
|
+
- [x] Create the first decision inventory over existing code-agent traces: failure-recovery, tool-select, and graph-completion.
|
|
216
|
+
- [x] Define first-pass trace extraction rules for Codex, Claude Code, OpenCode, Kimi Code, and Pi/PiGraph-shaped local traces.
|
|
217
217
|
- [ ] Define the minimum event fields needed from runtime and knowledge packages.
|
|
218
|
-
- [
|
|
218
|
+
- [x] Build the first replay-corpus adapter from existing `RunRecord` rows plus local code-agent session traces.
|
|
219
|
+
- [x] Add a one-call code-agent evidence corpus helper that joins session intake, decision extraction, and the research evidence gate.
|
|
219
220
|
- [ ] Label at least 200 decision points with outcome, cost, and whether the action was retrospectively correct.
|
|
220
|
-
- [
|
|
221
|
-
- [
|
|
221
|
+
- [x] Add first support diagnostics: missing outcomes, missing behavior/target propensities, and insufficient target support.
|
|
222
|
+
- [x] Decide the first decision kind for Phase 1 dogfooding: failure recovery after failed tool/patch actions.
|
|
223
|
+
- [x] Add a small research evidence gate that classifies selective vs counterfactual claim support.
|
|
222
224
|
|
|
223
225
|
Completion criteria:
|
|
224
226
|
|
|
@@ -228,6 +230,15 @@ Completion criteria:
|
|
|
228
230
|
- [ ] Backend and capture integrity are checked before analysis.
|
|
229
231
|
- [ ] No producerless schema fields are introduced.
|
|
230
232
|
- [ ] One baseline policy is recorded for every decision kind under study.
|
|
233
|
+
- [ ] A generated `BeliefDecisionResearchEvidencePacket` says `supported` for the intended claim scope.
|
|
234
|
+
|
|
235
|
+
Status on 2026-06-05: the experimental implementation exists in `src/belief-state/code-agent-corpus.ts` and `src/belief-state/research-evidence.ts`, with coverage in `src/belief-state/code-agent-corpus.test.ts` and `src/belief-state/research-evidence.test.ts`. A local smoke after build joined 33 private code-agent sessions to 33 `RunRecord`s and emitted 13,137 decision rows across Codex, Claude Code, Kimi Code, OpenCode, and PiGraph-shaped traces. This closes the infrastructure part of Phase 0, but not the empirical proof gate: the next corpus run still has to add split metadata, integrity checks, retrospective labels, and a recorded baseline per target before the work can claim Phase 0 completion. Missing behavior/target propensities now block counterfactual claims while still allowing selective-only claims to be evaluated.
|
|
236
|
+
|
|
237
|
+
Follow-up local smoke on 2026-06-05 over 50 recent Codex JSONL sessions under 20 MB produced 50 `RunRecord`s and 6,770 decision points: 5,610 tool-selection rows and 1,160 failure-recovery rows, with full outcome/confidence coverage and no propensity support. The default confidence-threshold policy did not clear the selective utility gate on failure recovery (`ci.lower = -0.1159` at threshold `0.6`), so the current result supports the extraction/evidence pipeline, not the belief-policy claim; the next empirical step is real logged confidence/propensity or retrospective labels, not more heuristic confidence tuning.
|
|
238
|
+
|
|
239
|
+
Runtime hook bridge on 2026-06-05: `src/belief-state/runtime-hooks.ts` now converts `agent-runtime` decision hooks into outcome-blind shadow-probe inputs, attaches matching lifecycle hook events as probe evidence, and only converts them into full `BeliefDecisionPoint` rows when the observed action is supplied. This keeps `agent-eval` trace/analysis-only while letting the runtime emit producer-backed decision boundaries and context for the next experiment.
|
|
240
|
+
|
|
241
|
+
Taxonomy tightening on 2026-06-05: `src/belief-state/types.ts` now exports stable decision kinds, evidence sources, evidence quality labels, evaluation criteria, and reason codes. The intent is to make future dashboards and paper artifacts aggregate by stable IDs (`calibration`, `ope-support`, `memory-health`, `surface-attribution`, `promotion`, etc.) instead of parsing prose diagnostics.
|
|
231
242
|
|
|
232
243
|
### Q4 2026 - Phase 1: Selective Prediction and Abstention
|
|
233
244
|
|
|
@@ -236,6 +247,7 @@ Completion criteria:
|
|
|
236
247
|
- [ ] Compare baseline policy vs selective policy on holdout.
|
|
237
248
|
- [ ] Add cost-aware utility: quality lift minus verification/ask/retry cost.
|
|
238
249
|
- [ ] Add report rows into `InsightReport` or an experimental research report.
|
|
250
|
+
- [ ] Export the real corpus packet into the paper artifact instead of hand-copying metrics.
|
|
239
251
|
- [ ] Run negative controls: shuffled confidence, random abstention, always-verify, never-verify.
|
|
240
252
|
- [ ] Pre-register thresholds before holdout.
|
|
241
253
|
|
|
@@ -364,6 +376,21 @@ Do not call belief-state work "done" until these are true:
|
|
|
364
376
|
- [ ] No runtime ownership boundary is crossed from `agent-eval`.
|
|
365
377
|
- [ ] Negative results are recorded instead of hidden.
|
|
366
378
|
|
|
379
|
+
Stable criterion IDs:
|
|
380
|
+
|
|
381
|
+
- `capture-integrity`
|
|
382
|
+
- `decision-completeness`
|
|
383
|
+
- `evidence-quality`
|
|
384
|
+
- `outcome-quality`
|
|
385
|
+
- `calibration`
|
|
386
|
+
- `accepted-region-risk`
|
|
387
|
+
- `policy-value`
|
|
388
|
+
- `ope-support`
|
|
389
|
+
- `memory-health`
|
|
390
|
+
- `surface-attribution`
|
|
391
|
+
- `generalization`
|
|
392
|
+
- `promotion`
|
|
393
|
+
|
|
367
394
|
## Kill Criteria
|
|
368
395
|
|
|
369
396
|
Stop or pivot if any of these persist for two consecutive phases:
|
|
@@ -408,12 +435,14 @@ The most succinct integration is an experimental `src/belief-state/` module that
|
|
|
408
435
|
|
|
409
436
|
| File | Purpose | Notes |
|
|
410
437
|
|---|---|---|
|
|
411
|
-
| `src/belief-state/types.ts` | Defines `BeliefDecisionPoint`, `BeliefDecisionKind`, `
|
|
438
|
+
| `src/belief-state/types.ts` | Defines `BeliefDecisionPoint`, `BeliefDecisionKind`, `BeliefDecisionOutcome`, `BeliefEvidenceRef`, `BeliefPolicyEvaluationReport`, support diagnostics, stable criteria, and reason codes. | No runtime dependency. Keep taxonomy compact and producer-backed. |
|
|
412
439
|
| `src/belief-state/extract.ts` | Extracts decision points from `TraceStore` runs/spans/events. | Structural parsing only. Unknown events are skipped with diagnostics. |
|
|
413
440
|
| `src/belief-state/selective.ts` | Evaluates continue/verify/ask/retry/stop policies against observed outcomes. | Computes coverage, accepted-error rate, rejected-action lift, cost-adjusted utility. |
|
|
414
441
|
| `src/belief-state/calibration.ts` | Computes confidence calibration for decision predictions. | Calls shared `calibrationFromPairs()` once added. |
|
|
415
442
|
| `src/belief-state/ope.ts` | Converts decision rows into `OffPolicyTrajectory[]` for an explicit named target policy and calls `offPolicyEstimateAll`. | Must report ESS and support mismatch; no silent value claims. |
|
|
416
443
|
| `src/belief-state/report.ts` | Orchestrates extraction + selective eval + calibration + OPE into one report. | Returns honest negative / need-more-data when unsupported. |
|
|
444
|
+
| `src/belief-state/code-agent-corpus.ts` | Converts local code-agent sessions into belief decision points, inventories targets, selects the first supported target, and runs the experimental policy report. | Supports Codex, Claude Code, OpenCode, Kimi Code, and Pi/PiGraph-shaped traces. Does not invent behavior or target propensities. |
|
|
445
|
+
| `src/belief-state/runtime-hooks.ts` | Bridges structurally typed `agent-runtime` decision hooks into shadow probes or completed belief decision rows. | No runtime dependency. Pre-action hooks do not fake `chosenAction`. |
|
|
417
446
|
| `src/belief-state/index.ts` | Experimental barrel for the module. | Keep out of root barrel and expose only through `./experimental/belief-state` while evidence gates are open. |
|
|
418
447
|
|
|
419
448
|
### Files to Change First
|
|
@@ -536,6 +565,9 @@ Promotion:
|
|
|
536
565
|
| `src/belief-state/calibration.test.ts` | ECE bins; equal-width/equal-frequency behavior; too-few-pairs returns unsupported. |
|
|
537
566
|
| `src/belief-state/ope.test.ts` | converts to `OffPolicyTrajectory`; explicit target policy required; invalid propensity disables OPE without throwing; low ESS support mismatch; estimator agreement surfaced. |
|
|
538
567
|
| `src/belief-state/report.test.ts` | full report status: `ship`, `hold`, `need_more_data`; recommendation cannot ship on OPE alone. |
|
|
568
|
+
| `src/belief-state/code-agent-corpus.test.ts` | extracts code-agent decision corpora across Codex, Claude Code, OpenCode, Kimi Code, and Pi/PiGraph-shaped traces; inventories targets; picks failure recovery first; holds when OPE propensities are absent. |
|
|
569
|
+
| `src/belief-state/runtime-hooks.test.ts` | converts runtime decision hooks to outcome-blind shadow probes; attaches matching lifecycle hook events as evidence; requires observed action for full belief rows; collector stays structurally compatible with runtime hooks. |
|
|
570
|
+
| `src/belief-state/types.test.ts` | guards the stable decision kinds, evidence sources/qualities, evaluation criteria, and reason-code taxonomy. |
|
|
539
571
|
| `src/meta-eval/calibration.test.ts` | existing `calibrationCurve()` still works after extracting pure helper. |
|
|
540
572
|
|
|
541
573
|
### Verification Commands
|