@tangle-network/agent-eval 0.80.0 → 0.82.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (95) hide show
  1. package/CHANGELOG.md +27 -0
  2. package/README.md +53 -152
  3. package/dist/adapters/http.d.ts +2 -2
  4. package/dist/adapters/langchain.d.ts +2 -2
  5. package/dist/adapters/otel.d.ts +4 -4
  6. package/dist/{agent-profile-aSEaJ9Pl.d.ts → agent-profile-D0PBIWlV.d.ts} +14 -2
  7. package/dist/analyst/index.d.ts +10 -10
  8. package/dist/analyst/index.js +3 -3
  9. package/dist/{analyst-t7zZS3TV.d.ts → analyst-C8HHvfJp.d.ts} +1 -1
  10. package/dist/belief-state/index.d.ts +344 -8
  11. package/dist/belief-state/index.js +1518 -142
  12. package/dist/belief-state/index.js.map +1 -1
  13. package/dist/benchmarks/index.d.ts +2 -2
  14. package/dist/campaign/index.d.ts +40 -120
  15. package/dist/campaign/index.js +129 -238
  16. package/dist/campaign/index.js.map +1 -1
  17. package/dist/{chunk-RPLZ4OIB.js → chunk-BABOZOSN.js} +7 -4
  18. package/dist/{chunk-RPLZ4OIB.js.map → chunk-BABOZOSN.js.map} +1 -1
  19. package/dist/{chunk-IHDHUN2X.js → chunk-CVVHBFGN.js} +3 -3
  20. package/dist/chunk-CVVHBFGN.js.map +1 -0
  21. package/dist/{chunk-B26KI423.js → chunk-FZWAFVAA.js} +2 -2
  22. package/dist/{chunk-ITBRCT73.js → chunk-IDVBLYCY.js} +2 -10
  23. package/dist/chunk-IDVBLYCY.js.map +1 -0
  24. package/dist/{chunk-5LVWPNS5.js → chunk-L5G7OUKD.js} +4 -4
  25. package/dist/{chunk-5LVWPNS5.js.map → chunk-L5G7OUKD.js.map} +1 -1
  26. package/dist/{chunk-GXHLRXDI.js → chunk-OTYQPHPL.js} +4 -4
  27. package/dist/{chunk-6REHLN5J.js → chunk-QS3RBQPI.js} +2 -2
  28. package/dist/{chunk-GWGO2K6Y.js → chunk-RBNA5AZT.js} +2 -2
  29. package/dist/chunk-S42AWHMP.js +697 -0
  30. package/dist/chunk-S42AWHMP.js.map +1 -0
  31. package/dist/chunk-VI2UW6B6.js +162 -0
  32. package/dist/chunk-VI2UW6B6.js.map +1 -0
  33. package/dist/{chunk-CF67I6QY.js → chunk-VIDQF3F5.js} +2 -2
  34. package/dist/{chunk-XXNIODOM.js → chunk-WJL2NJXN.js} +3 -3
  35. package/dist/{chunk-LB2UOI5F.js → chunk-YGYXHNAQ.js} +47 -160
  36. package/dist/chunk-YGYXHNAQ.js.map +1 -0
  37. package/dist/{chunk-KX6F6NCG.js → chunk-Z7VFTS2J.js} +2 -2
  38. package/dist/{chunk-ZPSKPT3V.js → chunk-ZZ2HOPME.js} +5 -2
  39. package/dist/chunk-ZZ2HOPME.js.map +1 -0
  40. package/dist/cli.js +2 -2
  41. package/dist/code-agent-session-BRXmavYv.d.ts +80 -0
  42. package/dist/contract/index.d.ts +45 -15
  43. package/dist/contract/index.js +32 -7
  44. package/dist/contract/index.js.map +1 -1
  45. package/dist/{control-CehLtoET.d.ts → control-GeE8OhpN.d.ts} +1 -1
  46. package/dist/control.d.ts +2 -2
  47. package/dist/hosted/index.d.ts +4 -4
  48. package/dist/{index-B1RKber3.d.ts → index-DE3RXAXD.d.ts} +1 -1
  49. package/dist/index.d.ts +78 -287
  50. package/dist/index.js +87 -410
  51. package/dist/index.js.map +1 -1
  52. package/dist/{insight-report-dlpEzQDi.d.ts → insight-report-3ADTfClO.d.ts} +1 -1
  53. package/dist/{kind-factory-DqV2t1Xk.d.ts → kind-factory-CVecZZG_.d.ts} +2 -2
  54. package/dist/{llm-client-DbjLfz-K.d.ts → llm-client-CuUg2Mn3.d.ts} +1 -1
  55. package/dist/meta-eval/index.d.ts +2 -2
  56. package/dist/openapi.json +1 -1
  57. package/dist/pipelines/index.js +2 -2
  58. package/dist/{provenance-jG-Gngg8.d.ts → provenance-DPpNIOJD.d.ts} +4 -4
  59. package/dist/{registry-BK0Zee01.d.ts → registry-DrEQ3Luj.d.ts} +1 -1
  60. package/dist/{release-report-CXXZlR8g.d.ts → release-report-hlNtD12q.d.ts} +2 -2
  61. package/dist/reporting.d.ts +5 -5
  62. package/dist/reporting.js +3 -3
  63. package/dist/{researcher-rInLj9De.d.ts → researcher-BLPHBbNV.d.ts} +3 -3
  64. package/dist/rl.d.ts +7 -7
  65. package/dist/rl.js +4 -4
  66. package/dist/{rubric-predictive-validity-CLPuwiUw.d.ts → rubric-predictive-validity-CnEl9Jc8.d.ts} +1 -1
  67. package/dist/{run-campaign-OVEZF24D.js → run-campaign-4Y5V5CN3.js} +3 -3
  68. package/dist/{run-improvement-loop-BAl_aVOZ.d.ts → run-improvement-loop-CNqQckTj.d.ts} +3 -3
  69. package/dist/{run-record-sItO5ftF.d.ts → run-record-De9VarXR.d.ts} +1 -1
  70. package/dist/{semantic-concept-judge-qXEUV2w7.d.ts → semantic-concept-judge-DIEgr_6v.d.ts} +6 -6
  71. package/dist/{statistics-B7yCbi9i.d.ts → statistics-CnC1FMbx.d.ts} +3 -7
  72. package/dist/{store-GmBE2pZZ.d.ts → store-C1YxJDEK.d.ts} +1 -1
  73. package/dist/{summary-report-BTaXq1TS.d.ts → summary-report-Db0dDSWP.d.ts} +1 -1
  74. package/dist/traces.d.ts +5 -5
  75. package/dist/{types-DRvV0zRo.d.ts → types-Cu3u_x59.d.ts} +3 -3
  76. package/dist/{types-4mm2msnR.d.ts → types-D7lLRYe9.d.ts} +1 -1
  77. package/dist/wire/index.js +2 -2
  78. package/dist/workflow/index.d.ts +7 -7
  79. package/dist/workflow/index.js +1 -1
  80. package/docs/concepts.md +1 -0
  81. package/docs/research/belief-state-agent-eval-roadmap.md +39 -7
  82. package/docs/self-improvement-map.md +111 -0
  83. package/package.json +2 -2
  84. package/dist/chunk-IHDHUN2X.js.map +0 -1
  85. package/dist/chunk-ITBRCT73.js.map +0 -1
  86. package/dist/chunk-LB2UOI5F.js.map +0 -1
  87. package/dist/chunk-ZPSKPT3V.js.map +0 -1
  88. /package/dist/{chunk-B26KI423.js.map → chunk-FZWAFVAA.js.map} +0 -0
  89. /package/dist/{chunk-GXHLRXDI.js.map → chunk-OTYQPHPL.js.map} +0 -0
  90. /package/dist/{chunk-6REHLN5J.js.map → chunk-QS3RBQPI.js.map} +0 -0
  91. /package/dist/{chunk-GWGO2K6Y.js.map → chunk-RBNA5AZT.js.map} +0 -0
  92. /package/dist/{chunk-CF67I6QY.js.map → chunk-VIDQF3F5.js.map} +0 -0
  93. /package/dist/{chunk-XXNIODOM.js.map → chunk-WJL2NJXN.js.map} +0 -0
  94. /package/dist/{chunk-KX6F6NCG.js.map → chunk-Z7VFTS2J.js.map} +0 -0
  95. /package/dist/{run-campaign-OVEZF24D.js.map → run-campaign-4Y5V5CN3.js.map} +0 -0
@@ -1,4 +1,4 @@
1
- import { G as GainDistributionBin, P as ParetoFigureSpec } from './summary-report-BTaXq1TS.js';
1
+ import { G as GainDistributionBin, P as ParetoFigureSpec } from './summary-report-Db0dDSWP.js';
2
2
  import { a as ContinuousAgreement } from './judge-calibration-DilmB3Ml.js';
3
3
 
4
4
  /**
@@ -1,7 +1,7 @@
1
1
  import { AxAIService, AxFunction } from '@ax-llm/ax';
2
- import { T as TraceAnalysisStore } from './store-GmBE2pZZ.js';
2
+ import { T as TraceAnalysisStore } from './store-C1YxJDEK.js';
3
3
  import { z } from 'zod';
4
- import { g as AnalystCost, a as AnalystContext, A as Analyst } from './types-DRvV0zRo.js';
4
+ import { g as AnalystCost, a as AnalystContext, A as Analyst } from './types-Cu3u_x59.js';
5
5
 
6
6
  /**
7
7
  * Typed Ax output for analyst findings.
@@ -52,7 +52,7 @@ interface LlmCallRequest {
52
52
  };
53
53
  temperature?: number;
54
54
  maxTokens?: number;
55
- /** Per-call timeout, default 60s. */
55
+ /** Per-call timeout, default 300s. */
56
56
  timeoutMs?: number;
57
57
  }
58
58
  interface LlmUsage {
@@ -1,7 +1,7 @@
1
1
  export { C as CalibrationBin, a as CalibrationOptions, b as CalibrationPair, c as CalibrationReport, d as CorrelationResult, e as CorrelationStudyOptions, f as CorrelationStudyResult, E as EvalMetricSpec, O as OutcomePair, g as calibrationCurve, h as calibrationFromPairs, i as correlationStudy } from '../calibration-Cpr3WaX3.js';
2
2
  export { D as DeploymentOutcome, F as FileSystemOutcomeStore, a as FileSystemOutcomeStoreOptions, I as InMemoryOutcomeStore, O as OutcomeFilter, b as OutcomeStore } from '../outcome-store-rnXLEqSn.js';
3
- export { R as RubricOutcomePair, a as RubricPredictiveValidityInput, b as RubricPredictiveValidityReport, c as RubricRanking, r as rubricPredictiveValidity } from '../rubric-predictive-validity-CLPuwiUw.js';
3
+ export { R as RubricOutcomePair, a as RubricPredictiveValidityInput, b as RubricPredictiveValidityReport, c as RubricRanking, r as rubricPredictiveValidity } from '../rubric-predictive-validity-CnEl9Jc8.js';
4
4
  import '../store-CKUAgsJz.js';
5
5
  import '../schema-m0gsnbt3.js';
6
- import '../run-record-sItO5ftF.js';
6
+ import '../run-record-De9VarXR.js';
7
7
  import '../errors-Dwqw-T_m.js';
package/dist/openapi.json CHANGED
@@ -2,7 +2,7 @@
2
2
  "openapi": "3.1.0",
3
3
  "info": {
4
4
  "title": "@tangle-network/agent-eval — wire protocol",
5
- "version": "0.80.0",
5
+ "version": "0.82.0",
6
6
  "description": "HTTP and stdio RPC interface to agent-eval. The TypeScript runtime is the source of truth; this spec is the contract that cross-language clients (Python, Rust, Go) generate from.\n\nWire-protocol version: 1.0.0. Bumps on breaking changes to request/response schemas.",
7
7
  "contact": {
8
8
  "name": "Tangle Network",
@@ -3,13 +3,13 @@ import {
3
3
  classifyFailure,
4
4
  compareToBaseline,
5
5
  computeToolUseMetrics
6
- } from "../chunk-GWGO2K6Y.js";
6
+ } from "../chunk-RBNA5AZT.js";
7
7
  import {
8
8
  buildTrajectory
9
9
  } from "../chunk-RZTMDUO7.js";
10
10
  import {
11
11
  interRaterReliability
12
- } from "../chunk-ITBRCT73.js";
12
+ } from "../chunk-IDVBLYCY.js";
13
13
  import {
14
14
  aggregateLlm,
15
15
  argHash,
@@ -1,9 +1,9 @@
1
- import { o as Mutator, I as ImprovementDriver, S as Scenario, G as Gate, k as GateResult, j as GateContext, g as CampaignResult, M as MutableSurface, c as GateDecision } from './types-4mm2msnR.js';
1
+ import { o as Mutator, I as ImprovementDriver, S as Scenario, G as Gate, k as GateResult, j as GateContext, g as CampaignResult, M as MutableSurface, c as GateDecision } from './types-D7lLRYe9.js';
2
2
  import { R as RedTeamCase } from './red-team-DW9Ca_tj.js';
3
- import { R as RunRecord } from './run-record-sItO5ftF.js';
3
+ import { R as RunRecord } from './run-record-De9VarXR.js';
4
4
  import { D as Direction } from './pareto-E-pembql.js';
5
- import { a as PairedBootstrapResult } from './statistics-B7yCbi9i.js';
6
- import { a as RunCampaignOptions, C as CampaignStorage } from './run-improvement-loop-BAl_aVOZ.js';
5
+ import { a as PairedBootstrapResult } from './statistics-CnC1FMbx.js';
6
+ import { b as RunCampaignOptions, C as CampaignStorage } from './run-improvement-loop-CNqQckTj.js';
7
7
  import { HostedClient, TraceSpanEvent } from './hosted/index.js';
8
8
 
9
9
  /**
@@ -1,4 +1,4 @@
1
- import { A as Analyst, a as AnalystContext, b as AnalystRunSummary, c as AnalystFinding, d as AnalystRunResult, C as ChatClient, e as AnalystRunInputs, f as AnalystRunEvent } from './types-DRvV0zRo.js';
1
+ import { A as Analyst, a as AnalystContext, b as AnalystRunSummary, c as AnalystFinding, d as AnalystRunResult, C as ChatClient, e as AnalystRunInputs, f as AnalystRunEvent } from './types-Cu3u_x59.js';
2
2
 
3
3
  /**
4
4
  * AnalystRegistry — orchestrate N analysts against one run.
@@ -1,6 +1,6 @@
1
1
  import { D as DatasetSplit, c as DatasetManifest, a as DatasetScenario } from './dataset-B2kL-fSM.js';
2
- import { m as GateDecision } from './summary-report-BTaXq1TS.js';
3
- import { R as RunRecord, b as RunSplitTag } from './run-record-sItO5ftF.js';
2
+ import { m as GateDecision } from './summary-report-Db0dDSWP.js';
3
+ import { R as RunRecord, a as RunSplitTag } from './run-record-De9VarXR.js';
4
4
 
5
5
  /**
6
6
  * Release confidence gate.
@@ -1,9 +1,9 @@
1
- export { R as RubricOutcomePair, a as RubricPredictiveValidityInput, b as RubricPredictiveValidityReport, c as RubricRanking, r as rubricPredictiveValidity } from './rubric-predictive-validity-CLPuwiUw.js';
2
- export { B as BootstrapOptions, a as BootstrapResult, J as JudgeReplayGateArgs, R as ReleaseConfidenceAxis, b as ReleaseConfidenceAxisName, c as ReleaseConfidenceInput, d as ReleaseConfidenceIssue, e as ReleaseConfidenceMetrics, f as ReleaseConfidenceScorecard, g as ReleaseConfidenceStatus, h as ReleaseConfidenceThresholds, i as ReleaseTraceEvidence, j as RenderReleaseReportOptions, V as Verdict, k as assertReleaseConfidence, l as bootstrapCi, m as evaluateReleaseConfidence, n as judgeReplayGate, r as renderReleaseReport } from './release-report-CXXZlR8g.js';
1
+ export { R as RubricOutcomePair, a as RubricPredictiveValidityInput, b as RubricPredictiveValidityReport, c as RubricRanking, r as rubricPredictiveValidity } from './rubric-predictive-validity-CnEl9Jc8.js';
2
+ export { B as BootstrapOptions, a as BootstrapResult, J as JudgeReplayGateArgs, R as ReleaseConfidenceAxis, b as ReleaseConfidenceAxisName, c as ReleaseConfidenceInput, d as ReleaseConfidenceIssue, e as ReleaseConfidenceMetrics, f as ReleaseConfidenceScorecard, g as ReleaseConfidenceStatus, h as ReleaseConfidenceThresholds, i as ReleaseTraceEvidence, j as RenderReleaseReportOptions, V as Verdict, k as assertReleaseConfidence, l as bootstrapCi, m as evaluateReleaseConfidence, n as judgeReplayGate, r as renderReleaseReport } from './release-report-hlNtD12q.js';
3
3
  export { I as InterimReleaseConfidence, a as InterimReleaseConfidenceInput, P as PairedEvalueOptions, b as PairedEvalueSequence, c as PairedEvalueStep, S as SequentialDecision, e as evaluateInterimReleaseConfidence, p as pairedEvalueSequence } from './sequential-5iSVfzl2.js';
4
- export { P as PairedBootstrapOptions, a as PairedBootstrapResult, b as benjaminiHochberg, p as pairedBootstrap, w as wilcoxonSignedRank } from './statistics-B7yCbi9i.js';
5
- export { G as GainDistributionBin, a as GainDistributionFigureSpec, b as GainDistributionOptions, P as ParetoFigureSpec, c as ParetoPoint, R as RESEARCH_REPORT_HARD_PAIR_FLOOR, d as ResearchReport, e as ResearchReportCandidate, f as ResearchReportDecision, g as ResearchReportMethodology, h as ResearchReportOptions, i as ResearchReportRecommendation, S as SummaryTable, j as SummaryTableOptions, k as SummaryTableRow, l as gainHistogram, p as paretoChart, r as researchReport, s as summaryTable } from './summary-report-BTaXq1TS.js';
6
- import './run-record-sItO5ftF.js';
4
+ export { P as PairedBootstrapOptions, a as PairedBootstrapResult, b as benjaminiHochberg, p as pairedBootstrap, w as wilcoxonSignedRank } from './statistics-CnC1FMbx.js';
5
+ export { G as GainDistributionBin, a as GainDistributionFigureSpec, b as GainDistributionOptions, P as ParetoFigureSpec, c as ParetoPoint, R as RESEARCH_REPORT_HARD_PAIR_FLOOR, d as ResearchReport, e as ResearchReportCandidate, f as ResearchReportDecision, g as ResearchReportMethodology, h as ResearchReportOptions, i as ResearchReportRecommendation, S as SummaryTable, j as SummaryTableOptions, k as SummaryTableRow, l as gainHistogram, p as paretoChart, r as researchReport, s as summaryTable } from './summary-report-Db0dDSWP.js';
6
+ import './run-record-De9VarXR.js';
7
7
  import './errors-Dwqw-T_m.js';
8
8
  import './schema-m0gsnbt3.js';
9
9
  import './outcome-store-rnXLEqSn.js';
package/dist/reporting.js CHANGED
@@ -4,7 +4,7 @@ import {
4
4
  evaluateReleaseConfidence,
5
5
  judgeReplayGate,
6
6
  renderReleaseReport
7
- } from "./chunk-B26KI423.js";
7
+ } from "./chunk-FZWAFVAA.js";
8
8
  import {
9
9
  rubricPredictiveValidity
10
10
  } from "./chunk-YRZ4M5GS.js";
@@ -18,12 +18,12 @@ import {
18
18
  paretoChart,
19
19
  researchReport,
20
20
  summaryTable
21
- } from "./chunk-KX6F6NCG.js";
21
+ } from "./chunk-Z7VFTS2J.js";
22
22
  import {
23
23
  benjaminiHochberg,
24
24
  pairedBootstrap,
25
25
  wilcoxonSignedRank
26
- } from "./chunk-ITBRCT73.js";
26
+ } from "./chunk-IDVBLYCY.js";
27
27
  import "./chunk-VSMTAMNK.js";
28
28
  import "./chunk-3BFEG2F6.js";
29
29
  import "./chunk-PZ5AY32C.js";
@@ -1,6 +1,6 @@
1
- import { b as RunSplitTag, a as RunTokenUsage, c as RunJudgeMetadata, J as JudgeScoresRecord, A as AgentProfileCell, d as AgentProfileCellInput, R as RunRecord } from './run-record-sItO5ftF.js';
2
- import { L as LlmClientOptions, a as LlmRouteRequirements } from './llm-client-DbjLfz-K.js';
3
- import { h as ResearchReportOptions, d as ResearchReport, m as GateDecision } from './summary-report-BTaXq1TS.js';
1
+ import { a as RunSplitTag, b as RunTokenUsage, c as RunJudgeMetadata, J as JudgeScoresRecord, A as AgentProfileCell, d as AgentProfileCellInput, R as RunRecord } from './run-record-De9VarXR.js';
2
+ import { L as LlmClientOptions, a as LlmRouteRequirements } from './llm-client-CuUg2Mn3.js';
3
+ import { h as ResearchReportOptions, d as ResearchReport, m as GateDecision } from './summary-report-Db0dDSWP.js';
4
4
  import { T as TraceEmitter, R as RunCompleteHook } from './emitter-DEZwY14K.js';
5
5
  import { R as RunIntegrityExpectations, a as RunIntegrityReport } from './integrity-CJzrpUua.js';
6
6
  import { R as RawProviderSink } from './raw-provider-sink-C46HDghv.js';
package/dist/rl.d.ts CHANGED
@@ -1,19 +1,19 @@
1
1
  export { O as OffPolicyEstimate, a as OffPolicyOptions, b as OffPolicyTrajectory, d as doublyRobust, i as inverseProbabilityWeighting, o as offPolicyEstimateAll, s as selfNormalizedImportanceWeighting } from './off-policy-DiwuKKg7.js';
2
- import { R as RunRecord, b as RunSplitTag } from './run-record-sItO5ftF.js';
3
- import { g as CampaignResult } from './types-4mm2msnR.js';
2
+ import { R as RunRecord, a as RunSplitTag } from './run-record-De9VarXR.js';
3
+ import { g as CampaignResult } from './types-D7lLRYe9.js';
4
4
  import { a as VerificationReport } from './multi-layer-verifier-DlWCXuxL.js';
5
5
  import { S as Span } from './schema-m0gsnbt3.js';
6
6
  import { T as TraceStore } from './store-CKUAgsJz.js';
7
7
  import { b as OutcomeStore } from './outcome-store-rnXLEqSn.js';
8
8
  export { D as DeploymentOutcome, F as FileSystemOutcomeStore, a as FileSystemOutcomeStoreOptions, I as InMemoryOutcomeStore } from './outcome-store-rnXLEqSn.js';
9
- import { b as RubricPredictiveValidityReport } from './rubric-predictive-validity-CLPuwiUw.js';
10
- import { R as Researcher, F as FailureMode, S as SteeringChange, E as ExperimentPlan, a as ExperimentResult, b as EvalCampaignResult, c as EvalCampaignOptions } from './researcher-rInLj9De.js';
11
- export { r as runEvalCampaign } from './researcher-rInLj9De.js';
9
+ import { b as RubricPredictiveValidityReport } from './rubric-predictive-validity-CnEl9Jc8.js';
10
+ import { R as Researcher, F as FailureMode, S as SteeringChange, E as ExperimentPlan, a as ExperimentResult, b as EvalCampaignResult, c as EvalCampaignOptions } from './researcher-BLPHBbNV.js';
11
+ export { r as runEvalCampaign } from './researcher-BLPHBbNV.js';
12
12
  import { I as InterimReleaseConfidence } from './sequential-5iSVfzl2.js';
13
13
  import './errors-Dwqw-T_m.js';
14
- import './llm-client-DbjLfz-K.js';
14
+ import './llm-client-CuUg2Mn3.js';
15
15
  import './raw-provider-sink-C46HDghv.js';
16
- import './summary-report-BTaXq1TS.js';
16
+ import './summary-report-Db0dDSWP.js';
17
17
  import './failure-cluster-CL7IVgkJ.js';
18
18
  import './emitter-DEZwY14K.js';
19
19
  import './integrity-CJzrpUua.js';
package/dist/rl.js CHANGED
@@ -16,19 +16,19 @@ import {
16
16
  } from "./chunk-3RF76KTD.js";
17
17
  import {
18
18
  runEvalCampaign
19
- } from "./chunk-XXNIODOM.js";
20
- import "./chunk-IHDHUN2X.js";
19
+ } from "./chunk-WJL2NJXN.js";
20
+ import "./chunk-CVVHBFGN.js";
21
21
  import {
22
22
  rubricPredictiveValidity
23
23
  } from "./chunk-YRZ4M5GS.js";
24
24
  import {
25
25
  evaluateInterimReleaseConfidence
26
26
  } from "./chunk-MAZ26DC7.js";
27
- import "./chunk-KX6F6NCG.js";
27
+ import "./chunk-Z7VFTS2J.js";
28
28
  import {
29
29
  benjaminiHochberg,
30
30
  wilcoxonSignedRank
31
- } from "./chunk-ITBRCT73.js";
31
+ } from "./chunk-IDVBLYCY.js";
32
32
  import "./chunk-SBCB6VZY.js";
33
33
  import "./chunk-PC4UYEBM.js";
34
34
  import "./chunk-KWRRMR3J.js";
@@ -1,4 +1,4 @@
1
- import { R as RunRecord } from './run-record-sItO5ftF.js';
1
+ import { R as RunRecord } from './run-record-De9VarXR.js';
2
2
  import { b as OutcomeStore } from './outcome-store-rnXLEqSn.js';
3
3
 
4
4
  /**
@@ -1,10 +1,10 @@
1
1
  import {
2
2
  runCampaign
3
- } from "./chunk-ZPSKPT3V.js";
4
- import "./chunk-ITBRCT73.js";
3
+ } from "./chunk-ZZ2HOPME.js";
4
+ import "./chunk-IDVBLYCY.js";
5
5
  import "./chunk-3BFEG2F6.js";
6
6
  import "./chunk-PZ5AY32C.js";
7
7
  export {
8
8
  runCampaign
9
9
  };
10
- //# sourceMappingURL=run-campaign-OVEZF24D.js.map
10
+ //# sourceMappingURL=run-campaign-4Y5V5CN3.js.map
@@ -1,5 +1,5 @@
1
- import { L as LlmClientOptions } from './llm-client-DbjLfz-K.js';
2
- import { I as ImprovementDriver, S as Scenario, g as CampaignResult, k as GateResult, D as DispatchFn, a as JudgeConfig, L as LabeledScenarioStore, h as CampaignTraceWriter, m as GenerationRecord, M as MutableSurface, P as ParetoParent, G as Gate } from './types-4mm2msnR.js';
1
+ import { L as LlmClientOptions } from './llm-client-CuUg2Mn3.js';
2
+ import { I as ImprovementDriver, S as Scenario, g as CampaignResult, k as GateResult, D as DispatchFn, a as JudgeConfig, L as LabeledScenarioStore, h as CampaignTraceWriter, m as GenerationRecord, M as MutableSurface, P as ParetoParent, G as Gate } from './types-D7lLRYe9.js';
3
3
 
4
4
  /**
5
5
  * @experimental
@@ -424,4 +424,4 @@ interface RunImprovementLoopResult<TArtifact, TScenario extends Scenario> extend
424
424
  declare function runImprovementLoop<TScenario extends Scenario, TArtifact>(opts: RunImprovementLoopOptions<TScenario, TArtifact>): Promise<RunImprovementLoopResult<TArtifact, TScenario>>;
425
425
  declare function defaultRenderDiff(winnerSurface: MutableSurface, baselineSurface: MutableSurface): string;
426
426
 
427
- export { type CampaignStorage as C, type GepaDriverOptions as G, type OpenAutoPrOptions as O, type RunImprovementLoopResult as R, type RunCampaignOptions as a, type RunImprovementLoopOptions as b, runImprovementLoop as c, type GepaDriverConstraints as d, type OpenAutoPrResult as e, fsCampaignStorage as f, gepaDriver as g, type RunOptimizationOptions as h, inMemoryCampaignStorage as i, type RunOptimizationResult as j, countSentenceEdits as k, defaultRenderDiff as l, extractH2Sections as m, runOptimization as n, openAutoPr as o, runCampaign as r, surfaceHash as s };
427
+ export { type CampaignStorage as C, type GepaDriverOptions as G, type OpenAutoPrOptions as O, type RunOptimizationOptions as R, type RunImprovementLoopResult as a, type RunCampaignOptions as b, type RunImprovementLoopOptions as c, runImprovementLoop as d, type GepaDriverConstraints as e, fsCampaignStorage as f, gepaDriver as g, type OpenAutoPrResult as h, inMemoryCampaignStorage as i, type RunOptimizationResult as j, countSentenceEdits as k, defaultRenderDiff as l, extractH2Sections as m, runOptimization as n, openAutoPr as o, runCampaign as r, surfaceHash as s };
@@ -315,4 +315,4 @@ declare function parseRunRecordSafe(input: unknown): {
315
315
  /** Round-trip helper — `JSON.parse(JSON.stringify(record))` then validate. */
316
316
  declare function roundTripRunRecord(record: RunRecord): RunRecord;
317
317
 
318
- export { type AgentProfileCell as A, validateAgentProfileCell as B, validateRunRecord as C, verifyAgentProfileCell as D, type JudgeScoresRecord as J, type RunRecord as R, type SandboxAgentProfileLike as S, type RunTokenUsage as a, type RunSplitTag as b, type RunJudgeMetadata as c, type AgentProfileCellInput as d, AGENT_PROFILE_KINDS as e, type AgentProfileCellSchemaVersion as f, AgentProfileCellValidationError as g, type AgentProfileDimensionValue as h, type AgentProfileHarness as i, type AgentProfileJson as j, type AgentProfileKind as k, type AgentProfileSource as l, type AgentProfileSourceInput as m, type RunOutcome as n, RunRecordValidationError as o, agentProfileCellHashMaterial as p, agentProfileCellKey as q, assertRunAgentProfileCell as r, buildAgentProfileCell as s, buildSandboxAgentProfileCell as t, groupRunsByAgentProfileCell as u, isRunRecord as v, parseRunRecordSafe as w, requireAgentProfileCell as x, roundTripRunRecord as y, toAgentProfileJson as z };
318
+ export { type AgentProfileCell as A, validateAgentProfileCell as B, validateRunRecord as C, verifyAgentProfileCell as D, type JudgeScoresRecord as J, type RunRecord as R, type SandboxAgentProfileLike as S, type RunSplitTag as a, type RunTokenUsage as b, type RunJudgeMetadata as c, type AgentProfileCellInput as d, AGENT_PROFILE_KINDS as e, type AgentProfileCellSchemaVersion as f, AgentProfileCellValidationError as g, type AgentProfileDimensionValue as h, type AgentProfileHarness as i, type AgentProfileJson as j, type AgentProfileKind as k, type AgentProfileSource as l, type AgentProfileSourceInput as m, type RunOutcome as n, RunRecordValidationError as o, agentProfileCellHashMaterial as p, agentProfileCellKey as q, assertRunAgentProfileCell as r, buildAgentProfileCell as s, buildSandboxAgentProfileCell as t, groupRunsByAgentProfileCell as u, isRunRecord as v, parseRunRecordSafe as w, requireAgentProfileCell as x, roundTripRunRecord as y, toAgentProfileJson as z };
@@ -1,10 +1,10 @@
1
1
  import { AxAIService } from '@ax-llm/ax';
2
- import { c as TraceAnalystKindSpec } from './kind-factory-DqV2t1Xk.js';
3
- import { b as AnalystRegistryOptions, a as AnalystRegistry } from './registry-BK0Zee01.js';
2
+ import { c as TraceAnalystKindSpec } from './kind-factory-CVecZZG_.js';
3
+ import { b as AnalystRegistryOptions, a as AnalystRegistry } from './registry-DrEQ3Luj.js';
4
4
  import { z } from 'zod';
5
- import { c as AnalystFinding, A as Analyst, a as AnalystContext } from './types-DRvV0zRo.js';
6
- import { a as TraceAnalystSpan } from './store-GmBE2pZZ.js';
7
- import { L as LlmClientOptions } from './llm-client-DbjLfz-K.js';
5
+ import { c as AnalystFinding, A as Analyst, a as AnalystContext } from './types-Cu3u_x59.js';
6
+ import { a as TraceAnalystSpan } from './store-C1YxJDEK.js';
7
+ import { L as LlmClientOptions } from './llm-client-CuUg2Mn3.js';
8
8
  import { S as Severity } from './multi-layer-verifier-DlWCXuxL.js';
9
9
 
10
10
  interface CreateAnalystAiConfig {
@@ -615,7 +615,7 @@ declare const DEFAULT_COMPLEXITY_WEIGHTS: Record<ConceptComplexity, number>;
615
615
  interface SemanticConceptJudgeOptions {
616
616
  /** Model id to call. Default 'claude-sonnet-4-6' via agent-eval defaults. */
617
617
  model?: string;
618
- /** Per-call timeout. Default 180s. */
618
+ /** Per-call timeout. Default 300s. */
619
619
  timeoutMs?: number;
620
620
  /** Pipeline budget for the prompt (source blob truncation). Default 45000. */
621
621
  maxSourceChars?: number;
@@ -1,13 +1,9 @@
1
1
  import { C as ContinuousAgreementOptions, a as ContinuousAgreement } from './judge-calibration-DilmB3Ml.js';
2
2
  import { J as JudgeScore } from './types-Croy5h7V.js';
3
3
 
4
- /**
5
- * Normalize scores so all dimensions follow "higher = better".
6
- * Inverted dimensions (hallucination, false_confidence, worst_failure)
7
- * already use inverted scoring in the prompt (10 = no hallucination),
8
- * but this function ensures consistency if raw scores leak through.
9
- */
10
- declare function normalizeScores(scores: JudgeScore[]): JudgeScore[];
4
+ /** Identity: dimensions already follow "higher = better" by prompt convention
5
+ * (inverted dims like hallucination are scored 10 = best at the source). */
6
+ declare const normalizeScores: (scores: JudgeScore[]) => JudgeScore[];
11
7
  /** Weighted mean — falls back to uniform weights when omitted */
12
8
  declare function weightedMean(scores: {
13
9
  score: number;
@@ -245,4 +245,4 @@ interface TraceAnalysisStore {
245
245
  }): Promise<SearchSpanResult>;
246
246
  }
247
247
 
248
- export { DEFAULT_TRACE_ANALYST_BUDGETS as D, type QueryTracesPage as Q, type SearchSpanResult as S, type TraceAnalysisStore as T, type ViewSpansResult as V, type TraceAnalystSpan as a, type DatasetOverview as b, type SearchTraceResult as c, type SpanMatchRecord as d, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX as e, type TraceAnalystByteBudgets as f, type TraceAnalystFilters as g, type TraceAnalystSpanKind as h, type TraceAnalystSpanStatus as i, type TraceAnalystTraceSummary as j, type ViewTraceOversized as k, type ViewTraceResult as l };
248
+ export { DEFAULT_TRACE_ANALYST_BUDGETS as D, type ErrorCluster as E, type QueryTracesPage as Q, type SearchSpanResult as S, type TraceAnalysisStore as T, type ViewSpansResult as V, type TraceAnalystSpan as a, type DatasetOverview as b, type SearchTraceResult as c, type SpanMatchRecord as d, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX as e, type TraceAnalystByteBudgets as f, type TraceAnalystFilters as g, type TraceAnalystSpanKind as h, type TraceAnalystSpanStatus as i, type TraceAnalystTraceSummary as j, type ViewTraceOversized as k, type ViewTraceResult as l };
@@ -1,4 +1,4 @@
1
- import { R as RunRecord } from './run-record-sItO5ftF.js';
1
+ import { R as RunRecord } from './run-record-De9VarXR.js';
2
2
  import { F as FailureClusterReport } from './failure-cluster-CL7IVgkJ.js';
3
3
 
4
4
  /**
package/dist/traces.d.ts CHANGED
@@ -10,11 +10,11 @@ export { a as aggregateLlm, b as argHash, g as groupBy, j as judgeSpans, l as ll
10
10
  export { D as DEFAULT_REDACTION_RULES, b as REDACTION_VERSION, a as RedactionReport, R as RedactionRule, r as redactString, c as redactValue } from './redact-B40YG2M_.js';
11
11
  import { R as Run } from './schema-m0gsnbt3.js';
12
12
  export { A as Artifact, B as BudgetLedgerEntry, h as BudgetSpec, E as EventKind, i as FAILURE_CLASSES, F as FailureClass, G as GenericSpan, J as JudgeSpan, L as LlmSpan, M as Message, d as RetrievalSpan, g as RunLayer, b as RunOutcome, f as RunStatus, e as SandboxSpan, S as Span, j as SpanBase, c as SpanKind, k as SpanStatus, l as TRACE_SCHEMA_VERSION, T as ToolSpan, a as TraceEvent, m as isJudgeSpan, n as isLlmSpan, o as isRetrievalSpan, p as isSandboxSpan, q as isToolSpan } from './schema-m0gsnbt3.js';
13
- import { A as AnalyzeTracesOptions, b as AnalyzeTracesResult } from './analyst-t7zZS3TV.js';
14
- export { a as AnalyzeTracesInput, c as AnalyzeTracesTurnSnapshot, d as analyzeTraces } from './analyst-t7zZS3TV.js';
15
- import { h as TraceAnalystSpanKind, i as TraceAnalystSpanStatus, T as TraceAnalysisStore, g as TraceAnalystFilters, b as DatasetOverview, Q as QueryTracesPage, l as ViewTraceResult, V as ViewSpansResult, c as SearchTraceResult, S as SearchSpanResult } from './store-GmBE2pZZ.js';
16
- export { D as DEFAULT_TRACE_ANALYST_BUDGETS, d as SpanMatchRecord, e as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, f as TraceAnalystByteBudgets, a as TraceAnalystSpan, j as TraceAnalystTraceSummary, k as ViewTraceOversized } from './store-GmBE2pZZ.js';
17
- import { b as RunSplitTag, a as RunTokenUsage, R as RunRecord } from './run-record-sItO5ftF.js';
13
+ import { A as AnalyzeTracesOptions, b as AnalyzeTracesResult } from './analyst-C8HHvfJp.js';
14
+ export { a as AnalyzeTracesInput, c as AnalyzeTracesTurnSnapshot, d as analyzeTraces } from './analyst-C8HHvfJp.js';
15
+ import { h as TraceAnalystSpanKind, i as TraceAnalystSpanStatus, T as TraceAnalysisStore, g as TraceAnalystFilters, b as DatasetOverview, Q as QueryTracesPage, l as ViewTraceResult, V as ViewSpansResult, c as SearchTraceResult, S as SearchSpanResult } from './store-C1YxJDEK.js';
16
+ export { D as DEFAULT_TRACE_ANALYST_BUDGETS, E as ErrorCluster, d as SpanMatchRecord, e as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, f as TraceAnalystByteBudgets, a as TraceAnalystSpan, j as TraceAnalystTraceSummary, k as ViewTraceOversized } from './store-C1YxJDEK.js';
17
+ import { a as RunSplitTag, b as RunTokenUsage, R as RunRecord } from './run-record-De9VarXR.js';
18
18
  import { AxFunction } from '@ax-llm/ax';
19
19
 
20
20
  /**
@@ -1,7 +1,7 @@
1
- import { R as RunRecord } from './run-record-sItO5ftF.js';
2
- import { T as TraceAnalysisStore } from './store-GmBE2pZZ.js';
1
+ import { R as RunRecord } from './run-record-De9VarXR.js';
2
+ import { T as TraceAnalysisStore } from './store-C1YxJDEK.js';
3
3
  import { a as JudgeInput } from './types-Croy5h7V.js';
4
- import { b as LlmCallRequest, c as LlmCallResult } from './llm-client-DbjLfz-K.js';
4
+ import { b as LlmCallRequest, c as LlmCallResult } from './llm-client-CuUg2Mn3.js';
5
5
 
6
6
  /**
7
7
  * ChatClient — the single LLM abstraction analysts call.
@@ -1,4 +1,4 @@
1
- import { a as RunTokenUsage } from './run-record-sItO5ftF.js';
1
+ import { b as RunTokenUsage } from './run-record-De9VarXR.js';
2
2
 
3
3
  /**
4
4
  * @experimental
@@ -34,8 +34,8 @@ import {
34
34
  runRpcOnce,
35
35
  startServer,
36
36
  startServerAsync
37
- } from "../chunk-6REHLN5J.js";
38
- import "../chunk-IHDHUN2X.js";
37
+ } from "../chunk-QS3RBQPI.js";
38
+ import "../chunk-CVVHBFGN.js";
39
39
  import "../chunk-PC4UYEBM.js";
40
40
  import "../chunk-3BFEG2F6.js";
41
41
  import "../chunk-PZ5AY32C.js";
@@ -1,24 +1,24 @@
1
1
  import { W as WorkflowTopology } from '../harness-optimizer-EnEnQPsr.js';
2
- import { b as RunSplitTag, a as RunTokenUsage, R as RunRecord } from '../run-record-sItO5ftF.js';
3
- import { c as AnalystFinding, h as AnalystSeverity, E as EvidenceRef } from '../types-DRvV0zRo.js';
4
- import { F as FailureClusterInsight } from '../insight-report-dlpEzQDi.js';
2
+ import { a as RunSplitTag, b as RunTokenUsage, R as RunRecord } from '../run-record-De9VarXR.js';
3
+ import { c as AnalystFinding, h as AnalystSeverity, E as EvidenceRef } from '../types-Cu3u_x59.js';
4
+ import { F as FailureClusterInsight } from '../insight-report-3ADTfClO.js';
5
5
  import { a as VerificationReport, L as LayerResult } from '../multi-layer-verifier-DlWCXuxL.js';
6
6
  import { F as FailureClusterReport } from '../failure-cluster-CL7IVgkJ.js';
7
7
  import { R as RedactionRule, a as RedactionReport } from '../redact-B40YG2M_.js';
8
8
  import { D as DatasetSplit } from '../dataset-B2kL-fSM.js';
9
9
  import { a as FeedbackTrajectory } from '../feedback-trajectory-B3rErRsh.js';
10
- import { a as PairedBootstrapResult } from '../statistics-B7yCbi9i.js';
10
+ import { a as PairedBootstrapResult } from '../statistics-CnC1FMbx.js';
11
11
  import '../pareto-E-pembql.js';
12
12
  import '../run-critic-BAIjX99r.js';
13
13
  import '../schema-m0gsnbt3.js';
14
14
  import '../store-CKUAgsJz.js';
15
15
  import '../errors-Dwqw-T_m.js';
16
- import '../store-GmBE2pZZ.js';
16
+ import '../store-C1YxJDEK.js';
17
17
  import '../types-Croy5h7V.js';
18
18
  import '@tangle-network/tcloud';
19
- import '../llm-client-DbjLfz-K.js';
19
+ import '../llm-client-CuUg2Mn3.js';
20
20
  import '../raw-provider-sink-C46HDghv.js';
21
- import '../summary-report-BTaXq1TS.js';
21
+ import '../summary-report-Db0dDSWP.js';
22
22
  import '../judge-calibration-DilmB3Ml.js';
23
23
  import '../control-runtime-DuFBYg7A.js';
24
24
  import '../emitter-DEZwY14K.js';
@@ -1,6 +1,6 @@
1
1
  import {
2
2
  pairedBootstrap
3
- } from "../chunk-ITBRCT73.js";
3
+ } from "../chunk-IDVBLYCY.js";
4
4
  import {
5
5
  DEFAULT_REDACTION_RULES,
6
6
  redactString
package/docs/concepts.md CHANGED
@@ -182,6 +182,7 @@ release decision.
182
182
 
183
183
  ## Where to go next
184
184
 
185
+ - **Confused by "GEPA / HALO / trace analysis / drivers everywhere"?** → [self-improvement-map.md](./self-improvement-map.md) — one loop, four roles, the seven-driver catalog (production vs bench-only), and why `gepa-refine` is the same loop on a test bench.
185
186
  - **Need the layman feature map?** → [feature-guide.md](./feature-guide.md) — what each primitive does, when to use it, integration patterns, and guardrails.
186
187
  - **Just want to score a string against a rubric?** → [wire-protocol.md](./wire-protocol.md) — HTTP/RPC interface, pluggable from any language.
187
188
  - **Need a reusable driver/worker/evaluator loop?** → [control-runtime.md](./control-runtime.md) — generic runtime plus coding, browser, computer-use, and research integration patterns.
@@ -211,14 +211,16 @@ The API should report uncertainty and support problems, not hide them behind a s
211
211
 
212
212
  ### Q3 2026 - Phase 0: Tracking, Corpus, and Decision Inventory
213
213
 
214
- - [ ] Keep this document current as the research tracker.
215
- - [ ] Create a decision inventory over existing traces: continue, verify, retry, ask, stop, memory-write, memory-read, tool-select, skill-select, workflow-select, prompt-promote.
216
- - [ ] Define trace extraction rules for each decision kind.
214
+ - [x] Keep this document current as the research tracker.
215
+ - [x] Create the first decision inventory over existing code-agent traces: failure-recovery, tool-select, and graph-completion.
216
+ - [x] Define first-pass trace extraction rules for Codex, Claude Code, OpenCode, Kimi Code, and Pi/PiGraph-shaped local traces.
217
217
  - [ ] Define the minimum event fields needed from runtime and knowledge packages.
218
- - [ ] Build a replay corpus from existing `RunRecord` and trace stores.
218
+ - [x] Build the first replay-corpus adapter from existing `RunRecord` rows plus local code-agent session traces.
219
+ - [x] Add a one-call code-agent evidence corpus helper that joins session intake, decision extraction, and the research evidence gate.
219
220
  - [ ] Label at least 200 decision points with outcome, cost, and whether the action was retrospectively correct.
220
- - [ ] Add support diagnostics: missing candidates, missing outcomes, no cost, no raw trace, no held-out split.
221
- - [ ] Decide which decision kind has enough data for Phase 1.
221
+ - [x] Add first support diagnostics: missing outcomes, missing behavior/target propensities, and insufficient target support.
222
+ - [x] Decide the first decision kind for Phase 1 dogfooding: failure recovery after failed tool/patch actions.
223
+ - [x] Add a small research evidence gate that classifies selective vs counterfactual claim support.
222
224
 
223
225
  Completion criteria:
224
226
 
@@ -228,6 +230,15 @@ Completion criteria:
228
230
  - [ ] Backend and capture integrity are checked before analysis.
229
231
  - [ ] No producerless schema fields are introduced.
230
232
  - [ ] One baseline policy is recorded for every decision kind under study.
233
+ - [ ] A generated `BeliefDecisionResearchEvidencePacket` says `supported` for the intended claim scope.
234
+
235
+ Status on 2026-06-05: the experimental implementation exists in `src/belief-state/code-agent-corpus.ts` and `src/belief-state/research-evidence.ts`, with coverage in `src/belief-state/code-agent-corpus.test.ts` and `src/belief-state/research-evidence.test.ts`. A local smoke after build joined 33 private code-agent sessions to 33 `RunRecord`s and emitted 13,137 decision rows across Codex, Claude Code, Kimi Code, OpenCode, and PiGraph-shaped traces. This closes the infrastructure part of Phase 0, but not the empirical proof gate: the next corpus run still has to add split metadata, integrity checks, retrospective labels, and a recorded baseline per target before the work can claim Phase 0 completion. Missing behavior/target propensities now block counterfactual claims while still allowing selective-only claims to be evaluated.
236
+
237
+ Follow-up local smoke on 2026-06-05 over 50 recent Codex JSONL sessions under 20 MB produced 50 `RunRecord`s and 6,770 decision points: 5,610 tool-selection rows and 1,160 failure-recovery rows, with full outcome/confidence coverage and no propensity support. The default confidence-threshold policy did not clear the selective utility gate on failure recovery (`ci.lower = -0.1159` at threshold `0.6`), so the current result supports the extraction/evidence pipeline, not the belief-policy claim; the next empirical step is real logged confidence/propensity or retrospective labels, not more heuristic confidence tuning.
238
+
239
+ Runtime hook bridge on 2026-06-05: `src/belief-state/runtime-hooks.ts` now converts `agent-runtime` decision hooks into outcome-blind shadow-probe inputs, attaches matching lifecycle hook events as probe evidence, and only converts them into full `BeliefDecisionPoint` rows when the observed action is supplied. This keeps `agent-eval` trace/analysis-only while letting the runtime emit producer-backed decision boundaries and context for the next experiment.
240
+
241
+ Taxonomy tightening on 2026-06-05: `src/belief-state/types.ts` now exports stable decision kinds, evidence sources, evidence quality labels, evaluation criteria, and reason codes. The intent is to make future dashboards and paper artifacts aggregate by stable IDs (`calibration`, `ope-support`, `memory-health`, `surface-attribution`, `promotion`, etc.) instead of parsing prose diagnostics.
231
242
 
232
243
  ### Q4 2026 - Phase 1: Selective Prediction and Abstention
233
244
 
@@ -236,6 +247,7 @@ Completion criteria:
236
247
  - [ ] Compare baseline policy vs selective policy on holdout.
237
248
  - [ ] Add cost-aware utility: quality lift minus verification/ask/retry cost.
238
249
  - [ ] Add report rows into `InsightReport` or an experimental research report.
250
+ - [ ] Export the real corpus packet into the paper artifact instead of hand-copying metrics.
239
251
  - [ ] Run negative controls: shuffled confidence, random abstention, always-verify, never-verify.
240
252
  - [ ] Pre-register thresholds before holdout.
241
253
 
@@ -364,6 +376,21 @@ Do not call belief-state work "done" until these are true:
364
376
  - [ ] No runtime ownership boundary is crossed from `agent-eval`.
365
377
  - [ ] Negative results are recorded instead of hidden.
366
378
 
379
+ Stable criterion IDs:
380
+
381
+ - `capture-integrity`
382
+ - `decision-completeness`
383
+ - `evidence-quality`
384
+ - `outcome-quality`
385
+ - `calibration`
386
+ - `accepted-region-risk`
387
+ - `policy-value`
388
+ - `ope-support`
389
+ - `memory-health`
390
+ - `surface-attribution`
391
+ - `generalization`
392
+ - `promotion`
393
+
367
394
  ## Kill Criteria
368
395
 
369
396
  Stop or pivot if any of these persist for two consecutive phases:
@@ -408,12 +435,14 @@ The most succinct integration is an experimental `src/belief-state/` module that
408
435
 
409
436
  | File | Purpose | Notes |
410
437
  |---|---|---|
411
- | `src/belief-state/types.ts` | Defines `BeliefDecisionPoint`, `BeliefDecisionKind`, `BeliefActionChoice`, `BeliefDecisionOutcome`, `BeliefEvidenceRef`, `BeliefPolicyEvaluationReport`, `SupportDiagnostics`. | Pure types. No runtime dependency. |
438
+ | `src/belief-state/types.ts` | Defines `BeliefDecisionPoint`, `BeliefDecisionKind`, `BeliefDecisionOutcome`, `BeliefEvidenceRef`, `BeliefPolicyEvaluationReport`, support diagnostics, stable criteria, and reason codes. | No runtime dependency. Keep taxonomy compact and producer-backed. |
412
439
  | `src/belief-state/extract.ts` | Extracts decision points from `TraceStore` runs/spans/events. | Structural parsing only. Unknown events are skipped with diagnostics. |
413
440
  | `src/belief-state/selective.ts` | Evaluates continue/verify/ask/retry/stop policies against observed outcomes. | Computes coverage, accepted-error rate, rejected-action lift, cost-adjusted utility. |
414
441
  | `src/belief-state/calibration.ts` | Computes confidence calibration for decision predictions. | Calls shared `calibrationFromPairs()` once added. |
415
442
  | `src/belief-state/ope.ts` | Converts decision rows into `OffPolicyTrajectory[]` for an explicit named target policy and calls `offPolicyEstimateAll`. | Must report ESS and support mismatch; no silent value claims. |
416
443
  | `src/belief-state/report.ts` | Orchestrates extraction + selective eval + calibration + OPE into one report. | Returns honest negative / need-more-data when unsupported. |
444
+ | `src/belief-state/code-agent-corpus.ts` | Converts local code-agent sessions into belief decision points, inventories targets, selects the first supported target, and runs the experimental policy report. | Supports Codex, Claude Code, OpenCode, Kimi Code, and Pi/PiGraph-shaped traces. Does not invent behavior or target propensities. |
445
+ | `src/belief-state/runtime-hooks.ts` | Bridges structurally typed `agent-runtime` decision hooks into shadow probes or completed belief decision rows. | No runtime dependency. Pre-action hooks do not fake `chosenAction`. |
417
446
  | `src/belief-state/index.ts` | Experimental barrel for the module. | Keep out of root barrel and expose only through `./experimental/belief-state` while evidence gates are open. |
418
447
 
419
448
  ### Files to Change First
@@ -536,6 +565,9 @@ Promotion:
536
565
  | `src/belief-state/calibration.test.ts` | ECE bins; equal-width/equal-frequency behavior; too-few-pairs returns unsupported. |
537
566
  | `src/belief-state/ope.test.ts` | converts to `OffPolicyTrajectory`; explicit target policy required; invalid propensity disables OPE without throwing; low ESS support mismatch; estimator agreement surfaced. |
538
567
  | `src/belief-state/report.test.ts` | full report status: `ship`, `hold`, `need_more_data`; recommendation cannot ship on OPE alone. |
568
+ | `src/belief-state/code-agent-corpus.test.ts` | extracts code-agent decision corpora across Codex, Claude Code, OpenCode, Kimi Code, and Pi/PiGraph-shaped traces; inventories targets; picks failure recovery first; holds when OPE propensities are absent. |
569
+ | `src/belief-state/runtime-hooks.test.ts` | converts runtime decision hooks to outcome-blind shadow probes; attaches matching lifecycle hook events as evidence; requires observed action for full belief rows; collector stays structurally compatible with runtime hooks. |
570
+ | `src/belief-state/types.test.ts` | guards the stable decision kinds, evidence sources/qualities, evaluation criteria, and reason-code taxonomy. |
539
571
  | `src/meta-eval/calibration.test.ts` | existing `calibrationCurve()` still works after extracting pure helper. |
540
572
 
541
573
  ### Verification Commands