@tangle-network/agent-eval 0.82.0 → 0.84.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -4,6 +4,14 @@ All notable changes to `@tangle-network/agent-eval` and its sibling `agent-eval-
4
4
 
5
5
  ---
6
6
 
7
+ ## [0.83.0] — 2026-06-05 — hostedTenantFromEnv
8
+
9
+ ### Added
10
+
11
+ - **`hostedTenantFromEnv` (`/hosted`).** Builds a `HostedTenant` config from env (the input `selfImprove({ hostedTenant })` and `emitLoopProvenance` take), with the same env precedence + overrides as `hostedClientFromEnv` — which now composes it. Returns `undefined` (not an error) when unconfigured, so a product wires `hostedTenant: hostedTenantFromEnv({ tenantId: 'my-agent' })` unconditionally and hosted ingest stays off until the env is set. Removes the env→tenant mapping every product would otherwise hand-roll when collapsing onto `selfImprove`.
12
+
13
+ ---
14
+
7
15
  ## [0.82.0] — 2026-06-05 — selfImprove forwards the full loop surface
8
16
 
9
17
  ### Changed
@@ -21,7 +29,7 @@ A deterministic offline test that drives `selfImprove` with a mock agent must no
21
29
 
22
30
  ### Added
23
31
 
24
- - **`aggregateJudgeVerdicts<D>` (root).** Generic judge-ensemble reducer: fan out N uncorrelated judges, mean each rubric dimension over the SURVIVORS, report the inter-rater disagreement spread, sum cost. Replaces the same reduction hand-rolled in legal (`aggregateEnsemble`), creative (`production-loop/judges.ts`), and tax (`judge-ensemble.ts`). Fail-loud: a failed judge (`perDimension: null`) is recorded in `failedJudges`, never folded into a zero; all-failed throws; a failed judge's cost is still summed. Composite reuses `weightedComposite`.
32
+ - **`aggregateJudgeVerdicts<D>` (root).** Generic judge-ensemble reducer: fan out N uncorrelated judges, mean each rubric dimension over the SURVIVORS, report the inter-rater disagreement spread, sum cost. Replaces the same reduction hand-rolled across multiple product agents. Fail-loud: a failed judge (`perDimension: null`) is recorded in `failedJudges`, never folded into a zero; all-failed throws; a failed judge's cost is still summed. Composite reuses `weightedComposite`.
25
33
  - **`createTokenRecallChecker` (root).** The deterministic, no-LLM `CorrectnessChecker` — sibling of `createLlmCorrectnessChecker`. A produced item fulfils a requirement when its content is substantive and recalls ≥ `minRecall` of the requirement title's significant tokens. The default completion gate for apps/tests without an LLM judge.
26
34
  - **`ErrorCluster` (root + `/analyst`).** The failure-cluster element type is now a named export, so consumers import it instead of deriving `DatasetOverview['error_clusters'][number]`.
27
35
 
@@ -1,8 +1,9 @@
1
1
  import { c as CalibrationReport } from '../calibration-Cpr3WaX3.js';
2
2
  import { O as OffPolicyEstimate, a as OffPolicyOptions, b as OffPolicyTrajectory } from '../off-policy-DiwuKKg7.js';
3
3
  import { d as CodeAgentSessionSource, a as CodeAgentSessionIntakeOptions, c as CodeAgentSessionMetrics, C as CodeAgentSessionDiagnostic } from '../code-agent-session-BRXmavYv.js';
4
- import { R as RunRecord } from '../run-record-De9VarXR.js';
4
+ import { R as RunRecord, a as RunSplitTag } from '../run-record-De9VarXR.js';
5
5
  import { T as TraceStore } from '../store-CKUAgsJz.js';
6
+ import { R as RuntimeTrajectoryRecord, P as ProjectRuntimeTrajectoryEvidenceOptions, a as RuntimeTrajectoryEvidenceProjection } from '../runtime-trajectory-BLRiaifm.js';
6
7
  import '../schema-m0gsnbt3.js';
7
8
  import '../outcome-store-rnXLEqSn.js';
8
9
  import '../errors-Dwqw-T_m.js';
@@ -521,4 +522,82 @@ declare function runtimeDecisionPointToBeliefShadowProbeInput(point: RuntimeBeli
521
522
  declare function runtimeDecisionPointToBeliefDecisionPoint(point: RuntimeBeliefDecisionPoint, options: RuntimeBeliefDecisionPointOptions): RuntimeBeliefDecisionPointReport;
522
523
  declare function createBeliefRuntimeHookCollector(defaults: RuntimeBeliefShadowProbeInputOptions): BeliefRuntimeHookCollector;
523
524
 
524
- export { type AnalyzeBeliefDecisionCorpusOptions, type AnalyzeBeliefPolicyOpeOptions, type AnalyzeBeliefPolicyOptions, BELIEF_DECISION_KINDS, BELIEF_EVALUATION_CRITERIA, BELIEF_EVIDENCE_QUALITIES, BELIEF_EVIDENCE_SOURCES, type BeliefCalibrationOptions, type BeliefCalibrationRegion, type BeliefCalibrationStatus, type BeliefDecisionCorpusEvaluation, type BeliefDecisionExtractionDiagnostic, type BeliefDecisionExtractionReport, type BeliefDecisionInventoryBucket, type BeliefDecisionInventoryReport, type BeliefDecisionKind, type BeliefDecisionOutcome, type BeliefDecisionPoint, type BeliefDecisionReason, type BeliefDecisionReasonCode, type BeliefDecisionResearchEvidencePacket, type BeliefDecisionTargetSelection, type BeliefEvaluationCriterionId, type BeliefEvaluationStatus, type BeliefEvidenceQuality, type BeliefEvidenceRef, type BeliefEvidenceSource, type BeliefOffPolicyTrajectoryReport, type BeliefOpeOptions, type BeliefOpeReport, type BeliefOpeStatus, type BeliefOpeSupportDiagnostics, type BeliefOpeTargetPolicy, type BeliefPolicyAction, type BeliefPolicyDecision, type BeliefPolicyEvaluationReport, type BeliefResearchClaimScope, type BeliefResearchEvidenceGate, type BeliefResearchEvidenceStatus, type BeliefResearchGateId, type BeliefRuntimeHookCollector, type BeliefSelectivePolicy, type BeliefSelectivePolicyMetrics, type BeliefShadowProbeDiagnostic, type BeliefShadowProbeEvidenceRef, type BeliefShadowProbeInput, type BeliefShadowProbeRecord, type BeliefShadowProbeResponse, type BeliefShadowProbeRun, type BeliefShadowProbeSummary, type BeliefUtilityOptions, type BuildBeliefDecisionResearchEvidencePacketOptions, type BuildCodeAgentBeliefEvidenceCorpusOptions, type CodeAgentBeliefDecisionTargetId, type CodeAgentBeliefEvidenceCorpus, type CodeAgentBeliefSession, type EvaluateBeliefSelectivePolicyOptions, type ExtractBeliefDecisionPointsOptions, type ExtractCodeAgentBeliefDecisionPointsOptions, type RunBeliefShadowProbeOptions, type RuntimeBeliefConversionDiagnostic, type RuntimeBeliefDecisionEvidenceRef, type RuntimeBeliefDecisionPoint, type RuntimeBeliefDecisionPointOptions, type RuntimeBeliefDecisionPointReport, type RuntimeBeliefHookContext, type RuntimeBeliefHookEvent, type RuntimeBeliefHooks, type RuntimeBeliefShadowProbeInputOptions, type RuntimeBeliefShadowProbeInputReport, type SelectBeliefDecisionTargetOptions, analyzeBeliefDecisionCorpus, analyzeBeliefPolicy, beliefDecisionsToOffPolicyTrajectories, buildBeliefDecisionResearchEvidencePacket, buildCodeAgentBeliefEvidenceCorpus, calibrateBeliefDecisions, createBeliefRuntimeHookCollector, embeddedBeliefOpeTargetPolicy, evaluateBeliefOffPolicy, evaluateBeliefSelectivePolicy, extractBeliefDecisionPoints, extractCodeAgentBeliefDecisionPoints, formatBeliefShadowProbePrompt, inventoryBeliefDecisionPoints, isBeliefDecisionKind, isBeliefEvidenceSource, runBeliefShadowProbe, runtimeDecisionPointToBeliefDecisionPoint, runtimeDecisionPointToBeliefShadowProbeInput, selectBeliefDecisionTarget, thresholdSelectivePolicy };
525
+ interface RuntimeBeliefPhase0RunRecord {
526
+ runId: string;
527
+ scenarioId?: string;
528
+ splitTag: RunSplitTag;
529
+ }
530
+ interface RuntimeBeliefDecisionLabel {
531
+ decisionId: string;
532
+ chosenAction: string;
533
+ outcome: BeliefDecisionOutcome;
534
+ confidence?: number;
535
+ behaviorProb?: number;
536
+ targetProb?: number;
537
+ qHat?: number | null;
538
+ costUsd?: number;
539
+ splitTag?: RunSplitTag;
540
+ metadata?: Record<string, unknown>;
541
+ }
542
+ interface BuildRuntimeBeliefPhase0MeasurementOptions extends Omit<BuildBeliefDecisionResearchEvidencePacketOptions, 'points'> {
543
+ runs: RuntimeBeliefPhase0RunRecord[];
544
+ decisions: RuntimeBeliefDecisionPoint[];
545
+ events?: RuntimeBeliefHookEvent[];
546
+ labels: RuntimeBeliefDecisionLabel[];
547
+ baselinePolicyId?: string;
548
+ }
549
+ interface RuntimeBeliefPhase0MeasurementSummary {
550
+ runCount: number;
551
+ producerDecisionCount: number;
552
+ lifecycleEventCount: number;
553
+ labelCount: number;
554
+ completedPointCount: number;
555
+ runJoinRate: number;
556
+ labelJoinRate: number;
557
+ missingRunRecordCount: number;
558
+ missingLabelCount: number;
559
+ withEvidence: number;
560
+ withOutcome: number;
561
+ withSplit: number;
562
+ withBehaviorProb: number;
563
+ withTargetProb: number;
564
+ baselinePolicyId: string;
565
+ packetStatus: BeliefDecisionResearchEvidencePacket['status'];
566
+ claimScope: BeliefDecisionResearchEvidencePacket['claimScope'];
567
+ }
568
+ interface RuntimeBeliefPhase0Measurement {
569
+ points: BeliefDecisionPoint[];
570
+ packet: BeliefDecisionResearchEvidencePacket;
571
+ summary: RuntimeBeliefPhase0MeasurementSummary;
572
+ diagnostics: string[];
573
+ }
574
+ declare function buildRuntimeBeliefPhase0Measurement(options: BuildRuntimeBeliefPhase0MeasurementOptions): RuntimeBeliefPhase0Measurement;
575
+
576
+ type RuntimeBenchmarkTrajectoryRecord = RuntimeTrajectoryRecord & {
577
+ benchmark?: unknown;
578
+ condition?: unknown;
579
+ instanceId?: unknown;
580
+ };
581
+ interface BuildRuntimeBenchmarkBeliefPhase0MeasurementOptions extends Omit<BuildRuntimeBeliefPhase0MeasurementOptions, 'runs' | 'events' | 'decisions' | 'labels'> {
582
+ records: RuntimeBenchmarkTrajectoryRecord[];
583
+ decisions?: RuntimeBeliefDecisionPoint[];
584
+ defaultSplitTag?: ProjectRuntimeTrajectoryEvidenceOptions['defaultSplitTag'];
585
+ labels?: RuntimeBeliefDecisionLabel[];
586
+ }
587
+ interface RuntimeBenchmarkBeliefPhase0Summary {
588
+ decisionCount: number;
589
+ labelCount: number;
590
+ }
591
+ interface RuntimeBenchmarkBeliefPhase0Measurement {
592
+ runs: RuntimeBeliefPhase0RunRecord[];
593
+ events: RuntimeBeliefHookEvent[];
594
+ decisions: RuntimeBeliefDecisionPoint[];
595
+ labels: RuntimeBeliefDecisionLabel[];
596
+ trajectory: RuntimeTrajectoryEvidenceProjection;
597
+ measurement: RuntimeBeliefPhase0Measurement;
598
+ summary: RuntimeBenchmarkBeliefPhase0Summary;
599
+ diagnostics: string[];
600
+ }
601
+ declare function buildRuntimeBenchmarkBeliefPhase0Measurement(options: BuildRuntimeBenchmarkBeliefPhase0MeasurementOptions): RuntimeBenchmarkBeliefPhase0Measurement;
602
+
603
+ export { type AnalyzeBeliefDecisionCorpusOptions, type AnalyzeBeliefPolicyOpeOptions, type AnalyzeBeliefPolicyOptions, BELIEF_DECISION_KINDS, BELIEF_EVALUATION_CRITERIA, BELIEF_EVIDENCE_QUALITIES, BELIEF_EVIDENCE_SOURCES, type BeliefCalibrationOptions, type BeliefCalibrationRegion, type BeliefCalibrationStatus, type BeliefDecisionCorpusEvaluation, type BeliefDecisionExtractionDiagnostic, type BeliefDecisionExtractionReport, type BeliefDecisionInventoryBucket, type BeliefDecisionInventoryReport, type BeliefDecisionKind, type BeliefDecisionOutcome, type BeliefDecisionPoint, type BeliefDecisionReason, type BeliefDecisionReasonCode, type BeliefDecisionResearchEvidencePacket, type BeliefDecisionTargetSelection, type BeliefEvaluationCriterionId, type BeliefEvaluationStatus, type BeliefEvidenceQuality, type BeliefEvidenceRef, type BeliefEvidenceSource, type BeliefOffPolicyTrajectoryReport, type BeliefOpeOptions, type BeliefOpeReport, type BeliefOpeStatus, type BeliefOpeSupportDiagnostics, type BeliefOpeTargetPolicy, type BeliefPolicyAction, type BeliefPolicyDecision, type BeliefPolicyEvaluationReport, type BeliefResearchClaimScope, type BeliefResearchEvidenceGate, type BeliefResearchEvidenceStatus, type BeliefResearchGateId, type BeliefRuntimeHookCollector, type BeliefSelectivePolicy, type BeliefSelectivePolicyMetrics, type BeliefShadowProbeDiagnostic, type BeliefShadowProbeEvidenceRef, type BeliefShadowProbeInput, type BeliefShadowProbeRecord, type BeliefShadowProbeResponse, type BeliefShadowProbeRun, type BeliefShadowProbeSummary, type BeliefUtilityOptions, type BuildBeliefDecisionResearchEvidencePacketOptions, type BuildCodeAgentBeliefEvidenceCorpusOptions, type BuildRuntimeBeliefPhase0MeasurementOptions, type BuildRuntimeBenchmarkBeliefPhase0MeasurementOptions, type CodeAgentBeliefDecisionTargetId, type CodeAgentBeliefEvidenceCorpus, type CodeAgentBeliefSession, type EvaluateBeliefSelectivePolicyOptions, type ExtractBeliefDecisionPointsOptions, type ExtractCodeAgentBeliefDecisionPointsOptions, type RunBeliefShadowProbeOptions, type RuntimeBeliefConversionDiagnostic, type RuntimeBeliefDecisionEvidenceRef, type RuntimeBeliefDecisionLabel, type RuntimeBeliefDecisionPoint, type RuntimeBeliefDecisionPointOptions, type RuntimeBeliefDecisionPointReport, type RuntimeBeliefHookContext, type RuntimeBeliefHookEvent, type RuntimeBeliefHooks, type RuntimeBeliefPhase0Measurement, type RuntimeBeliefPhase0MeasurementSummary, type RuntimeBeliefPhase0RunRecord, type RuntimeBeliefShadowProbeInputOptions, type RuntimeBeliefShadowProbeInputReport, type RuntimeBenchmarkBeliefPhase0Measurement, type RuntimeBenchmarkBeliefPhase0Summary, type SelectBeliefDecisionTargetOptions, analyzeBeliefDecisionCorpus, analyzeBeliefPolicy, beliefDecisionsToOffPolicyTrajectories, buildBeliefDecisionResearchEvidencePacket, buildCodeAgentBeliefEvidenceCorpus, buildRuntimeBeliefPhase0Measurement, buildRuntimeBenchmarkBeliefPhase0Measurement, calibrateBeliefDecisions, createBeliefRuntimeHookCollector, embeddedBeliefOpeTargetPolicy, evaluateBeliefOffPolicy, evaluateBeliefSelectivePolicy, extractBeliefDecisionPoints, extractCodeAgentBeliefDecisionPoints, formatBeliefShadowProbePrompt, inventoryBeliefDecisionPoints, isBeliefDecisionKind, isBeliefEvidenceSource, runBeliefShadowProbe, runtimeDecisionPointToBeliefDecisionPoint, runtimeDecisionPointToBeliefShadowProbeInput, selectBeliefDecisionTarget, thresholdSelectivePolicy };
@@ -8,6 +8,9 @@ import {
8
8
  import {
9
9
  calibrationFromPairs
10
10
  } from "../chunk-NPCTHQIO.js";
11
+ import {
12
+ projectRuntimeTrajectoryEvidence
13
+ } from "../chunk-T4SQEITX.js";
11
14
  import "../chunk-VI2UW6B6.js";
12
15
  import {
13
16
  offPolicyEstimateAll
@@ -127,7 +130,7 @@ function evaluateBeliefOffPolicy(points, targetPolicy, options = {}) {
127
130
  return { targetPolicyId: targetPolicy.id, ...estimates, support };
128
131
  }
129
132
  function supportDiagnostics(estimate, options) {
130
- const ratio = estimate.n > 0 ? estimate.effectiveSampleSize / estimate.n : 0;
133
+ const ratio2 = estimate.n > 0 ? estimate.effectiveSampleSize / estimate.n : 0;
131
134
  const reasons = [...options.diagnostics];
132
135
  if (estimate.n === 0) {
133
136
  reasons.push("no valid OPE trajectories");
@@ -140,9 +143,9 @@ function supportDiagnostics(estimate, options) {
140
143
  `effective sample size ${estimate.effectiveSampleSize.toFixed(2)} below ${options.minEffectiveSampleSize}`
141
144
  );
142
145
  }
143
- if (ratio < options.minEffectiveSampleRatio) {
146
+ if (ratio2 < options.minEffectiveSampleRatio) {
144
147
  reasons.push(
145
- `effective sample ratio ${ratio.toFixed(2)} below ${options.minEffectiveSampleRatio}`
148
+ `effective sample ratio ${ratio2.toFixed(2)} below ${options.minEffectiveSampleRatio}`
146
149
  );
147
150
  }
148
151
  if (estimate.maxImportanceWeight > 10) {
@@ -153,7 +156,7 @@ function supportDiagnostics(estimate, options) {
153
156
  n: estimate.n,
154
157
  dropped: options.dropped,
155
158
  effectiveSampleSize: estimate.effectiveSampleSize,
156
- effectiveSampleRatio: ratio,
159
+ effectiveSampleRatio: ratio2,
157
160
  maxImportanceWeight: estimate.maxImportanceWeight,
158
161
  reasons
159
162
  };
@@ -1622,6 +1625,157 @@ function nonNegativeNumberOrUndefined(value) {
1622
1625
  return number !== void 0 && number >= 0 ? number : void 0;
1623
1626
  }
1624
1627
 
1628
+ // src/belief-state/phase0-measurement.ts
1629
+ var DEFAULT_BASELINE_POLICY_ID = "always-accept-observed-action";
1630
+ function buildRuntimeBeliefPhase0Measurement(options) {
1631
+ const runsById = new Map(options.runs.map((run) => [run.runId, run]));
1632
+ const labelsByDecisionId = /* @__PURE__ */ new Map();
1633
+ const diagnostics = [];
1634
+ for (const label of options.labels) {
1635
+ if (labelsByDecisionId.has(label.decisionId)) {
1636
+ diagnostics.push(`${label.decisionId}: duplicate label; using the last label`);
1637
+ }
1638
+ labelsByDecisionId.set(label.decisionId, label);
1639
+ }
1640
+ const points = [];
1641
+ let missingRunRecordCount = 0;
1642
+ let missingLabelCount = 0;
1643
+ for (const decision of options.decisions) {
1644
+ const run = runsById.get(decision.runId);
1645
+ if (!run) {
1646
+ missingRunRecordCount += 1;
1647
+ diagnostics.push(`${decision.id}: missing RunRecord join for runId ${decision.runId}`);
1648
+ continue;
1649
+ }
1650
+ const label = labelsByDecisionId.get(decision.id);
1651
+ if (!label) {
1652
+ missingLabelCount += 1;
1653
+ diagnostics.push(`${decision.id}: missing observed action/outcome label`);
1654
+ continue;
1655
+ }
1656
+ const splitTag = label.splitTag ?? run.splitTag;
1657
+ const report = runtimeDecisionPointToBeliefDecisionPoint(
1658
+ { ...decision, scenarioId: decision.scenarioId ?? run.scenarioId },
1659
+ {
1660
+ chosenAction: label.chosenAction,
1661
+ confidence: label.confidence,
1662
+ behaviorProb: label.behaviorProb,
1663
+ targetProb: label.targetProb,
1664
+ qHat: label.qHat,
1665
+ costUsd: label.costUsd,
1666
+ outcome: label.outcome,
1667
+ lifecycleEvents: options.events,
1668
+ metadata: compactMetadata2({
1669
+ baselinePolicyId: options.baselinePolicyId ?? DEFAULT_BASELINE_POLICY_ID,
1670
+ splitTag,
1671
+ ...label.metadata
1672
+ })
1673
+ }
1674
+ );
1675
+ diagnostics.push(...report.diagnostics.map((item) => `${item.decisionId}: ${item.reason}`));
1676
+ if (report.point) points.push(report.point);
1677
+ }
1678
+ const packet = buildBeliefDecisionResearchEvidencePacket({
1679
+ ...options,
1680
+ points
1681
+ });
1682
+ return {
1683
+ points,
1684
+ packet,
1685
+ summary: summarizePhase0Measurement(options, points, packet, {
1686
+ missingRunRecordCount,
1687
+ missingLabelCount
1688
+ }),
1689
+ diagnostics
1690
+ };
1691
+ }
1692
+ function summarizePhase0Measurement(options, points, packet, counts) {
1693
+ const producerDecisionCount = options.decisions.length;
1694
+ return {
1695
+ runCount: options.runs.length,
1696
+ producerDecisionCount,
1697
+ lifecycleEventCount: options.events?.length ?? 0,
1698
+ labelCount: options.labels.length,
1699
+ completedPointCount: points.length,
1700
+ runJoinRate: ratio(producerDecisionCount - counts.missingRunRecordCount, producerDecisionCount),
1701
+ labelJoinRate: ratio(points.length, producerDecisionCount),
1702
+ missingRunRecordCount: counts.missingRunRecordCount,
1703
+ missingLabelCount: counts.missingLabelCount,
1704
+ withEvidence: points.filter((point) => point.evidence.length > 0).length,
1705
+ withOutcome: points.filter((point) => point.outcome).length,
1706
+ withSplit: points.filter((point) => typeof point.metadata?.splitTag === "string").length,
1707
+ withBehaviorProb: points.filter((point) => point.behaviorProb !== void 0).length,
1708
+ withTargetProb: points.filter((point) => point.targetProb !== void 0).length,
1709
+ baselinePolicyId: options.baselinePolicyId ?? DEFAULT_BASELINE_POLICY_ID,
1710
+ packetStatus: packet.status,
1711
+ claimScope: packet.claimScope
1712
+ };
1713
+ }
1714
+ function ratio(numerator, denominator) {
1715
+ return denominator > 0 ? numerator / denominator : 0;
1716
+ }
1717
+ function compactMetadata2(values) {
1718
+ const entries = Object.entries(values).filter(([, value]) => value !== void 0);
1719
+ return entries.length > 0 ? Object.fromEntries(entries) : void 0;
1720
+ }
1721
+
1722
+ // src/belief-state/runtime-benchmark-corpus.ts
1723
+ function buildRuntimeBenchmarkBeliefPhase0Measurement(options) {
1724
+ const diagnostics = [];
1725
+ const trajectory = projectRuntimeTrajectoryEvidence({
1726
+ records: options.records,
1727
+ defaultSplitTag: options.defaultSplitTag,
1728
+ recordIdOf: runtimeBenchmarkRecordId,
1729
+ scenarioIdOf: runtimeBenchmarkScenarioId
1730
+ });
1731
+ const decisions = options.decisions ?? [];
1732
+ const labels = options.labels ?? [];
1733
+ if (decisions.length === 0) {
1734
+ diagnostics.push(
1735
+ "no runtime decision points supplied; benchmark lifecycle events alone cannot produce belief decision rows"
1736
+ );
1737
+ }
1738
+ if (labels.length === 0 && decisions.length > 0) {
1739
+ diagnostics.push(
1740
+ "no decision labels supplied; observed action/outcome joins will be incomplete"
1741
+ );
1742
+ }
1743
+ const measurement = buildRuntimeBeliefPhase0Measurement({
1744
+ ...options,
1745
+ runs: trajectory.runs,
1746
+ events: trajectory.events,
1747
+ decisions,
1748
+ labels
1749
+ });
1750
+ return {
1751
+ runs: trajectory.runs,
1752
+ events: trajectory.events,
1753
+ decisions,
1754
+ labels,
1755
+ trajectory,
1756
+ measurement,
1757
+ summary: {
1758
+ decisionCount: decisions.length,
1759
+ labelCount: labels.length
1760
+ },
1761
+ diagnostics: [...trajectory.diagnostics, ...diagnostics, ...measurement.diagnostics]
1762
+ };
1763
+ }
1764
+ function runtimeBenchmarkRecordId(record2) {
1765
+ const parts = [
1766
+ nonEmptyString(record2.benchmark),
1767
+ nonEmptyString(record2.instanceId),
1768
+ nonEmptyString(record2.condition)
1769
+ ].filter((part) => part !== void 0);
1770
+ return parts.length > 0 ? parts.join(":") : void 0;
1771
+ }
1772
+ function runtimeBenchmarkScenarioId(record2) {
1773
+ return nonEmptyString(record2.instanceId);
1774
+ }
1775
+ function nonEmptyString(value) {
1776
+ return typeof value === "string" && value.length > 0 ? value : void 0;
1777
+ }
1778
+
1625
1779
  // src/belief-state/shadow-probe.ts
1626
1780
  var DEFAULT_CONCURRENCY = 4;
1627
1781
  var DEFAULT_MAX_CONTEXT_CHARS2 = 12e3;
@@ -1842,6 +1996,8 @@ export {
1842
1996
  beliefDecisionsToOffPolicyTrajectories,
1843
1997
  buildBeliefDecisionResearchEvidencePacket,
1844
1998
  buildCodeAgentBeliefEvidenceCorpus,
1999
+ buildRuntimeBeliefPhase0Measurement,
2000
+ buildRuntimeBenchmarkBeliefPhase0Measurement,
1845
2001
  calibrateBeliefDecisions,
1846
2002
  createBeliefRuntimeHookCollector,
1847
2003
  embeddedBeliefOpeTargetPolicy,