@tangle-network/agent-eval 0.116.0 → 0.117.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (135) hide show
  1. package/CHANGELOG.md +38 -0
  2. package/dist/analyst/index.d.ts +18 -11
  3. package/dist/analyst/index.js +10 -7
  4. package/dist/analyst/index.js.map +1 -1
  5. package/dist/{analyst-CFBc14Wc.d.ts → analyst-C8HHvfJp.d.ts} +1 -1
  6. package/dist/{analyze-runs-0rz_m29H.d.ts → analyze-runs--2x39HZ7.d.ts} +3 -3
  7. package/dist/{baseline-DsNteOgR.d.ts → baseline-DKq3gJpP.d.ts} +6 -3
  8. package/dist/belief-state/index.d.ts +6 -6
  9. package/dist/belief-state/index.js +1 -1
  10. package/dist/benchmarks/index.d.ts +11 -8
  11. package/dist/benchmarks/index.js +11 -10
  12. package/dist/builder-eval/index.d.ts +4 -4
  13. package/dist/builder-eval/index.js +1 -1
  14. package/dist/{calibration-Dz8TQV4y.d.ts → calibration-C8MTS7cw.d.ts} +2 -2
  15. package/dist/campaign/index.d.ts +54 -30
  16. package/dist/campaign/index.js +18 -13
  17. package/dist/chunk-3YYRZDON.js +45 -0
  18. package/dist/chunk-3YYRZDON.js.map +1 -0
  19. package/dist/{chunk-RPDDVKI7.js → chunk-4JLWXDYA.js} +2 -2
  20. package/dist/{chunk-NBSS5NDZ.js → chunk-CCZIVI3F.js} +54 -115
  21. package/dist/chunk-CCZIVI3F.js.map +1 -0
  22. package/dist/{chunk-J6P6PK2R.js → chunk-FQNLDL4D.js} +3 -3
  23. package/dist/{chunk-ONM6PEAE.js → chunk-GQCZRZ7L.js} +2 -2
  24. package/dist/chunk-HHWE3POT.js +94 -0
  25. package/dist/chunk-HHWE3POT.js.map +1 -0
  26. package/dist/{chunk-3274WNK7.js → chunk-HQPHZGL6.js} +687 -44
  27. package/dist/chunk-HQPHZGL6.js.map +1 -0
  28. package/dist/{chunk-FAOEFFRT.js → chunk-IDZTTFRR.js} +390 -78
  29. package/dist/chunk-IDZTTFRR.js.map +1 -0
  30. package/dist/{chunk-3LXTCTWL.js → chunk-JSDVRFAP.js} +2 -2
  31. package/dist/{chunk-GSW3OBHK.js → chunk-JSJZ4PJ6.js} +406 -726
  32. package/dist/chunk-JSJZ4PJ6.js.map +1 -0
  33. package/dist/{chunk-MHNQWM4I.js → chunk-LQUTGLOZ.js} +5 -1
  34. package/dist/chunk-LQUTGLOZ.js.map +1 -0
  35. package/dist/{chunk-4D5RVB3W.js → chunk-LTVG32KX.js} +30 -5
  36. package/dist/chunk-LTVG32KX.js.map +1 -0
  37. package/dist/{chunk-CIUOICJT.js → chunk-MGEHEHSN.js} +62 -15
  38. package/dist/chunk-MGEHEHSN.js.map +1 -0
  39. package/dist/{chunk-GY4SYVPJ.js → chunk-NJC7U437.js} +97 -25
  40. package/dist/chunk-NJC7U437.js.map +1 -0
  41. package/dist/{chunk-NYFUT3B3.js → chunk-ODVOOEWQ.js} +31 -10
  42. package/dist/chunk-ODVOOEWQ.js.map +1 -0
  43. package/dist/{chunk-LNQEP766.js → chunk-S2F4J57L.js} +44 -4
  44. package/dist/chunk-S2F4J57L.js.map +1 -0
  45. package/dist/chunk-VCTY3W6J.js +798 -0
  46. package/dist/chunk-VCTY3W6J.js.map +1 -0
  47. package/dist/chunk-VF3XSYTI.js +545 -0
  48. package/dist/chunk-VF3XSYTI.js.map +1 -0
  49. package/dist/{chunk-TLDB7WRY.js → chunk-YZPO4UHR.js} +28 -31
  50. package/dist/chunk-YZPO4UHR.js.map +1 -0
  51. package/dist/cli.js +4 -2
  52. package/dist/cli.js.map +1 -1
  53. package/dist/{code-agent-session-CdxteG0y.d.ts → code-agent-session-CjZsVd19.d.ts} +1 -1
  54. package/dist/contract/index.d.ts +43 -29
  55. package/dist/contract/index.js +56 -19
  56. package/dist/contract/index.js.map +1 -1
  57. package/dist/{control-DbcDxouY.d.ts → control-6vuGfmDH.d.ts} +5 -5
  58. package/dist/control.d.ts +6 -6
  59. package/dist/cost-ledger-DWy3XdJc.d.ts +183 -0
  60. package/dist/{default-registry-DDfv22MQ.d.ts → default-registry-DaK8b3fv.d.ts} +2 -2
  61. package/dist/{emitter-BRchAAAx.d.ts → emitter-CjD7vUwv.d.ts} +2 -2
  62. package/dist/{failure-cluster-C48PiReX.d.ts → failure-cluster-DOAcSJ87.d.ts} +2 -2
  63. package/dist/{feedback-trajectory-pDcz1lQ1.d.ts → feedback-trajectory-BUnM58xL.d.ts} +3 -3
  64. package/dist/fuzz.d.ts +8 -16
  65. package/dist/fuzz.js +72 -42
  66. package/dist/fuzz.js.map +1 -1
  67. package/dist/{gepa-CQelRtuC.d.ts → gepa-eESocoDi.d.ts} +56 -6
  68. package/dist/hosted/index.d.ts +13 -10
  69. package/dist/{index-DbCXJfZ1.d.ts → index-PdX4VnPA.d.ts} +3 -3
  70. package/dist/index.d.ts +102 -57
  71. package/dist/index.js +328 -235
  72. package/dist/index.js.map +1 -1
  73. package/dist/{insight-report-oMVxDTxl.d.ts → insight-report-DY4nDW9Q.d.ts} +1 -1
  74. package/dist/{integrity-C6PZ73iC.d.ts → integrity-DqlBiLyK.d.ts} +2 -2
  75. package/dist/{kind-factory-DWOvXjR_.d.ts → kind-factory-ClZmO25A.d.ts} +2 -2
  76. package/dist/llm-client-qoDd18Qz.d.ts +289 -0
  77. package/dist/meta-eval/index.d.ts +8 -7
  78. package/dist/meta-eval/index.js +1 -1
  79. package/dist/multishot/index.d.ts +9 -6
  80. package/dist/openapi.json +1 -1
  81. package/dist/pipelines/index.d.ts +16 -6
  82. package/dist/pipelines/index.js +119 -23
  83. package/dist/pipelines/index.js.map +1 -1
  84. package/dist/{policy-edit-Clb2v6Oa.d.ts → policy-edit-wG9uFEFm.d.ts} +13 -266
  85. package/dist/{pre-registration--vU0mMtD.d.ts → pre-registration-BWQhJ3vz.d.ts} +24 -5
  86. package/dist/{provenance-BbVagC68.d.ts → provenance-DpjwyseI.d.ts} +6 -6
  87. package/dist/{query-Ck190MOd.d.ts → query-CF7PG61p.d.ts} +5 -3
  88. package/dist/raw-provider-sink-C46HDghv.d.ts +132 -0
  89. package/dist/{release-report-CamNDe90.d.ts → release-report-C8G2i5Xi.d.ts} +2 -2
  90. package/dist/reporting.d.ts +10 -9
  91. package/dist/{researcher-Dwbo_Fxx.d.ts → researcher-C8XyxQsu.d.ts} +8 -8
  92. package/dist/rl.d.ts +18 -15
  93. package/dist/rl.js +2 -2
  94. package/dist/{rubric-predictive-validity-BIdf9h4R.d.ts → rubric-predictive-validity-p49lLVrE.d.ts} +1 -1
  95. package/dist/{run-campaign-UADIM77S.js → run-campaign-IM26A6PD.js} +4 -2
  96. package/dist/{run-record-CZmcpWPo.d.ts → run-record-BDH49H2E.d.ts} +1 -1
  97. package/dist/{runtime-trajectory-CC0jx9ql.d.ts → runtime-trajectory-DGBIUt4B.d.ts} +1 -1
  98. package/dist/{schema-SGWcK9wa.d.ts → schema-B3Q3l9Z_.d.ts} +2 -0
  99. package/dist/{semantic-concept-judge-CKjePUMh.d.ts → semantic-concept-judge-CXnPEJbf.d.ts} +24 -6
  100. package/dist/{statistics-oUbOJe-S.d.ts → statistics-KUnG73jH.d.ts} +1 -1
  101. package/dist/{storage-Dw_f7WMt.d.ts → storage-DrX3v_5B.d.ts} +12 -1
  102. package/dist/{store-9cAScOcb.d.ts → store-C1YxJDEK.d.ts} +1 -132
  103. package/dist/{store-BsVi7ncX.d.ts → store-DGqD0Pyo.d.ts} +1 -1
  104. package/dist/storyboard/index.d.ts +1 -1
  105. package/dist/{summary-report-DTNgQycC.d.ts → summary-report-C5bKFfm-.d.ts} +2 -2
  106. package/dist/{test-graded-scenario-mzYBKspu.d.ts → test-graded-scenario-B0ybnPY7.d.ts} +3 -3
  107. package/dist/traces.d.ts +25 -14
  108. package/dist/traces.js +16 -4
  109. package/dist/{types-Ca_63YSD.d.ts → types-BSw1rOUB.d.ts} +41 -39
  110. package/dist/{types-C7DGg5ex.d.ts → types-BkfcQnxV.d.ts} +15 -0
  111. package/dist/wire/index.d.ts +28 -19
  112. package/dist/wire/index.js +4 -2
  113. package/docs/distributed-driver.md +1 -1
  114. package/package.json +3 -3
  115. package/dist/chunk-3274WNK7.js.map +0 -1
  116. package/dist/chunk-4D5RVB3W.js.map +0 -1
  117. package/dist/chunk-7GKEAIAD.js +0 -205
  118. package/dist/chunk-7GKEAIAD.js.map +0 -1
  119. package/dist/chunk-CIUOICJT.js.map +0 -1
  120. package/dist/chunk-FAOEFFRT.js.map +0 -1
  121. package/dist/chunk-GSW3OBHK.js.map +0 -1
  122. package/dist/chunk-GY4SYVPJ.js.map +0 -1
  123. package/dist/chunk-LNQEP766.js.map +0 -1
  124. package/dist/chunk-MHNQWM4I.js.map +0 -1
  125. package/dist/chunk-MPHTT5HE.js +0 -74
  126. package/dist/chunk-MPHTT5HE.js.map +0 -1
  127. package/dist/chunk-NBSS5NDZ.js.map +0 -1
  128. package/dist/chunk-NYFUT3B3.js.map +0 -1
  129. package/dist/chunk-TLDB7WRY.js.map +0 -1
  130. package/dist/cost-ledger-DuSqlw5B.d.ts +0 -113
  131. /package/dist/{chunk-RPDDVKI7.js.map → chunk-4JLWXDYA.js.map} +0 -0
  132. /package/dist/{chunk-J6P6PK2R.js.map → chunk-FQNLDL4D.js.map} +0 -0
  133. /package/dist/{chunk-ONM6PEAE.js.map → chunk-GQCZRZ7L.js.map} +0 -0
  134. /package/dist/{chunk-3LXTCTWL.js.map → chunk-JSDVRFAP.js.map} +0 -0
  135. /package/dist/{run-campaign-UADIM77S.js.map → run-campaign-IM26A6PD.js.map} +0 -0
package/CHANGELOG.md CHANGED
@@ -4,6 +4,44 @@ All notable changes to `@tangle-network/agent-eval` and its sibling `agent-eval-
4
4
 
5
5
  ---
6
6
 
7
+ ## [0.117.1] — 2026-07-13 — retry-safe code-candidate cleanup
8
+
9
+ ### Fixed
10
+
11
+ - `gitWorktreeAdapter().discard()` now reconciles worktree and branch removal independently.
12
+ Repeated cleanup is safe, partial cleanup can be retried, and a Git command that reports an error after completing its mutation no longer strands candidate branches or worktrees.
13
+
14
+ ## [0.117.0] — 2026-07-13 — durable cost and bounded behavioral evidence
15
+
16
+ ### Added
17
+
18
+ - `createReferenceEquivalenceJudge()` and `runReferenceEquivalenceJudge()` score whether an answer preserves the meaning of one or more references, with the same cost and transport accounting as other judges.
19
+
20
+ ### Changed
21
+
22
+ - `CostLedger.runPaidCall()` is now the single paid-call path across campaigns, proposers, judges, analysts, and distillation.
23
+ It durably reserves maximum spend before dispatch, records provider receipts, blocks unresolved crash state, and enforces the run ceiling before another paid call starts.
24
+ - `ToolSpan.argsCaptured` distinguishes a call with unavailable arguments from a captured no-argument call.
25
+ Repeated-call analysis, failure clustering, tool-use metrics, and per-step redundancy grading no longer compare uncaptured arguments.
26
+ Every OTLP export path uses one mapping that preserves this distinction.
27
+
28
+ ### Breaking
29
+
30
+ - `CostLedger.record()` is removed because recording spend after a provider call cannot enforce a cost limit or survive a crash.
31
+ Use `CostLedger.runPaidCall()` for billable work, `CostLedger` receipt import for already-settled calls, or `costForUsage()` for pure estimates.
32
+ - `computeTraceMetrics()` now rejects mixed-trace input, and `BehavioralMetrics` adds required `traceId` and `tokenSequences` fields.
33
+ The convenience token trajectories now expose the longest proven-serial sequence instead of flattening parallel branches.
34
+ - `ToolUseMetrics` and `ToolStats` add required `callsWithCapturedArgs` fields.
35
+ `duplicateRate` now uses captured-argument calls as its denominator.
36
+
37
+ ### Fixed
38
+
39
+ - Repeated-call findings now require a contiguous, time-bounded, serial episode within one agent branch instead of grouping identical or concurrent calls across an entire run.
40
+ - Behavioral token findings now analyze each trace and serial agent timeline independently, use numeric time ordering across accepted timestamp formats, and only attribute output decay to context that actually grew.
41
+ - Behavioral issue IDs remain stable across trace runs while evidence retains exact trace identities and sampled prevalence.
42
+ - Partial timing isolates only the uncertain interval, and same-named root spans retain independent structural identity.
43
+ - Multi-trace behavioral findings use pattern-level claims while each trace's exact values remain in its evidence reference.
44
+
7
45
  ## [0.116.0] — 2026-07-12 — evidence-linked AgentProfile optimization
8
46
 
9
47
  ### Added
@@ -1,21 +1,24 @@
1
1
  import { M as MultiLayerVerifier, V as VerifyOptions, S as Severity } from '../multi-layer-verifier-BsqKuLyN.js';
2
- import { R as RunCritic, S as SemanticConceptJudgeOptions, a as RunTrace, b as SemanticConceptJudgeInput, B as BehavioralMetrics } from '../semantic-concept-judge-CKjePUMh.js';
3
- export { C as CreateAnalystAiConfig, D as DEFAULT_TRACE_ANALYST_KINDS, c as DiffPolicy, F as FAILURE_MODE_KIND_SPEC, d as FINDING_SUBJECT_GRAMMAR_PROMPT, e as FINDING_SUBJECT_KINDS, f as FINDING_SUBJECT_SYNTAX, g as FindingSubject, h as FindingSubjectKind, i as FindingSubjectStringSchema, j as FindingsDiff, k as FindingsStore, I as IMPROVEMENT_KIND_SPEC, K as KIND_EXPECTED_SUBJECTS, l as KNOWLEDGE_GAP_KIND_SPEC, m as KNOWLEDGE_POISONING_KIND_SPEC, P as PersistedFinding, n as SKILL_USAGE_ANALYST, o as SkillUsageAnalyst, p as SkillUsageRecord, q as SkillUsageReport, r as SkillUsageScanConfig, s as buildSkillUsageReport, t as createAnalystAi, u as defaultIsMaterial, v as diffFindings, w as emitSkillUsageFindings, x as findingSubjectGrammarPromptFor, y as parseFindingSubject, z as renderFindingSubject } from '../semantic-concept-judge-CKjePUMh.js';
4
- import { b as JudgeFn, a as JudgeInput } from '../types-C7DGg5ex.js';
5
- import { A as Analyst, i as AnalystSeverity, d as AnalystFinding, L as LlmClientOptions } from '../policy-edit-Clb2v6Oa.js';
6
- export { b as AnalystContext, h as AnalystCost, j as AnalystInputKind, k as AnalystRequirements, g as AnalystRunEvent, f as AnalystRunInputs, e as AnalystRunResult, c as AnalystRunSummary, l as ChatCallOpts, C as ChatClient, m as ChatRequest, n as ChatResponse, o as ChatTransport, p as CliBridgeTransportOpts, q as CreateChatClientOpts, D as DirectProviderTransportOpts, E as EvidenceRef, F as FindingToPolicyEditOptions, M as MockTransportOpts, r as POLICY_EDIT_AXES, s as POLICY_EDIT_CANDIDATE_RECORD_SCHEMA, t as POLICY_EDIT_TARGET_SURFACES, u as PolicyEdit, v as PolicyEditAdmission, w as PolicyEditAdmissionOptions, x as PolicyEditAxis, P as PolicyEditCandidateRecord, y as PolicyEditChange, z as PolicyEditExpectedGain, B as PolicyEditGainDirection, G as PolicyEditGainUnit, H as PolicyEditInit, I as PolicyEditRisk, J as PolicyEditSchemaVersion, K as PolicyEditSource, N as PolicyEditTarget, O as PolicyEditTargetSurface, Q as PolicyEditValidationError, R as RouterTransportOpts, S as SandboxSdkTransportOpts, T as admitPolicyEdit, U as applyPolicyEditToSurface, V as computeFindingId, W as computePolicyEditId, X as createChatClient, Y as isPolicyEdit, Z as makeFinding, _ as makePolicyEdit, $ as makePolicyEditCandidateRecord, a0 as policyEditFromFinding, a1 as policyEditsFromFindings, a2 as scorePolicyEditReadiness, a3 as validatePolicyEdit, a4 as validatePolicyEditCandidateRecord } from '../policy-edit-Clb2v6Oa.js';
2
+ import { R as RunCritic, S as SemanticConceptJudgeOptions, a as RunTrace, b as SemanticConceptJudgeInput, B as BehavioralMetrics } from '../semantic-concept-judge-CXnPEJbf.js';
3
+ export { C as CreateAnalystAiConfig, D as DEFAULT_TRACE_ANALYST_KINDS, c as DiffPolicy, F as FAILURE_MODE_KIND_SPEC, d as FINDING_SUBJECT_GRAMMAR_PROMPT, e as FINDING_SUBJECT_KINDS, f as FINDING_SUBJECT_SYNTAX, g as FindingSubject, h as FindingSubjectKind, i as FindingSubjectStringSchema, j as FindingsDiff, k as FindingsStore, I as IMPROVEMENT_KIND_SPEC, K as KIND_EXPECTED_SUBJECTS, l as KNOWLEDGE_GAP_KIND_SPEC, m as KNOWLEDGE_POISONING_KIND_SPEC, P as PersistedFinding, n as SKILL_USAGE_ANALYST, o as SkillUsageAnalyst, p as SkillUsageRecord, q as SkillUsageReport, r as SkillUsageScanConfig, s as buildSkillUsageReport, t as createAnalystAi, u as defaultIsMaterial, v as diffFindings, w as emitSkillUsageFindings, x as findingSubjectGrammarPromptFor, y as parseFindingSubject, z as renderFindingSubject } from '../semantic-concept-judge-CXnPEJbf.js';
4
+ import { b as JudgeFn, a as JudgeInput } from '../types-BkfcQnxV.js';
5
+ import { A as Analyst, h as AnalystSeverity, c as AnalystFinding } from '../policy-edit-wG9uFEFm.js';
6
+ export { a as AnalystContext, g as AnalystCost, i as AnalystInputKind, j as AnalystRequirements, f as AnalystRunEvent, e as AnalystRunInputs, d as AnalystRunResult, b as AnalystRunSummary, k as ChatCallOpts, C as ChatClient, l as ChatRequest, m as ChatResponse, n as ChatTransport, o as CliBridgeTransportOpts, p as CreateChatClientOpts, D as DirectProviderTransportOpts, E as EvidenceRef, F as FindingToPolicyEditOptions, M as MockTransportOpts, q as POLICY_EDIT_AXES, r as POLICY_EDIT_CANDIDATE_RECORD_SCHEMA, s as POLICY_EDIT_TARGET_SURFACES, t as PolicyEdit, u as PolicyEditAdmission, v as PolicyEditAdmissionOptions, w as PolicyEditAxis, P as PolicyEditCandidateRecord, x as PolicyEditChange, y as PolicyEditExpectedGain, z as PolicyEditGainDirection, B as PolicyEditGainUnit, G as PolicyEditInit, H as PolicyEditRisk, I as PolicyEditSchemaVersion, J as PolicyEditSource, K as PolicyEditTarget, L as PolicyEditTargetSurface, N as PolicyEditValidationError, R as RouterTransportOpts, S as SandboxSdkTransportOpts, O as admitPolicyEdit, Q as applyPolicyEditToSurface, T as computeFindingId, U as computePolicyEditId, V as createChatClient, W as isPolicyEdit, X as makeFinding, Y as makePolicyEdit, Z as makePolicyEditCandidateRecord, _ as policyEditFromFinding, $ as policyEditsFromFindings, a0 as scorePolicyEditReadiness, a1 as validatePolicyEdit, a2 as validatePolicyEditCandidateRecord } from '../policy-edit-wG9uFEFm.js';
7
7
  import { TCloud } from '@tangle-network/tcloud';
8
- import { T as TraceAnalysisStore } from '../store-9cAScOcb.js';
9
- export { a as AnalystHooks, A as AnalystRegistry, b as AnalystRegistryOptions, B as BudgetPolicy, D as DefaultAnalystRegistryOptions, R as RegistryRunOpts, c as buildDefaultAnalystRegistry } from '../default-registry-DDfv22MQ.js';
10
- export { A as ANALYST_SEVERITIES, C as CreateTraceAnalystKindOpts, R as RAW_FINDING_SCHEMA_PROMPT, a as RawAnalystFinding, b as RawAnalystFindingSchema, c as TraceAnalystGolden, T as TraceAnalystKindSpec, d as createTraceAnalystKind, p as parseRawFinding, r as renderPriorFindings } from '../kind-factory-DWOvXjR_.js';
8
+ import { T as TraceAnalysisStore } from '../store-C1YxJDEK.js';
9
+ export { a as AnalystHooks, A as AnalystRegistry, b as AnalystRegistryOptions, B as BudgetPolicy, D as DefaultAnalystRegistryOptions, R as RegistryRunOpts, c as buildDefaultAnalystRegistry } from '../default-registry-DaK8b3fv.js';
10
+ export { A as ANALYST_SEVERITIES, C as CreateTraceAnalystKindOpts, R as RAW_FINDING_SCHEMA_PROMPT, a as RawAnalystFinding, b as RawAnalystFindingSchema, c as TraceAnalystGolden, T as TraceAnalystKindSpec, d as createTraceAnalystKind, p as parseRawFinding, r as renderPriorFindings } from '../kind-factory-ClZmO25A.js';
11
+ import { C as CostLedger } from '../cost-ledger-DWy3XdJc.js';
12
+ import { a as LlmClientOptions } from '../llm-client-qoDd18Qz.js';
11
13
  import { AxFunction } from '@ax-llm/ax';
12
14
  import '../verdict-C9MlYujm.js';
13
15
  import 'zod';
14
- import '../schema-SGWcK9wa.js';
15
- import '../store-BsVi7ncX.js';
16
- import '../run-record-CZmcpWPo.js';
16
+ import '../schema-B3Q3l9Z_.js';
17
+ import '../store-DGqD0Pyo.js';
18
+ import '../run-record-BDH49H2E.js';
17
19
  import '@tangle-network/agent-interface';
18
20
  import '../errors-oeQrLqXC.js';
21
+ import '../raw-provider-sink-C46HDghv.js';
19
22
 
20
23
  /**
21
24
  * Adapter factories — lift each existing agent-eval primitive into the
@@ -186,6 +189,10 @@ interface StructureFindingsOptions {
186
189
  model: string;
187
190
  baseUrl: string;
188
191
  apiKey?: string;
192
+ /** Optional ledger for direct use. */
193
+ costLedger?: CostLedger;
194
+ costPhase?: string;
195
+ maxTokens?: number;
189
196
  /** Max reask attempts after a zero/invalid extraction. Default 1. */
190
197
  maxReasks?: number;
191
198
  /** Test seam: inject a fetch (no network in unit tests). */
@@ -6,18 +6,19 @@ import {
6
6
  SkillUsageAnalyst,
7
7
  buildSkillUsageReport,
8
8
  createAnalystAi,
9
- createChatClient,
10
9
  defaultIsMaterial,
11
10
  diffFindings,
12
11
  emitSkillUsageFindings,
13
12
  runSemanticConceptJudge
14
- } from "../chunk-NBSS5NDZ.js";
13
+ } from "../chunk-CCZIVI3F.js";
15
14
  import {
16
15
  behavioralAnalyst,
17
16
  buildDefaultAnalystRegistry,
17
+ createChatClient,
18
18
  deriveEfficiencyFindings
19
- } from "../chunk-7GKEAIAD.js";
20
- import "../chunk-MPHTT5HE.js";
19
+ } from "../chunk-VF3XSYTI.js";
20
+ import "../chunk-HHWE3POT.js";
21
+ import "../chunk-3YYRZDON.js";
21
22
  import {
22
23
  ANALYST_SEVERITIES,
23
24
  AnalystRegistry,
@@ -64,11 +65,13 @@ import {
64
65
  structureFindings,
65
66
  validatePolicyEdit,
66
67
  validatePolicyEditCandidateRecord
67
- } from "../chunk-CIUOICJT.js";
68
- import "../chunk-LNQEP766.js";
68
+ } from "../chunk-MGEHEHSN.js";
69
+ import "../chunk-S2F4J57L.js";
69
70
  import "../chunk-XJYR7XFV.js";
70
71
  import "../chunk-VSMTAMNK.js";
71
- import "../chunk-GY4SYVPJ.js";
72
+ import "../chunk-NJC7U437.js";
73
+ import "../chunk-VCTY3W6J.js";
74
+ import "../chunk-VI2UW6B6.js";
72
75
  import "../chunk-PC4UYEBM.js";
73
76
  import "../chunk-ONWEPEDO.js";
74
77
  import "../chunk-PZ5AY32C.js";
@@ -1 +1 @@
1
- {"version":3,"sources":["../../src/analyst/adapters.ts"],"sourcesContent":["/**\n * Adapter factories — lift each existing agent-eval primitive into the\n * Analyst contract without re-implementing it.\n *\n * Five primitives, five factories. Each one:\n * - Builds an Analyst with a stable id (caller chooses; defaults\n * given), a sensible default `inputKind`, a version derived from\n * the wrapped primitive's version + an adapter revision, and an\n * `analyze()` that calls the primitive and lifts its output to\n * AnalystFinding[] using `makeFinding()`.\n * - Maps severities: the existing `Severity` ('critical' | 'major' |\n * 'minor' | 'info') projects onto AnalystSeverity ('critical' |\n * 'high' | 'medium' | 'low' | 'info'); 'major' → 'high', 'minor' →\n * 'medium'. Domain analysts that want finer-grained mapping override.\n *\n * Adapters never own state. Calling the same factory twice with the\n * same primitive instance is safe.\n */\n\nimport type {\n Finding as LayerFinding,\n Severity as LayerSeverity,\n MultiLayerVerifier,\n VerifyOptions,\n} from '../multi-layer-verifier'\nimport { RunCritic, type RunTrace } from '../run-critic'\nimport {\n runSemanticConceptJudge,\n SEMANTIC_CONCEPT_JUDGE_VERSION,\n type SemanticConceptJudgeInput,\n type SemanticConceptJudgeOptions,\n} from '../semantic-concept-judge'\nimport type { JudgeFn, JudgeInput, JudgeScore, TCloud } from '../types'\nimport type { Analyst, AnalystFinding, AnalystSeverity } from './types'\nimport { makeFinding } from './types'\n\nconst ADAPTER_REV = '1'\n\n// ── Severity bridges ───────────────────────────────────────────────\n\nexport function liftSeverity(s: LayerSeverity): AnalystSeverity {\n switch (s) {\n case 'critical':\n return 'critical'\n case 'major':\n return 'high'\n case 'minor':\n return 'medium'\n case 'info':\n return 'info'\n }\n}\n\n// ── 1. MultiLayerVerifier → Analyst ─────────────────────────────────\n\nexport interface VerifierAdapterOpts<Env> {\n id?: string\n area?: string\n verifier: MultiLayerVerifier<Env>\n /**\n * The verifier expects an `env` per run. Adapters take it from\n * `AnalystRunInputs.custom[<id>]` via the registry's 'custom' routing.\n */\n options?: Omit<VerifyOptions<Env>, 'env'>\n}\n\nexport function createVerifierAdapter<Env>(opts: VerifierAdapterOpts<Env>): Analyst<Env> {\n const id = opts.id ?? 'multi-layer-verifier'\n const area = opts.area ?? 'verification'\n return {\n id,\n description:\n \"Runs a MultiLayerVerifier and lifts each layer's findings into the analyst envelope.\",\n inputKind: 'custom',\n cost: { kind: 'deterministic' },\n version: `verifier-${ADAPTER_REV}`,\n async analyze(env, ctx) {\n const report = await opts.verifier.run({ env, ...opts.options })\n const out: AnalystFinding[] = []\n for (const layer of report.layers) {\n for (const finding of layer.findings) {\n out.push(liftLayerFinding(id, area, layer.layer, finding))\n }\n // Layer-level signal: a failed/error layer is itself a finding\n // even if it didn't emit per-finding rows.\n if (layer.status === 'fail' || layer.status === 'error' || layer.status === 'timeout') {\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: layer.layer,\n claim: `layer \"${layer.layer}\" ${layer.status}: ${layer.reason ?? 'no reason given'}`,\n severity:\n layer.status === 'error' ? 'high' : layer.status === 'timeout' ? 'medium' : 'high',\n confidence: 1,\n evidence_refs: [],\n metadata: {\n layer_status: layer.status,\n duration_ms: layer.durationMs,\n score: layer.score,\n diagnostics: layer.diagnostics,\n },\n }),\n )\n }\n }\n ctx.log?.('verifier complete', {\n layers: report.layers.length,\n blended: report.blendedScore,\n all_pass: report.allPass,\n })\n return out\n },\n }\n}\n\nfunction liftLayerFinding(\n analyst_id: string,\n area: string,\n layer: string,\n f: LayerFinding,\n): AnalystFinding {\n return makeFinding({\n analyst_id,\n area,\n subject: f.layer ?? layer,\n claim: f.message,\n severity: liftSeverity(f.severity),\n confidence: 0.85,\n evidence_refs: f.evidence\n ? [{ kind: 'artifact', uri: 'inline:evidence', excerpt: f.evidence }]\n : [],\n metadata: f.detail,\n })\n}\n\n// ── 2. RunCritic → Analyst ──────────────────────────────────────────\n\nexport interface RunCriticAdapterOpts {\n id?: string\n area?: string\n critic?: RunCritic\n /** Optional threshold below which a dimension is reported as a finding. Default 0.5. */\n threshold?: number\n}\n\nexport function createRunCriticAdapter(opts: RunCriticAdapterOpts = {}): Analyst<RunTrace> {\n const id = opts.id ?? 'run-critic'\n const area = opts.area ?? 'run-quality'\n const critic = opts.critic ?? new RunCritic()\n const threshold = opts.threshold ?? 0.5\n return {\n id,\n description:\n 'Scores a single run across success / grounding / drift / tool-quality and surfaces below-threshold dimensions.',\n inputKind: 'custom',\n cost: { kind: 'deterministic' },\n version: `run-critic-${ADAPTER_REV}`,\n async analyze(trace) {\n const score = critic.scoreTrace(trace)\n const out: AnalystFinding[] = []\n const dims: Array<[keyof typeof score, AnalystSeverity, string]> = [\n ['success', 'critical', 'run did not complete successfully'],\n ['goalProgress', 'high', 'goal progress is low'],\n ['repoGroundedness', 'high', 'output is poorly grounded in the repository'],\n ['toolUseQuality', 'medium', 'tool use quality is low'],\n ['patchQuality', 'medium', 'no real patch/edit evidence'],\n ['testReality', 'high', 'no real test/build evidence'],\n ['finalGate', 'critical', 'final gate is blocking'],\n ]\n for (const [dim, sev, msg] of dims) {\n const value = score[dim] as number\n if (typeof value === 'number' && value < threshold) {\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: dim,\n claim: msg,\n rationale: `${dim}=${value.toFixed(2)} below threshold ${threshold}`,\n severity: sev,\n confidence: 1,\n evidence_refs: [],\n metadata: { dimension: dim, value, threshold, run_id: trace.run.runId },\n }),\n )\n }\n }\n // Drift penalty is high → surface as a finding (inverse threshold).\n if (score.driftPenalty > 1 - threshold) {\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: 'drift',\n claim: 'agent output drifted from repository signal',\n rationale: `driftPenalty=${score.driftPenalty.toFixed(2)}`,\n severity: 'medium',\n confidence: 0.9,\n evidence_refs: [],\n metadata: { drift_penalty: score.driftPenalty, notes: score.notes },\n }),\n )\n }\n return out\n },\n }\n}\n\n// ── 3. JudgeFn → Analyst ────────────────────────────────────────────\n\nexport interface JudgeAdapterOpts {\n id?: string\n area?: string\n judge: JudgeFn\n /** TCloud handle the JudgeFn calls. */\n tcloud: TCloud\n /** Optional cost classification — most judges call an LLM. */\n cost?: Analyst['cost']\n /** Optional threshold below which a JudgeScore becomes a finding. Default 6 (on 0-10 scale). */\n threshold?: number\n}\n\nexport function createJudgeAdapter(opts: JudgeAdapterOpts): Analyst<JudgeInput> {\n const id = opts.id ?? 'judge'\n const area = opts.area ?? 'judge'\n const threshold = opts.threshold ?? 6\n return {\n id,\n description:\n 'Wraps an agent-eval JudgeFn into an analyst; below-threshold dimensions surface as findings.',\n inputKind: 'judge-input',\n cost: opts.cost ?? { kind: 'llm' },\n version: `judge-${ADAPTER_REV}`,\n async analyze(input) {\n const scores = await opts.judge(opts.tcloud, input)\n return scores\n .filter((s) => normalize10(s.score) < threshold)\n .map((s) => liftJudgeScore(id, area, s))\n },\n }\n}\n\nfunction normalize10(s: number): number {\n // JudgeScore convention is 0-10 but some judges emit 0-1. Coerce to 0-10.\n return s <= 1 ? s * 10 : s\n}\n\nfunction liftJudgeScore(analyst_id: string, area: string, s: JudgeScore): AnalystFinding {\n const score10 = normalize10(s.score)\n const severity: AnalystSeverity =\n score10 < 3 ? 'critical' : score10 < 5 ? 'high' : score10 < 7 ? 'medium' : 'low'\n return makeFinding({\n analyst_id,\n area,\n subject: s.dimension,\n claim: `${s.judgeName}/${s.dimension} scored ${score10.toFixed(1)}/10`,\n rationale: s.reasoning,\n severity,\n confidence: 0.8,\n evidence_refs: s.evidence\n ? [{ kind: 'artifact', uri: 'inline:evidence', excerpt: s.evidence }]\n : [],\n // Provenance: this finding IS a judge verdict (an acceptance score), not an\n // observation of behavior. The steer firewall (assertNoJudgeVerdict) rejects\n // it from steering — even when it cites an artifact above — because letting a\n // verdict steer the next attempt is the held-out judge leaking into the loop.\n derived_from_judge: true,\n metadata: { judge_name: s.judgeName, dimension: s.dimension, score_10: score10 },\n })\n}\n\n// ── 4. SemanticConceptJudge → Analyst ──────────────────────────────\n\nexport interface SemanticConceptJudgeAdapterOpts {\n id?: string\n area?: string\n options?: SemanticConceptJudgeOptions\n}\n\nexport function createSemanticConceptJudgeAdapter(\n opts: SemanticConceptJudgeAdapterOpts = {},\n): Analyst<SemanticConceptJudgeInput> {\n const id = opts.id ?? 'semantic-concept-judge'\n const area = opts.area ?? 'concept-coverage'\n return {\n id,\n description:\n 'Runs the semantic-concept judge and surfaces missing / weak concepts as findings.',\n inputKind: 'custom',\n cost: { kind: 'llm', models: opts.options?.model ? [opts.options.model] : undefined },\n version: `${SEMANTIC_CONCEPT_JUDGE_VERSION}-adapter-${ADAPTER_REV}`,\n async analyze(input) {\n const result = await runSemanticConceptJudge(input, opts.options)\n if (!result.available) {\n return [\n makeFinding({\n analyst_id: id,\n area,\n claim: 'semantic-concept judge unavailable',\n rationale: result.error,\n severity: 'info',\n confidence: 1,\n evidence_refs: [],\n metadata: { reason: result.error },\n }),\n ]\n }\n const out: AnalystFinding[] = []\n for (const f of result.findings) {\n // Only surface gaps: missing concepts or low scores. Concepts at\n // 7+/10 with present=true are not findings — they're successes.\n if (f.present && f.score >= 7) continue\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: f.concept,\n claim: f.present\n ? `concept \"${f.concept}\" is weak (${f.score}/10)`\n : `concept \"${f.concept}\" is missing`,\n rationale: f.evidence,\n severity: liftSeverity(f.severity),\n confidence: 0.85,\n evidence_refs: [{ kind: 'artifact', uri: 'inline:evidence', excerpt: f.evidence }],\n metadata: {\n concept: f.concept,\n present: f.present,\n score_10: f.score,\n cost_usd: result.costUsd ?? undefined,\n },\n }),\n )\n }\n return out\n },\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAoCA,IAAM,cAAc;AAIb,SAAS,aAAa,GAAmC;AAC9D,UAAQ,GAAG;AAAA,IACT,KAAK;AACH,aAAO;AAAA,IACT,KAAK;AACH,aAAO;AAAA,IACT,KAAK;AACH,aAAO;AAAA,IACT,KAAK;AACH,aAAO;AAAA,EACX;AACF;AAeO,SAAS,sBAA2B,MAA8C;AACvF,QAAM,KAAK,KAAK,MAAM;AACtB,QAAM,OAAO,KAAK,QAAQ;AAC1B,SAAO;AAAA,IACL;AAAA,IACA,aACE;AAAA,IACF,WAAW;AAAA,IACX,MAAM,EAAE,MAAM,gBAAgB;AAAA,IAC9B,SAAS,YAAY,WAAW;AAAA,IAChC,MAAM,QAAQ,KAAK,KAAK;AACtB,YAAM,SAAS,MAAM,KAAK,SAAS,IAAI,EAAE,KAAK,GAAG,KAAK,QAAQ,CAAC;AAC/D,YAAM,MAAwB,CAAC;AAC/B,iBAAW,SAAS,OAAO,QAAQ;AACjC,mBAAW,WAAW,MAAM,UAAU;AACpC,cAAI,KAAK,iBAAiB,IAAI,MAAM,MAAM,OAAO,OAAO,CAAC;AAAA,QAC3D;AAGA,YAAI,MAAM,WAAW,UAAU,MAAM,WAAW,WAAW,MAAM,WAAW,WAAW;AACrF,cAAI;AAAA,YACF,YAAY;AAAA,cACV,YAAY;AAAA,cACZ;AAAA,cACA,SAAS,MAAM;AAAA,cACf,OAAO,UAAU,MAAM,KAAK,KAAK,MAAM,MAAM,KAAK,MAAM,UAAU,iBAAiB;AAAA,cACnF,UACE,MAAM,WAAW,UAAU,SAAS,MAAM,WAAW,YAAY,WAAW;AAAA,cAC9E,YAAY;AAAA,cACZ,eAAe,CAAC;AAAA,cAChB,UAAU;AAAA,gBACR,cAAc,MAAM;AAAA,gBACpB,aAAa,MAAM;AAAA,gBACnB,OAAO,MAAM;AAAA,gBACb,aAAa,MAAM;AAAA,cACrB;AAAA,YACF,CAAC;AAAA,UACH;AAAA,QACF;AAAA,MACF;AACA,UAAI,MAAM,qBAAqB;AAAA,QAC7B,QAAQ,OAAO,OAAO;AAAA,QACtB,SAAS,OAAO;AAAA,QAChB,UAAU,OAAO;AAAA,MACnB,CAAC;AACD,aAAO;AAAA,IACT;AAAA,EACF;AACF;AAEA,SAAS,iBACP,YACA,MACA,OACA,GACgB;AAChB,SAAO,YAAY;AAAA,IACjB;AAAA,IACA;AAAA,IACA,SAAS,EAAE,SAAS;AAAA,IACpB,OAAO,EAAE;AAAA,IACT,UAAU,aAAa,EAAE,QAAQ;AAAA,IACjC,YAAY;AAAA,IACZ,eAAe,EAAE,WACb,CAAC,EAAE,MAAM,YAAY,KAAK,mBAAmB,SAAS,EAAE,SAAS,CAAC,IAClE,CAAC;AAAA,IACL,UAAU,EAAE;AAAA,EACd,CAAC;AACH;AAYO,SAAS,uBAAuB,OAA6B,CAAC,GAAsB;AACzF,QAAM,KAAK,KAAK,MAAM;AACtB,QAAM,OAAO,KAAK,QAAQ;AAC1B,QAAM,SAAS,KAAK,UAAU,IAAI,UAAU;AAC5C,QAAM,YAAY,KAAK,aAAa;AACpC,SAAO;AAAA,IACL;AAAA,IACA,aACE;AAAA,IACF,WAAW;AAAA,IACX,MAAM,EAAE,MAAM,gBAAgB;AAAA,IAC9B,SAAS,cAAc,WAAW;AAAA,IAClC,MAAM,QAAQ,OAAO;AACnB,YAAM,QAAQ,OAAO,WAAW,KAAK;AACrC,YAAM,MAAwB,CAAC;AAC/B,YAAM,OAA6D;AAAA,QACjE,CAAC,WAAW,YAAY,mCAAmC;AAAA,QAC3D,CAAC,gBAAgB,QAAQ,sBAAsB;AAAA,QAC/C,CAAC,oBAAoB,QAAQ,6CAA6C;AAAA,QAC1E,CAAC,kBAAkB,UAAU,yBAAyB;AAAA,QACtD,CAAC,gBAAgB,UAAU,6BAA6B;AAAA,QACxD,CAAC,eAAe,QAAQ,6BAA6B;AAAA,QACrD,CAAC,aAAa,YAAY,wBAAwB;AAAA,MACpD;AACA,iBAAW,CAAC,KAAK,KAAK,GAAG,KAAK,MAAM;AAClC,cAAM,QAAQ,MAAM,GAAG;AACvB,YAAI,OAAO,UAAU,YAAY,QAAQ,WAAW;AAClD,cAAI;AAAA,YACF,YAAY;AAAA,cACV,YAAY;AAAA,cACZ;AAAA,cACA,SAAS;AAAA,cACT,OAAO;AAAA,cACP,WAAW,GAAG,GAAG,IAAI,MAAM,QAAQ,CAAC,CAAC,oBAAoB,SAAS;AAAA,cAClE,UAAU;AAAA,cACV,YAAY;AAAA,cACZ,eAAe,CAAC;AAAA,cAChB,UAAU,EAAE,WAAW,KAAK,OAAO,WAAW,QAAQ,MAAM,IAAI,MAAM;AAAA,YACxE,CAAC;AAAA,UACH;AAAA,QACF;AAAA,MACF;AAEA,UAAI,MAAM,eAAe,IAAI,WAAW;AACtC,YAAI;AAAA,UACF,YAAY;AAAA,YACV,YAAY;AAAA,YACZ;AAAA,YACA,SAAS;AAAA,YACT,OAAO;AAAA,YACP,WAAW,gBAAgB,MAAM,aAAa,QAAQ,CAAC,CAAC;AAAA,YACxD,UAAU;AAAA,YACV,YAAY;AAAA,YACZ,eAAe,CAAC;AAAA,YAChB,UAAU,EAAE,eAAe,MAAM,cAAc,OAAO,MAAM,MAAM;AAAA,UACpE,CAAC;AAAA,QACH;AAAA,MACF;AACA,aAAO;AAAA,IACT;AAAA,EACF;AACF;AAgBO,SAAS,mBAAmB,MAA6C;AAC9E,QAAM,KAAK,KAAK,MAAM;AACtB,QAAM,OAAO,KAAK,QAAQ;AAC1B,QAAM,YAAY,KAAK,aAAa;AACpC,SAAO;AAAA,IACL;AAAA,IACA,aACE;AAAA,IACF,WAAW;AAAA,IACX,MAAM,KAAK,QAAQ,EAAE,MAAM,MAAM;AAAA,IACjC,SAAS,SAAS,WAAW;AAAA,IAC7B,MAAM,QAAQ,OAAO;AACnB,YAAM,SAAS,MAAM,KAAK,MAAM,KAAK,QAAQ,KAAK;AAClD,aAAO,OACJ,OAAO,CAAC,MAAM,YAAY,EAAE,KAAK,IAAI,SAAS,EAC9C,IAAI,CAAC,MAAM,eAAe,IAAI,MAAM,CAAC,CAAC;AAAA,IAC3C;AAAA,EACF;AACF;AAEA,SAAS,YAAY,GAAmB;AAEtC,SAAO,KAAK,IAAI,IAAI,KAAK;AAC3B;AAEA,SAAS,eAAe,YAAoB,MAAc,GAA+B;AACvF,QAAM,UAAU,YAAY,EAAE,KAAK;AACnC,QAAM,WACJ,UAAU,IAAI,aAAa,UAAU,IAAI,SAAS,UAAU,IAAI,WAAW;AAC7E,SAAO,YAAY;AAAA,IACjB;AAAA,IACA;AAAA,IACA,SAAS,EAAE;AAAA,IACX,OAAO,GAAG,EAAE,SAAS,IAAI,EAAE,SAAS,WAAW,QAAQ,QAAQ,CAAC,CAAC;AAAA,IACjE,WAAW,EAAE;AAAA,IACb;AAAA,IACA,YAAY;AAAA,IACZ,eAAe,EAAE,WACb,CAAC,EAAE,MAAM,YAAY,KAAK,mBAAmB,SAAS,EAAE,SAAS,CAAC,IAClE,CAAC;AAAA;AAAA;AAAA;AAAA;AAAA,IAKL,oBAAoB;AAAA,IACpB,UAAU,EAAE,YAAY,EAAE,WAAW,WAAW,EAAE,WAAW,UAAU,QAAQ;AAAA,EACjF,CAAC;AACH;AAUO,SAAS,kCACd,OAAwC,CAAC,GACL;AACpC,QAAM,KAAK,KAAK,MAAM;AACtB,QAAM,OAAO,KAAK,QAAQ;AAC1B,SAAO;AAAA,IACL;AAAA,IACA,aACE;AAAA,IACF,WAAW;AAAA,IACX,MAAM,EAAE,MAAM,OAAO,QAAQ,KAAK,SAAS,QAAQ,CAAC,KAAK,QAAQ,KAAK,IAAI,OAAU;AAAA,IACpF,SAAS,GAAG,8BAA8B,YAAY,WAAW;AAAA,IACjE,MAAM,QAAQ,OAAO;AACnB,YAAM,SAAS,MAAM,wBAAwB,OAAO,KAAK,OAAO;AAChE,UAAI,CAAC,OAAO,WAAW;AACrB,eAAO;AAAA,UACL,YAAY;AAAA,YACV,YAAY;AAAA,YACZ;AAAA,YACA,OAAO;AAAA,YACP,WAAW,OAAO;AAAA,YAClB,UAAU;AAAA,YACV,YAAY;AAAA,YACZ,eAAe,CAAC;AAAA,YAChB,UAAU,EAAE,QAAQ,OAAO,MAAM;AAAA,UACnC,CAAC;AAAA,QACH;AAAA,MACF;AACA,YAAM,MAAwB,CAAC;AAC/B,iBAAW,KAAK,OAAO,UAAU;AAG/B,YAAI,EAAE,WAAW,EAAE,SAAS,EAAG;AAC/B,YAAI;AAAA,UACF,YAAY;AAAA,YACV,YAAY;AAAA,YACZ;AAAA,YACA,SAAS,EAAE;AAAA,YACX,OAAO,EAAE,UACL,YAAY,EAAE,OAAO,cAAc,EAAE,KAAK,SAC1C,YAAY,EAAE,OAAO;AAAA,YACzB,WAAW,EAAE;AAAA,YACb,UAAU,aAAa,EAAE,QAAQ;AAAA,YACjC,YAAY;AAAA,YACZ,eAAe,CAAC,EAAE,MAAM,YAAY,KAAK,mBAAmB,SAAS,EAAE,SAAS,CAAC;AAAA,YACjF,UAAU;AAAA,cACR,SAAS,EAAE;AAAA,cACX,SAAS,EAAE;AAAA,cACX,UAAU,EAAE;AAAA,cACZ,UAAU,OAAO,WAAW;AAAA,YAC9B;AAAA,UACF,CAAC;AAAA,QACH;AAAA,MACF;AACA,aAAO;AAAA,IACT;AAAA,EACF;AACF;","names":[]}
1
+ {"version":3,"sources":["../../src/analyst/adapters.ts"],"sourcesContent":["/**\n * Adapter factories — lift each existing agent-eval primitive into the\n * Analyst contract without re-implementing it.\n *\n * Five primitives, five factories. Each one:\n * - Builds an Analyst with a stable id (caller chooses; defaults\n * given), a sensible default `inputKind`, a version derived from\n * the wrapped primitive's version + an adapter revision, and an\n * `analyze()` that calls the primitive and lifts its output to\n * AnalystFinding[] using `makeFinding()`.\n * - Maps severities: the existing `Severity` ('critical' | 'major' |\n * 'minor' | 'info') projects onto AnalystSeverity ('critical' |\n * 'high' | 'medium' | 'low' | 'info'); 'major' → 'high', 'minor' →\n * 'medium'. Domain analysts that want finer-grained mapping override.\n *\n * Adapters never own state. Calling the same factory twice with the\n * same primitive instance is safe.\n */\n\nimport type {\n Finding as LayerFinding,\n Severity as LayerSeverity,\n MultiLayerVerifier,\n VerifyOptions,\n} from '../multi-layer-verifier'\nimport { RunCritic, type RunTrace } from '../run-critic'\nimport {\n runSemanticConceptJudge,\n SEMANTIC_CONCEPT_JUDGE_VERSION,\n type SemanticConceptJudgeInput,\n type SemanticConceptJudgeOptions,\n} from '../semantic-concept-judge'\nimport type { JudgeFn, JudgeInput, JudgeScore, TCloud } from '../types'\nimport type { Analyst, AnalystFinding, AnalystSeverity } from './types'\nimport { makeFinding } from './types'\n\nconst ADAPTER_REV = '1'\n\n// ── Severity bridges ───────────────────────────────────────────────\n\nexport function liftSeverity(s: LayerSeverity): AnalystSeverity {\n switch (s) {\n case 'critical':\n return 'critical'\n case 'major':\n return 'high'\n case 'minor':\n return 'medium'\n case 'info':\n return 'info'\n }\n}\n\n// ── 1. MultiLayerVerifier → Analyst ─────────────────────────────────\n\nexport interface VerifierAdapterOpts<Env> {\n id?: string\n area?: string\n verifier: MultiLayerVerifier<Env>\n /**\n * The verifier expects an `env` per run. Adapters take it from\n * `AnalystRunInputs.custom[<id>]` via the registry's 'custom' routing.\n */\n options?: Omit<VerifyOptions<Env>, 'env'>\n}\n\nexport function createVerifierAdapter<Env>(opts: VerifierAdapterOpts<Env>): Analyst<Env> {\n const id = opts.id ?? 'multi-layer-verifier'\n const area = opts.area ?? 'verification'\n return {\n id,\n description:\n \"Runs a MultiLayerVerifier and lifts each layer's findings into the analyst envelope.\",\n inputKind: 'custom',\n cost: { kind: 'deterministic' },\n version: `verifier-${ADAPTER_REV}`,\n async analyze(env, ctx) {\n const report = await opts.verifier.run({ env, ...opts.options })\n const out: AnalystFinding[] = []\n for (const layer of report.layers) {\n for (const finding of layer.findings) {\n out.push(liftLayerFinding(id, area, layer.layer, finding))\n }\n // Layer-level signal: a failed/error layer is itself a finding\n // even if it didn't emit per-finding rows.\n if (layer.status === 'fail' || layer.status === 'error' || layer.status === 'timeout') {\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: layer.layer,\n claim: `layer \"${layer.layer}\" ${layer.status}: ${layer.reason ?? 'no reason given'}`,\n severity:\n layer.status === 'error' ? 'high' : layer.status === 'timeout' ? 'medium' : 'high',\n confidence: 1,\n evidence_refs: [],\n metadata: {\n layer_status: layer.status,\n duration_ms: layer.durationMs,\n score: layer.score,\n diagnostics: layer.diagnostics,\n },\n }),\n )\n }\n }\n ctx.log?.('verifier complete', {\n layers: report.layers.length,\n blended: report.blendedScore,\n all_pass: report.allPass,\n })\n return out\n },\n }\n}\n\nfunction liftLayerFinding(\n analyst_id: string,\n area: string,\n layer: string,\n f: LayerFinding,\n): AnalystFinding {\n return makeFinding({\n analyst_id,\n area,\n subject: f.layer ?? layer,\n claim: f.message,\n severity: liftSeverity(f.severity),\n confidence: 0.85,\n evidence_refs: f.evidence\n ? [{ kind: 'artifact', uri: 'inline:evidence', excerpt: f.evidence }]\n : [],\n metadata: f.detail,\n })\n}\n\n// ── 2. RunCritic → Analyst ──────────────────────────────────────────\n\nexport interface RunCriticAdapterOpts {\n id?: string\n area?: string\n critic?: RunCritic\n /** Optional threshold below which a dimension is reported as a finding. Default 0.5. */\n threshold?: number\n}\n\nexport function createRunCriticAdapter(opts: RunCriticAdapterOpts = {}): Analyst<RunTrace> {\n const id = opts.id ?? 'run-critic'\n const area = opts.area ?? 'run-quality'\n const critic = opts.critic ?? new RunCritic()\n const threshold = opts.threshold ?? 0.5\n return {\n id,\n description:\n 'Scores a single run across success / grounding / drift / tool-quality and surfaces below-threshold dimensions.',\n inputKind: 'custom',\n cost: { kind: 'deterministic' },\n version: `run-critic-${ADAPTER_REV}`,\n async analyze(trace) {\n const score = critic.scoreTrace(trace)\n const out: AnalystFinding[] = []\n const dims: Array<[keyof typeof score, AnalystSeverity, string]> = [\n ['success', 'critical', 'run did not complete successfully'],\n ['goalProgress', 'high', 'goal progress is low'],\n ['repoGroundedness', 'high', 'output is poorly grounded in the repository'],\n ['toolUseQuality', 'medium', 'tool use quality is low'],\n ['patchQuality', 'medium', 'no real patch/edit evidence'],\n ['testReality', 'high', 'no real test/build evidence'],\n ['finalGate', 'critical', 'final gate is blocking'],\n ]\n for (const [dim, sev, msg] of dims) {\n const value = score[dim] as number\n if (typeof value === 'number' && value < threshold) {\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: dim,\n claim: msg,\n rationale: `${dim}=${value.toFixed(2)} below threshold ${threshold}`,\n severity: sev,\n confidence: 1,\n evidence_refs: [],\n metadata: { dimension: dim, value, threshold, run_id: trace.run.runId },\n }),\n )\n }\n }\n // Drift penalty is high → surface as a finding (inverse threshold).\n if (score.driftPenalty > 1 - threshold) {\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: 'drift',\n claim: 'agent output drifted from repository signal',\n rationale: `driftPenalty=${score.driftPenalty.toFixed(2)}`,\n severity: 'medium',\n confidence: 0.9,\n evidence_refs: [],\n metadata: { drift_penalty: score.driftPenalty, notes: score.notes },\n }),\n )\n }\n return out\n },\n }\n}\n\n// ── 3. JudgeFn → Analyst ────────────────────────────────────────────\n\nexport interface JudgeAdapterOpts {\n id?: string\n area?: string\n judge: JudgeFn\n /** TCloud handle the JudgeFn calls. */\n tcloud: TCloud\n /** Optional cost classification — most judges call an LLM. */\n cost?: Analyst['cost']\n /** Optional threshold below which a JudgeScore becomes a finding. Default 6 (on 0-10 scale). */\n threshold?: number\n}\n\nexport function createJudgeAdapter(opts: JudgeAdapterOpts): Analyst<JudgeInput> {\n const id = opts.id ?? 'judge'\n const area = opts.area ?? 'judge'\n const threshold = opts.threshold ?? 6\n return {\n id,\n description:\n 'Wraps an agent-eval JudgeFn into an analyst; below-threshold dimensions surface as findings.',\n inputKind: 'judge-input',\n cost: opts.cost ?? { kind: 'llm' },\n version: `judge-${ADAPTER_REV}`,\n async analyze(input) {\n const scores = await opts.judge(opts.tcloud, input)\n return scores\n .filter((s) => normalize10(s.score) < threshold)\n .map((s) => liftJudgeScore(id, area, s))\n },\n }\n}\n\nfunction normalize10(s: number): number {\n // JudgeScore convention is 0-10 but some judges emit 0-1. Coerce to 0-10.\n return s <= 1 ? s * 10 : s\n}\n\nfunction liftJudgeScore(analyst_id: string, area: string, s: JudgeScore): AnalystFinding {\n const score10 = normalize10(s.score)\n const severity: AnalystSeverity =\n score10 < 3 ? 'critical' : score10 < 5 ? 'high' : score10 < 7 ? 'medium' : 'low'\n return makeFinding({\n analyst_id,\n area,\n subject: s.dimension,\n claim: `${s.judgeName}/${s.dimension} scored ${score10.toFixed(1)}/10`,\n rationale: s.reasoning,\n severity,\n confidence: 0.8,\n evidence_refs: s.evidence\n ? [{ kind: 'artifact', uri: 'inline:evidence', excerpt: s.evidence }]\n : [],\n // Provenance: this finding IS a judge verdict (an acceptance score), not an\n // observation of behavior. The steer firewall (assertNoJudgeVerdict) rejects\n // it from steering — even when it cites an artifact above — because letting a\n // verdict steer the next attempt is the held-out judge leaking into the loop.\n derived_from_judge: true,\n metadata: { judge_name: s.judgeName, dimension: s.dimension, score_10: score10 },\n })\n}\n\n// ── 4. SemanticConceptJudge → Analyst ──────────────────────────────\n\nexport interface SemanticConceptJudgeAdapterOpts {\n id?: string\n area?: string\n options?: SemanticConceptJudgeOptions\n}\n\nexport function createSemanticConceptJudgeAdapter(\n opts: SemanticConceptJudgeAdapterOpts = {},\n): Analyst<SemanticConceptJudgeInput> {\n const id = opts.id ?? 'semantic-concept-judge'\n const area = opts.area ?? 'concept-coverage'\n return {\n id,\n description:\n 'Runs the semantic-concept judge and surfaces missing / weak concepts as findings.',\n inputKind: 'custom',\n cost: { kind: 'llm', models: opts.options?.model ? [opts.options.model] : undefined },\n version: `${SEMANTIC_CONCEPT_JUDGE_VERSION}-adapter-${ADAPTER_REV}`,\n async analyze(input) {\n const result = await runSemanticConceptJudge(input, opts.options)\n if (!result.available) {\n return [\n makeFinding({\n analyst_id: id,\n area,\n claim: 'semantic-concept judge unavailable',\n rationale: result.error,\n severity: 'info',\n confidence: 1,\n evidence_refs: [],\n metadata: { reason: result.error },\n }),\n ]\n }\n const out: AnalystFinding[] = []\n for (const f of result.findings) {\n // Only surface gaps: missing concepts or low scores. Concepts at\n // 7+/10 with present=true are not findings — they're successes.\n if (f.present && f.score >= 7) continue\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: f.concept,\n claim: f.present\n ? `concept \"${f.concept}\" is weak (${f.score}/10)`\n : `concept \"${f.concept}\" is missing`,\n rationale: f.evidence,\n severity: liftSeverity(f.severity),\n confidence: 0.85,\n evidence_refs: [{ kind: 'artifact', uri: 'inline:evidence', excerpt: f.evidence }],\n metadata: {\n concept: f.concept,\n present: f.present,\n score_10: f.score,\n cost_usd: result.costUsd ?? undefined,\n },\n }),\n )\n }\n return out\n },\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAoCA,IAAM,cAAc;AAIb,SAAS,aAAa,GAAmC;AAC9D,UAAQ,GAAG;AAAA,IACT,KAAK;AACH,aAAO;AAAA,IACT,KAAK;AACH,aAAO;AAAA,IACT,KAAK;AACH,aAAO;AAAA,IACT,KAAK;AACH,aAAO;AAAA,EACX;AACF;AAeO,SAAS,sBAA2B,MAA8C;AACvF,QAAM,KAAK,KAAK,MAAM;AACtB,QAAM,OAAO,KAAK,QAAQ;AAC1B,SAAO;AAAA,IACL;AAAA,IACA,aACE;AAAA,IACF,WAAW;AAAA,IACX,MAAM,EAAE,MAAM,gBAAgB;AAAA,IAC9B,SAAS,YAAY,WAAW;AAAA,IAChC,MAAM,QAAQ,KAAK,KAAK;AACtB,YAAM,SAAS,MAAM,KAAK,SAAS,IAAI,EAAE,KAAK,GAAG,KAAK,QAAQ,CAAC;AAC/D,YAAM,MAAwB,CAAC;AAC/B,iBAAW,SAAS,OAAO,QAAQ;AACjC,mBAAW,WAAW,MAAM,UAAU;AACpC,cAAI,KAAK,iBAAiB,IAAI,MAAM,MAAM,OAAO,OAAO,CAAC;AAAA,QAC3D;AAGA,YAAI,MAAM,WAAW,UAAU,MAAM,WAAW,WAAW,MAAM,WAAW,WAAW;AACrF,cAAI;AAAA,YACF,YAAY;AAAA,cACV,YAAY;AAAA,cACZ;AAAA,cACA,SAAS,MAAM;AAAA,cACf,OAAO,UAAU,MAAM,KAAK,KAAK,MAAM,MAAM,KAAK,MAAM,UAAU,iBAAiB;AAAA,cACnF,UACE,MAAM,WAAW,UAAU,SAAS,MAAM,WAAW,YAAY,WAAW;AAAA,cAC9E,YAAY;AAAA,cACZ,eAAe,CAAC;AAAA,cAChB,UAAU;AAAA,gBACR,cAAc,MAAM;AAAA,gBACpB,aAAa,MAAM;AAAA,gBACnB,OAAO,MAAM;AAAA,gBACb,aAAa,MAAM;AAAA,cACrB;AAAA,YACF,CAAC;AAAA,UACH;AAAA,QACF;AAAA,MACF;AACA,UAAI,MAAM,qBAAqB;AAAA,QAC7B,QAAQ,OAAO,OAAO;AAAA,QACtB,SAAS,OAAO;AAAA,QAChB,UAAU,OAAO;AAAA,MACnB,CAAC;AACD,aAAO;AAAA,IACT;AAAA,EACF;AACF;AAEA,SAAS,iBACP,YACA,MACA,OACA,GACgB;AAChB,SAAO,YAAY;AAAA,IACjB;AAAA,IACA;AAAA,IACA,SAAS,EAAE,SAAS;AAAA,IACpB,OAAO,EAAE;AAAA,IACT,UAAU,aAAa,EAAE,QAAQ;AAAA,IACjC,YAAY;AAAA,IACZ,eAAe,EAAE,WACb,CAAC,EAAE,MAAM,YAAY,KAAK,mBAAmB,SAAS,EAAE,SAAS,CAAC,IAClE,CAAC;AAAA,IACL,UAAU,EAAE;AAAA,EACd,CAAC;AACH;AAYO,SAAS,uBAAuB,OAA6B,CAAC,GAAsB;AACzF,QAAM,KAAK,KAAK,MAAM;AACtB,QAAM,OAAO,KAAK,QAAQ;AAC1B,QAAM,SAAS,KAAK,UAAU,IAAI,UAAU;AAC5C,QAAM,YAAY,KAAK,aAAa;AACpC,SAAO;AAAA,IACL;AAAA,IACA,aACE;AAAA,IACF,WAAW;AAAA,IACX,MAAM,EAAE,MAAM,gBAAgB;AAAA,IAC9B,SAAS,cAAc,WAAW;AAAA,IAClC,MAAM,QAAQ,OAAO;AACnB,YAAM,QAAQ,OAAO,WAAW,KAAK;AACrC,YAAM,MAAwB,CAAC;AAC/B,YAAM,OAA6D;AAAA,QACjE,CAAC,WAAW,YAAY,mCAAmC;AAAA,QAC3D,CAAC,gBAAgB,QAAQ,sBAAsB;AAAA,QAC/C,CAAC,oBAAoB,QAAQ,6CAA6C;AAAA,QAC1E,CAAC,kBAAkB,UAAU,yBAAyB;AAAA,QACtD,CAAC,gBAAgB,UAAU,6BAA6B;AAAA,QACxD,CAAC,eAAe,QAAQ,6BAA6B;AAAA,QACrD,CAAC,aAAa,YAAY,wBAAwB;AAAA,MACpD;AACA,iBAAW,CAAC,KAAK,KAAK,GAAG,KAAK,MAAM;AAClC,cAAM,QAAQ,MAAM,GAAG;AACvB,YAAI,OAAO,UAAU,YAAY,QAAQ,WAAW;AAClD,cAAI;AAAA,YACF,YAAY;AAAA,cACV,YAAY;AAAA,cACZ;AAAA,cACA,SAAS;AAAA,cACT,OAAO;AAAA,cACP,WAAW,GAAG,GAAG,IAAI,MAAM,QAAQ,CAAC,CAAC,oBAAoB,SAAS;AAAA,cAClE,UAAU;AAAA,cACV,YAAY;AAAA,cACZ,eAAe,CAAC;AAAA,cAChB,UAAU,EAAE,WAAW,KAAK,OAAO,WAAW,QAAQ,MAAM,IAAI,MAAM;AAAA,YACxE,CAAC;AAAA,UACH;AAAA,QACF;AAAA,MACF;AAEA,UAAI,MAAM,eAAe,IAAI,WAAW;AACtC,YAAI;AAAA,UACF,YAAY;AAAA,YACV,YAAY;AAAA,YACZ;AAAA,YACA,SAAS;AAAA,YACT,OAAO;AAAA,YACP,WAAW,gBAAgB,MAAM,aAAa,QAAQ,CAAC,CAAC;AAAA,YACxD,UAAU;AAAA,YACV,YAAY;AAAA,YACZ,eAAe,CAAC;AAAA,YAChB,UAAU,EAAE,eAAe,MAAM,cAAc,OAAO,MAAM,MAAM;AAAA,UACpE,CAAC;AAAA,QACH;AAAA,MACF;AACA,aAAO;AAAA,IACT;AAAA,EACF;AACF;AAgBO,SAAS,mBAAmB,MAA6C;AAC9E,QAAM,KAAK,KAAK,MAAM;AACtB,QAAM,OAAO,KAAK,QAAQ;AAC1B,QAAM,YAAY,KAAK,aAAa;AACpC,SAAO;AAAA,IACL;AAAA,IACA,aACE;AAAA,IACF,WAAW;AAAA,IACX,MAAM,KAAK,QAAQ,EAAE,MAAM,MAAM;AAAA,IACjC,SAAS,SAAS,WAAW;AAAA,IAC7B,MAAM,QAAQ,OAAO;AACnB,YAAM,SAAS,MAAM,KAAK,MAAM,KAAK,QAAQ,KAAK;AAClD,aAAO,OACJ,OAAO,CAAC,MAAM,YAAY,EAAE,KAAK,IAAI,SAAS,EAC9C,IAAI,CAAC,MAAM,eAAe,IAAI,MAAM,CAAC,CAAC;AAAA,IAC3C;AAAA,EACF;AACF;AAEA,SAAS,YAAY,GAAmB;AAEtC,SAAO,KAAK,IAAI,IAAI,KAAK;AAC3B;AAEA,SAAS,eAAe,YAAoB,MAAc,GAA+B;AACvF,QAAM,UAAU,YAAY,EAAE,KAAK;AACnC,QAAM,WACJ,UAAU,IAAI,aAAa,UAAU,IAAI,SAAS,UAAU,IAAI,WAAW;AAC7E,SAAO,YAAY;AAAA,IACjB;AAAA,IACA;AAAA,IACA,SAAS,EAAE;AAAA,IACX,OAAO,GAAG,EAAE,SAAS,IAAI,EAAE,SAAS,WAAW,QAAQ,QAAQ,CAAC,CAAC;AAAA,IACjE,WAAW,EAAE;AAAA,IACb;AAAA,IACA,YAAY;AAAA,IACZ,eAAe,EAAE,WACb,CAAC,EAAE,MAAM,YAAY,KAAK,mBAAmB,SAAS,EAAE,SAAS,CAAC,IAClE,CAAC;AAAA;AAAA;AAAA;AAAA;AAAA,IAKL,oBAAoB;AAAA,IACpB,UAAU,EAAE,YAAY,EAAE,WAAW,WAAW,EAAE,WAAW,UAAU,QAAQ;AAAA,EACjF,CAAC;AACH;AAUO,SAAS,kCACd,OAAwC,CAAC,GACL;AACpC,QAAM,KAAK,KAAK,MAAM;AACtB,QAAM,OAAO,KAAK,QAAQ;AAC1B,SAAO;AAAA,IACL;AAAA,IACA,aACE;AAAA,IACF,WAAW;AAAA,IACX,MAAM,EAAE,MAAM,OAAO,QAAQ,KAAK,SAAS,QAAQ,CAAC,KAAK,QAAQ,KAAK,IAAI,OAAU;AAAA,IACpF,SAAS,GAAG,8BAA8B,YAAY,WAAW;AAAA,IACjE,MAAM,QAAQ,OAAO;AACnB,YAAM,SAAS,MAAM,wBAAwB,OAAO,KAAK,OAAO;AAChE,UAAI,CAAC,OAAO,WAAW;AACrB,eAAO;AAAA,UACL,YAAY;AAAA,YACV,YAAY;AAAA,YACZ;AAAA,YACA,OAAO;AAAA,YACP,WAAW,OAAO;AAAA,YAClB,UAAU;AAAA,YACV,YAAY;AAAA,YACZ,eAAe,CAAC;AAAA,YAChB,UAAU,EAAE,QAAQ,OAAO,MAAM;AAAA,UACnC,CAAC;AAAA,QACH;AAAA,MACF;AACA,YAAM,MAAwB,CAAC;AAC/B,iBAAW,KAAK,OAAO,UAAU;AAG/B,YAAI,EAAE,WAAW,EAAE,SAAS,EAAG;AAC/B,YAAI;AAAA,UACF,YAAY;AAAA,YACV,YAAY;AAAA,YACZ;AAAA,YACA,SAAS,EAAE;AAAA,YACX,OAAO,EAAE,UACL,YAAY,EAAE,OAAO,cAAc,EAAE,KAAK,SAC1C,YAAY,EAAE,OAAO;AAAA,YACzB,WAAW,EAAE;AAAA,YACb,UAAU,aAAa,EAAE,QAAQ;AAAA,YACjC,YAAY;AAAA,YACZ,eAAe,CAAC,EAAE,MAAM,YAAY,KAAK,mBAAmB,SAAS,EAAE,SAAS,CAAC;AAAA,YACjF,UAAU;AAAA,cACR,SAAS,EAAE;AAAA,cACX,SAAS,EAAE;AAAA,cACX,UAAU,EAAE;AAAA,cACZ,UAAU,OAAO,WAAW;AAAA,YAC9B;AAAA,UACF,CAAC;AAAA,QACH;AAAA,MACF;AACA,aAAO;AAAA,IACT;AAAA,EACF;AACF;","names":[]}
@@ -1,5 +1,5 @@
1
1
  import { AxAIService } from '@ax-llm/ax';
2
- import { T as TraceAnalysisStore } from './store-9cAScOcb.js';
2
+ import { T as TraceAnalysisStore } from './store-C1YxJDEK.js';
3
3
 
4
4
  interface AnalyzeTracesInput {
5
5
  /** The user-facing question. Domain framing belongs here, not in the
@@ -1,7 +1,7 @@
1
- import { A as AnalystRegistry } from './default-registry-DDfv22MQ.js';
1
+ import { A as AnalystRegistry } from './default-registry-DaK8b3fv.js';
2
2
  import { D as DatasetScenario } from './dataset-NENEzRgk.js';
3
- import { R as RunRecord } from './run-record-CZmcpWPo.js';
4
- import { I as InsightReport } from './insight-report-oMVxDTxl.js';
3
+ import { R as RunRecord } from './run-record-BDH49H2E.js';
4
+ import { I as InsightReport } from './insight-report-DY4nDW9Q.js';
5
5
 
6
6
  /**
7
7
  * # `analyzeRuns()` — turn a set of agent runs into an actionable decision packet.
@@ -1,5 +1,5 @@
1
- import { T as TraceStore } from './store-BsVi7ncX.js';
2
- import { S as Span, e as TraceEvent } from './schema-SGWcK9wa.js';
1
+ import { T as TraceStore } from './store-DGqD0Pyo.js';
2
+ import { S as Span, e as TraceEvent } from './schema-B3Q3l9Z_.js';
3
3
 
4
4
  /**
5
5
  * Tool-use metrics — derived purely from trace data.
@@ -13,9 +13,11 @@ import { S as Span, e as TraceEvent } from './schema-SGWcK9wa.js';
13
13
  interface ToolUseMetrics {
14
14
  runId: string;
15
15
  totalCalls: number;
16
+ /** Calls whose arguments were captured and can be compared for duplication. */
17
+ callsWithCapturedArgs: number;
16
18
  byTool: Record<string, ToolStats>;
17
19
  errorRate: number;
18
- /** Ratio of calls with identical (toolName, argHash) already seen earlier in the same run. */
20
+ /** Ratio of captured-argument calls already seen with the same tool name and arguments. */
19
21
  duplicateRate: number;
20
22
  /** Ratio of error calls followed by ≥1 retry on same tool. */
21
23
  retryRate: number;
@@ -24,6 +26,7 @@ interface ToolUseMetrics {
24
26
  }
25
27
  interface ToolStats {
26
28
  calls: number;
29
+ callsWithCapturedArgs: number;
27
30
  errors: number;
28
31
  avgLatencyMs: number;
29
32
  duplicates: number;
@@ -1,10 +1,10 @@
1
- import { c as CalibrationReport } from '../calibration-Dz8TQV4y.js';
1
+ import { c as CalibrationReport } from '../calibration-C8MTS7cw.js';
2
2
  import { O as OffPolicyEstimate, a as OffPolicyOptions, b as OffPolicyTrajectory } from '../off-policy-DiwuKKg7.js';
3
- import { d as CodeAgentSessionSource, a as CodeAgentSessionIntakeOptions, c as CodeAgentSessionMetrics, C as CodeAgentSessionDiagnostic } from '../code-agent-session-CdxteG0y.js';
4
- import { R as RunRecord, a as RunSplitTag } from '../run-record-CZmcpWPo.js';
5
- import { T as TraceStore } from '../store-BsVi7ncX.js';
6
- import { R as RuntimeTrajectoryRecord, P as ProjectRuntimeTrajectoryEvidenceOptions, a as RuntimeTrajectoryEvidenceProjection } from '../runtime-trajectory-CC0jx9ql.js';
7
- import '../schema-SGWcK9wa.js';
3
+ import { d as CodeAgentSessionSource, a as CodeAgentSessionIntakeOptions, c as CodeAgentSessionMetrics, C as CodeAgentSessionDiagnostic } from '../code-agent-session-CjZsVd19.js';
4
+ import { R as RunRecord, a as RunSplitTag } from '../run-record-BDH49H2E.js';
5
+ import { T as TraceStore } from '../store-DGqD0Pyo.js';
6
+ import { R as RuntimeTrajectoryRecord, P as ProjectRuntimeTrajectoryEvidenceOptions, a as RuntimeTrajectoryEvidenceProjection } from '../runtime-trajectory-DGBIUt4B.js';
7
+ import '../schema-B3Q3l9Z_.js';
8
8
  import '../outcome-store-rnXLEqSn.js';
9
9
  import '@tangle-network/agent-interface';
10
10
  import '../errors-oeQrLqXC.js';
@@ -11,13 +11,13 @@ import {
11
11
  import {
12
12
  projectRuntimeTrajectoryEvidence
13
13
  } from "../chunk-T4SQEITX.js";
14
- import "../chunk-VI2UW6B6.js";
15
14
  import {
16
15
  offPolicyEstimateAll
17
16
  } from "../chunk-DTJ6QUQB.js";
18
17
  import {
19
18
  confidenceInterval
20
19
  } from "../chunk-PJQFMIOX.js";
20
+ import "../chunk-VI2UW6B6.js";
21
21
  import {
22
22
  ValidationError
23
23
  } from "../chunk-ONWEPEDO.js";
@@ -1,11 +1,14 @@
1
- export { B as BENCHMARK_SPLIT_SEED, a as BenchmarkAdapter, b as BenchmarkDatasetItem, k as BenchmarkDistribution, c as BenchmarkEvaluation, d as BenchmarkFamily, l as BenchmarkMetricCalibrationOptions, m as BenchmarkMetricCalibrationResult, n as BenchmarkReport, e as BenchmarkResponder, o as BenchmarkRunOptions, p as BenchmarkRunResult, f as BenchmarkScenario, q as BenchmarkSliceSummary, g as BenchmarkSource, h as BenchmarkTaskKind, r as BuildStandardRetrievalItemsOptions, R as RetrievalIdAdapterOptions, S as StandardRetrievalArtifact, s as StandardRetrievalDocument, t as StandardRetrievalEvaluationOptions, u as StandardRetrievalPayload, v as StandardRetrievalQrel, w as StandardRetrievalQuery, x as StandardRetrievalResult, y as buildStandardRetrievalItems, z as calibrateBenchmarkMetric, A as createRetrievalIdBenchmarkAdapter, i as deterministicSplit, C as evaluateStandardRetrieval, D as normalizeRetrievedDocumentIds, E as parseBeirCorpusJsonl, F as parseBeirQueriesJsonl, G as parseJsonlRows, H as parseQrels, I as parseTsvRows, J as renderBenchmarkReportMarkdown, K as retrievalMetricsAtCutoff, L as routing, M as runBenchmarkAdapter, N as summarizeBenchmarkCampaign } from '../index-DbCXJfZ1.js';
2
- import '../types-Ca_63YSD.js';
3
- import '../policy-edit-Clb2v6Oa.js';
4
- import '../run-record-CZmcpWPo.js';
1
+ export { B as BENCHMARK_SPLIT_SEED, a as BenchmarkAdapter, b as BenchmarkDatasetItem, k as BenchmarkDistribution, c as BenchmarkEvaluation, d as BenchmarkFamily, l as BenchmarkMetricCalibrationOptions, m as BenchmarkMetricCalibrationResult, n as BenchmarkReport, e as BenchmarkResponder, o as BenchmarkRunOptions, p as BenchmarkRunResult, f as BenchmarkScenario, q as BenchmarkSliceSummary, g as BenchmarkSource, h as BenchmarkTaskKind, r as BuildStandardRetrievalItemsOptions, R as RetrievalIdAdapterOptions, S as StandardRetrievalArtifact, s as StandardRetrievalDocument, t as StandardRetrievalEvaluationOptions, u as StandardRetrievalPayload, v as StandardRetrievalQrel, w as StandardRetrievalQuery, x as StandardRetrievalResult, y as buildStandardRetrievalItems, z as calibrateBenchmarkMetric, A as createRetrievalIdBenchmarkAdapter, i as deterministicSplit, C as evaluateStandardRetrieval, D as normalizeRetrievedDocumentIds, E as parseBeirCorpusJsonl, F as parseBeirQueriesJsonl, G as parseJsonlRows, H as parseQrels, I as parseTsvRows, J as renderBenchmarkReportMarkdown, K as retrievalMetricsAtCutoff, L as routing, M as runBenchmarkAdapter, N as summarizeBenchmarkCampaign } from '../index-PdX4VnPA.js';
2
+ import '../types-BSw1rOUB.js';
3
+ import '../policy-edit-wG9uFEFm.js';
4
+ import '../run-record-BDH49H2E.js';
5
5
  import '@tangle-network/agent-interface';
6
6
  import '../errors-oeQrLqXC.js';
7
- import '../schema-SGWcK9wa.js';
8
- import '../store-9cAScOcb.js';
9
- import '../types-C7DGg5ex.js';
7
+ import '../schema-B3Q3l9Z_.js';
8
+ import '../store-C1YxJDEK.js';
9
+ import '../types-BkfcQnxV.js';
10
+ import '../cost-ledger-DWy3XdJc.js';
10
11
  import '@tangle-network/tcloud';
11
- import '../storage-Dw_f7WMt.js';
12
+ import '../llm-client-qoDd18Qz.js';
13
+ import '../raw-provider-sink-C46HDghv.js';
14
+ import '../storage-DrX3v_5B.js';
@@ -16,22 +16,23 @@ import {
16
16
  routing_exports,
17
17
  runBenchmarkAdapter,
18
18
  summarizeBenchmarkCampaign
19
- } from "../chunk-3LXTCTWL.js";
20
- import "../chunk-GSW3OBHK.js";
21
- import "../chunk-3274WNK7.js";
22
- import "../chunk-VI2UW6B6.js";
23
- import "../chunk-FAOEFFRT.js";
24
- import "../chunk-MPHTT5HE.js";
25
- import "../chunk-CIUOICJT.js";
19
+ } from "../chunk-JSDVRFAP.js";
20
+ import "../chunk-JSJZ4PJ6.js";
21
+ import "../chunk-HQPHZGL6.js";
22
+ import "../chunk-IDZTTFRR.js";
23
+ import "../chunk-3YYRZDON.js";
24
+ import "../chunk-MGEHEHSN.js";
26
25
  import "../chunk-ARU2PZFM.js";
27
26
  import "../chunk-PJQFMIOX.js";
28
- import "../chunk-RPDDVKI7.js";
27
+ import "../chunk-4JLWXDYA.js";
29
28
  import "../chunk-GGE4NNQT.js";
30
- import "../chunk-LNQEP766.js";
29
+ import "../chunk-S2F4J57L.js";
31
30
  import "../chunk-5UF54T55.js";
32
31
  import "../chunk-XJYR7XFV.js";
33
32
  import "../chunk-VSMTAMNK.js";
34
- import "../chunk-GY4SYVPJ.js";
33
+ import "../chunk-NJC7U437.js";
34
+ import "../chunk-VCTY3W6J.js";
35
+ import "../chunk-VI2UW6B6.js";
35
36
  import "../chunk-PC4UYEBM.js";
36
37
  import "../chunk-ONWEPEDO.js";
37
38
  import "../chunk-PZ5AY32C.js";
@@ -1,7 +1,7 @@
1
- import { S as SandboxDriver, H as HarnessConfig, a as SandboxHarnessResult, T as TestGradedScenario, b as TestGradedRunResult } from '../test-graded-scenario-mzYBKspu.js';
2
- import { T as TraceEmitter } from '../emitter-BRchAAAx.js';
3
- import { R as Run } from '../schema-SGWcK9wa.js';
4
- import { T as TraceStore } from '../store-BsVi7ncX.js';
1
+ import { S as SandboxDriver, H as HarnessConfig, a as SandboxHarnessResult, T as TestGradedScenario, b as TestGradedRunResult } from '../test-graded-scenario-B0ybnPY7.js';
2
+ import { T as TraceEmitter } from '../emitter-CjD7vUwv.js';
3
+ import { R as Run } from '../schema-B3Q3l9Z_.js';
4
+ import { T as TraceStore } from '../store-DGqD0Pyo.js';
5
5
 
6
6
  /**
7
7
  * BuilderSession — ties a builder-of-builders workflow together.
@@ -8,7 +8,7 @@ import {
8
8
  } from "../chunk-PJQFMIOX.js";
9
9
  import {
10
10
  judgeSpans
11
- } from "../chunk-MHNQWM4I.js";
11
+ } from "../chunk-LQUTGLOZ.js";
12
12
  import {
13
13
  TraceEmitter
14
14
  } from "../chunk-TVVP3ZZQ.js";
@@ -1,5 +1,5 @@
1
- import { T as TraceStore } from './store-BsVi7ncX.js';
2
- import { R as Run } from './schema-SGWcK9wa.js';
1
+ import { T as TraceStore } from './store-DGqD0Pyo.js';
2
+ import { R as Run } from './schema-B3Q3l9Z_.js';
3
3
  import { O as OutcomeFilter, b as OutcomeStore } from './outcome-store-rnXLEqSn.js';
4
4
 
5
5
  /**
@@ -1,34 +1,37 @@
1
- import { L as LlmClientOptions, w as PolicyEditAdmissionOptions, v as PolicyEditAdmission, u as PolicyEdit, d as AnalystFinding, F as FindingToPolicyEditOptions } from '../policy-edit-Clb2v6Oa.js';
2
- export { s as POLICY_EDIT_CANDIDATE_RECORD_SCHEMA, P as PolicyEditCandidateRecord, a4 as validatePolicyEditCandidateRecord } from '../policy-edit-Clb2v6Oa.js';
3
- import { P as PairedArmsComparison, S as SignedManifest, B as BackendIntegrityReport, C as CompletionRequirement, R as RuntimeEventLike, a as CompletionVerdict, b as ProducedState, c as CorrectnessChecker } from '../pre-registration--vU0mMtD.js';
4
- export { L as LlmJudgeDimension, d as LlmJudgeOptions, l as llmJudge } from '../pre-registration--vU0mMtD.js';
5
- import { A as AnalyzeTracesOptions, a as AnalyzeTracesInput, b as AnalyzeTracesResult } from '../analyst-CFBc14Wc.js';
6
- import { S as Scenario, M as MutableSurface, D as DispatchContext, b as JudgeConfig, g as Gate, e as GenerationRecord, J as JudgeScore, L as LabeledScenarioStore, s as LabeledScenarioWrite, t as LabeledScenarioSampleArgs, u as LabeledScenarioRecord, v as LabelTrust, f as SurfaceProposer, w as ProposedCandidate, x as ProposeContext, m as CodeSurface, y as LabeledScenarioSource, C as CampaignResult } from '../types-Ca_63YSD.js';
7
- export { i as CampaignAggregates, j as CampaignArtifactWriter, k as CampaignCellResult, l as CampaignCostMeter, z as CampaignTokenUsage, d as CampaignTraceWriter, c as DispatchFn, n as GateContext, h as GateDecision, G as GateResult, o as GenerationCandidate, A as JudgeAggregate, a as JudgeDimension, p as Mutator, O as OptimizationProposer, q as OptimizerConfig, P as ParetoParent, R as RedactionStatus, B as ScenarioAggregate, E as ScoredSurfaceOutcome, r as SessionScript, T as TraceSpan, F as isProposedCandidate, H as labelTrustRank } from '../types-Ca_63YSD.js';
8
- import { C as CampaignRunPlan, P as PlanCampaignRunOptions, b as RunCampaignOptions, c as RunImprovementLoopOptions } from '../gepa-CQelRtuC.js';
9
- export { f as CampaignRunPlanCell, h as GepaProposerConstraints, G as GepaProposerOptions, O as OpenAutoPrOptions, i as OpenAutoPrResult, a as RunImprovementLoopResult, R as RunOptimizationOptions, j as RunOptimizationResult, k as countSentenceEdits, l as defaultRenderDiff, m as extractH2Sections, g as gepaProposer, o as openAutoPr, p as planCampaignRun, r as runCampaign, d as runImprovementLoop, n as runOptimization } from '../gepa-CQelRtuC.js';
10
- import { a as PairedBootstrapResult, E as EProcessState } from '../statistics-oUbOJe-S.js';
11
- import { C as CampaignStorage } from '../storage-Dw_f7WMt.js';
12
- export { f as fsCampaignStorage, i as inMemoryCampaignStorage } from '../storage-Dw_f7WMt.js';
13
- export { A as AxisEvidence, a as AxisVerdict, B as BuildEvidenceVectorOptions, l as BuildLoopProvenanceArgs, D as DefaultProductionGateOptions, m as EmitLoopProvenanceArgs, n as EmitLoopProvenanceResult, E as EvidenceVector, b as EvolutionaryProposerOptions, H as HeldOutGateOptions, o as LoopProvenanceBackend, q as LoopProvenanceCandidate, L as LoopProvenanceRecord, O as ObjectiveSource, c as ParetoSignificanceGateOptions, P as PowerPreflight, s as PowerPreflightOptions, d as PromotionObjective, e as PromotionPolicy, R as RunEvalOptions, f as buildEvidenceVector, t as buildLoopProvenanceRecord, g as composeGate, h as defaultProductionGate, u as emitLoopProvenance, i as evolutionaryProposer, j as heldOutGate, v as loopProvenanceSpans, p as paretoPolicy, k as paretoSignificanceGate, w as powerPreflight, x as provenanceRecordPath, y as provenanceSpansPath, r as runEval } from '../provenance-BbVagC68.js';
1
+ import { v as PolicyEditAdmissionOptions, u as PolicyEditAdmission, t as PolicyEdit, c as AnalystFinding, F as FindingToPolicyEditOptions } from '../policy-edit-wG9uFEFm.js';
2
+ export { r as POLICY_EDIT_CANDIDATE_RECORD_SCHEMA, P as PolicyEditCandidateRecord, a2 as validatePolicyEditCandidateRecord } from '../policy-edit-wG9uFEFm.js';
3
+ import { P as PairedArmsComparison, S as SignedManifest, B as BackendIntegrityReport, C as CompletionRequirement, R as RuntimeEventLike, a as CompletionVerdict, b as ProducedState, c as CorrectnessChecker } from '../pre-registration-BWQhJ3vz.js';
4
+ export { L as LlmJudgeDimension, d as LlmJudgeOptions, l as llmJudge } from '../pre-registration-BWQhJ3vz.js';
5
+ import { C as CampaignRunPlan, P as PlanCampaignRunOptions, h as RunCampaignOptions, i as RunImprovementLoopOptions } from '../gepa-eESocoDi.js';
6
+ export { o as CampaignRunPlanCell, p as GepaProposerConstraints, G as GepaProposerOptions, O as OpenAutoPrOptions, q as OpenAutoPrResult, e as ReferenceEquivalenceJudgeOptions, g as ReferenceEquivalenceScenario, a as RunImprovementLoopResult, R as RunOptimizationOptions, s as RunOptimizationResult, t as countSentenceEdits, j as createReferenceEquivalenceJudge, u as defaultRenderDiff, v as extractH2Sections, k as gepaProposer, w as openAutoPr, x as planCampaignRun, r as runCampaign, l as runImprovementLoop, y as runOptimization } from '../gepa-eESocoDi.js';
7
+ import { A as AnalyzeTracesOptions, a as AnalyzeTracesInput, b as AnalyzeTracesResult } from '../analyst-C8HHvfJp.js';
8
+ import { S as Scenario, M as MutableSurface, D as DispatchContext, b as JudgeConfig, G as Gate, o as GenerationRecord, J as JudgeScore, L as LabeledScenarioStore, s as LabeledScenarioWrite, t as LabeledScenarioSampleArgs, u as LabeledScenarioRecord, v as LabelTrust, c as SurfaceProposer, w as ProposedCandidate, x as ProposeContext, j as CodeSurface, y as LabeledScenarioSource, C as CampaignResult } from '../types-BSw1rOUB.js';
9
+ export { e as CampaignAggregates, f as CampaignArtifactWriter, g as CampaignCellResult, h as CampaignCostMeter, z as CampaignTokenUsage, i as CampaignTraceWriter, k as DispatchFn, l as GateContext, d as GateDecision, m as GateResult, n as GenerationCandidate, A as JudgeAggregate, a as JudgeDimension, p as Mutator, O as OptimizationProposer, q as OptimizerConfig, P as ParetoParent, R as RedactionStatus, B as ScenarioAggregate, E as ScoredSurfaceOutcome, r as SessionScript, T as TraceSpan, F as isProposedCandidate, H as labelTrustRank } from '../types-BSw1rOUB.js';
10
+ import { a as PairedBootstrapResult, E as EProcessState } from '../statistics-KUnG73jH.js';
11
+ import { C as CampaignStorage } from '../storage-DrX3v_5B.js';
12
+ export { c as createRunCostLedger, f as fsCampaignStorage, i as inMemoryCampaignStorage } from '../storage-DrX3v_5B.js';
13
+ export { A as AxisEvidence, a as AxisVerdict, B as BuildEvidenceVectorOptions, l as BuildLoopProvenanceArgs, D as DefaultProductionGateOptions, m as EmitLoopProvenanceArgs, n as EmitLoopProvenanceResult, E as EvidenceVector, b as EvolutionaryProposerOptions, H as HeldOutGateOptions, o as LoopProvenanceBackend, q as LoopProvenanceCandidate, L as LoopProvenanceRecord, O as ObjectiveSource, c as ParetoSignificanceGateOptions, P as PowerPreflight, s as PowerPreflightOptions, d as PromotionObjective, e as PromotionPolicy, R as RunEvalOptions, f as buildEvidenceVector, t as buildLoopProvenanceRecord, g as composeGate, h as defaultProductionGate, u as emitLoopProvenance, i as evolutionaryProposer, j as heldOutGate, v as loopProvenanceSpans, p as paretoPolicy, k as paretoSignificanceGate, w as powerPreflight, x as provenanceRecordPath, y as provenanceSpansPath, r as runEval } from '../provenance-DpjwyseI.js';
14
+ import { a as LlmClientOptions } from '../llm-client-qoDd18Qz.js';
14
15
  import { AgentProfile } from '@tangle-network/agent-interface';
15
16
  import { A as AgentEvalError, V as ValidationError } from '../errors-oeQrLqXC.js';
16
- import { a as RunSplitTag, R as RunRecord } from '../run-record-CZmcpWPo.js';
17
- import { T as TraceAnalystKindSpec } from '../kind-factory-DWOvXjR_.js';
18
- import '../store-9cAScOcb.js';
19
- import '../types-C7DGg5ex.js';
17
+ import { a as RunSplitTag, R as RunRecord } from '../run-record-BDH49H2E.js';
18
+ import { C as CostLedger, b as CostLedgerSummary, M as MaximumCharge, c as CostReceiptInput } from '../cost-ledger-DWy3XdJc.js';
19
+ import { T as TraceAnalystKindSpec } from '../kind-factory-ClZmO25A.js';
20
+ import '../store-C1YxJDEK.js';
21
+ import '../types-BkfcQnxV.js';
20
22
  import '@tangle-network/tcloud';
23
+ import 'zod';
24
+ import '../raw-provider-sink-C46HDghv.js';
21
25
  import '../verdict-C9MlYujm.js';
22
- import '@ax-llm/ax';
23
26
  import '../dataset-NENEzRgk.js';
24
- import '../store-BsVi7ncX.js';
25
- import '../schema-SGWcK9wa.js';
27
+ import '../store-DGqD0Pyo.js';
28
+ import '../schema-B3Q3l9Z_.js';
29
+ import '@ax-llm/ax';
26
30
  import '../judge-calibration-7C-IDmKr.js';
27
31
  import '../hosted/index.js';
28
- import '../insight-report-oMVxDTxl.js';
29
- import '../summary-report-DTNgQycC.js';
30
- import '../failure-cluster-C48PiReX.js';
31
- import 'zod';
32
+ import '../insight-report-DY4nDW9Q.js';
33
+ import '../summary-report-C5bKFfm-.js';
34
+ import '../failure-cluster-DOAcSJ87.js';
32
35
 
33
36
  /**
34
37
  * Lineage DAG — a git-graph of improvement candidates.
@@ -1468,12 +1471,12 @@ declare function fapoEscalationEntry<TScenario extends Scenario, TArtifact>(conf
1468
1471
  * It runs `runCampaign` once per profile (reusing its seeds, reps, bootstrap
1469
1472
  * CIs, resumability, and the `LabeledScenarioStore` capture flywheel), maps
1470
1473
  * every cell to a validated `RunRecord` carrying the real `tokenUsage` the
1471
- * dispatch reported via `ctx.cost.observeTokens`, and runs `assertRealBackend`
1474
+ * dispatch committed via `ctx.cost.runPaidCall`, and runs `assertRealBackend`
1472
1475
  * BY CONSTRUCTION before returning — so a stub-backend run fails loudly instead
1473
1476
  * of reporting a clean 0/N leaderboard.
1474
1477
  *
1475
1478
  * Dispatch contract: a dispatch that calls an LLM MUST report usage via
1476
- * `ctx.cost.observeTokens({ input, output })` (and cost via `ctx.cost.observe`).
1479
+ * `ctx.cost.runPaidCall({ execute, receipt })`.
1477
1480
  * A dispatch that reports zero tokens is indistinguishable from a stub and the
1478
1481
  * integrity guard treats it as one.
1479
1482
  */
@@ -1485,8 +1488,8 @@ declare class ProfileMatrixError extends AgentEvalError {
1485
1488
  constructor(message: string);
1486
1489
  }
1487
1490
  /** Dispatch for one cell: render `profile` against `scenario`, returning the
1488
- * artifact the judges score. Report LLM usage via `ctx.cost.observeTokens`
1489
- * and `ctx.cost.observe` — the integrity guard depends on it. */
1491
+ * artifact the judges score. Run LLM work through `ctx.cost.runPaidCall`
1492
+ * the integrity check depends on its receipt. */
1490
1493
  type ProfileDispatchFn<TScenario extends Scenario, TArtifact> = (profile: AgentProfile, scenario: TScenario, ctx: DispatchContext) => Promise<TArtifact>;
1491
1494
  interface RunProfileMatrixOptions<TScenario extends Scenario, TArtifact> {
1492
1495
  /** Axis 3 — the agent-under-test configurations. Each is one column. */
@@ -1637,7 +1640,7 @@ interface PlaybackContext extends DispatchContext {
1637
1640
  * `SandboxPlaybackDriver` (real API / sandbox workspace) and
1638
1641
  * `PlaywrightPlaybackDriver` (real UI) — because they depend on runtime /
1639
1642
  * browser infra the substrate must not import. The driver MUST report LLM
1640
- * usage via `ctx.cost.observeTokens` so the backend-integrity guard sees real
1643
+ * usage through `ctx.cost.runPaidCall` so the backend-integrity check sees real
1641
1644
  * tokens (a run that never reports tokens reads as a stub).
1642
1645
  */
1643
1646
  interface PlaybackDriver<TStory extends UserStory = UserStory> {
@@ -1950,10 +1953,14 @@ interface ProposePatchesArgs {
1950
1953
  /** How many candidate patches to propose. */
1951
1954
  count: number;
1952
1955
  signal: AbortSignal;
1956
+ costLedger?: CostLedger;
1957
+ costPhase?: string;
1953
1958
  }
1954
1959
  interface SkillOptProposerOptions {
1955
1960
  llm: LlmClientOptions;
1956
1961
  model: string;
1962
+ /** Optional ledger for direct proposer use. Campaign context takes precedence. */
1963
+ costLedger?: CostLedger;
1957
1964
  /** What the skill document governs — orients the prompt. */
1958
1965
  target: string;
1959
1966
  /** Default ops-per-patch cap when used as a bare `SurfaceProposer`. The
@@ -2077,6 +2084,8 @@ interface RunSkillOptResult {
2077
2084
  /** Total cost across every scoring campaign (train evidence + holdout
2078
2085
  * acceptance) the hill-climb ran. */
2079
2086
  totalCostUsd: number;
2087
+ /** Run-wide spend, including scoring, proposals, and judges. */
2088
+ cost: CostLedgerSummary;
2080
2089
  }
2081
2090
  /**
2082
2091
  * SkillOpt sequential hill-climb: each epoch reflects on train-scenario weaknesses, proposes bounded patches, accepts the first patch that strictly improves the held-out composite, and anneals the edit budget on consecutive rejections.
@@ -2191,6 +2200,11 @@ interface HaloProposerOptions {
2191
2200
  model?: string;
2192
2201
  /** Model used to APPLY halo's findings to the prompt surface. Default = `model`. */
2193
2202
  applyModel?: string;
2203
+ /** Optional ledger for direct proposer use. Campaign context takes precedence. */
2204
+ costLedger?: CostLedger;
2205
+ analysisMaximumCharge?: MaximumCharge;
2206
+ analysisReceipt?: (report: string) => CostReceiptInput;
2207
+ applyMaxTokens?: number;
2194
2208
  /** The real halo binary. Default 'halo' (from `pip install halo-engine`). */
2195
2209
  haloBin?: string;
2196
2210
  /** Resolve the OTLP traces (JSONL string) halo should analyze for THIS
@@ -2323,6 +2337,8 @@ interface PolicyEditHistoryGenerationContext {
2323
2337
  interface LlmPolicyEditProposerOptions {
2324
2338
  llm: LlmClientOptions;
2325
2339
  model: string;
2340
+ /** Optional ledger for direct proposer use. Campaign context takes precedence. */
2341
+ costLedger?: CostLedger;
2326
2342
  /** Plain-language description of the JSON surface being improved. */
2327
2343
  target: string;
2328
2344
  /** PolicyEdit target surface every authored edit must retain. */
@@ -2396,6 +2412,8 @@ declare function projectPolicyEditHistory(history: readonly GenerationRecord[],
2396
2412
  */
2397
2413
 
2398
2414
  interface MemoryCurationProposerOptions {
2415
+ /** Optional ledger for direct proposer use. Campaign context takes precedence. */
2416
+ costLedger?: CostLedger;
2399
2417
  /** Top-K lessons retained in the surface memory block. Default 12. */
2400
2418
  maxEntries?: number;
2401
2419
  /** Heading rendered above the lessons inside the block. Default below. */
@@ -2408,6 +2426,7 @@ interface MemoryCurationProposerOptions {
2408
2426
  baseUrl: string;
2409
2427
  apiKey?: string;
2410
2428
  model: string;
2429
+ maxTokens?: number;
2411
2430
  fetchImpl?: LlmClientOptions['fetch'];
2412
2431
  };
2413
2432
  }
@@ -2496,6 +2515,11 @@ interface TraceAnalystProposerOptions {
2496
2515
  /** Model used to APPLY findings to the prompt surface. Default = `model`.
2497
2516
  * Keep this EQUAL to haloProposer's `applyModel` for an apples-to-apples run. */
2498
2517
  applyModel?: string;
2518
+ /** Optional ledger for direct proposer use. Campaign context takes precedence. */
2519
+ costLedger?: CostLedger;
2520
+ analysisMaximumCharge?: MaximumCharge;
2521
+ analysisReceipt?: (report: string) => CostReceiptInput;
2522
+ applyMaxTokens?: number;
2499
2523
  /** Ax provider name. Default 'openai' — works for any OpenAI-compatible base
2500
2524
  * via `apiURL`. Use 'deepseek' to hit DeepSeek's native provider. */
2501
2525
  provider?: string;
@@ -3045,7 +3069,7 @@ interface WorktreeAdapter {
3045
3069
  /** Commit pending changes, freeze the exact Git objects + binary patch, and
3046
3070
  * verify the worktree still matches that identity. */
3047
3071
  finalize(worktree: Worktree, summary: string): Promise<CodeSurface>;
3048
- /** Remove the worktree (and its branch) called for losing candidates. */
3072
+ /** Idempotently remove the worktree and branch. Safe to retry after partial cleanup. */
3049
3073
  discard(worktree: Worktree): Promise<void>;
3050
3074
  }
3051
3075
  /** Typed failure from a `WorktreeAdapter` operation (create/finalize/discard) — wraps the underlying git error as `cause`. */