@tangle-network/agent-eval 0.137.0 → 0.139.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (172) hide show
  1. package/CHANGELOG.md +78 -0
  2. package/README.md +34 -0
  3. package/dist/analyst/index.d.ts +485 -104
  4. package/dist/analyst/index.d.ts.map +1 -1
  5. package/dist/analyst/index.js +10 -607
  6. package/dist/analyst/index.js.map +1 -1
  7. package/dist/{benchmark-YDrpumqB.js → benchmark-CYtcIF2V.js} +299 -159
  8. package/dist/benchmark-CYtcIF2V.js.map +1 -0
  9. package/dist/{benchmark-CHX4orG7.d.ts → benchmark-DDVdWcwA.d.ts} +67 -15
  10. package/dist/benchmark-DDVdWcwA.d.ts.map +1 -0
  11. package/dist/benchmark-command-BKfjOBJ5.js +4537 -0
  12. package/dist/benchmark-command-BKfjOBJ5.js.map +1 -0
  13. package/dist/benchmarks/index.d.ts +1 -1
  14. package/dist/benchmarks/index.js +1 -1
  15. package/dist/{benchmarks-DCLkQOmc.js → benchmarks-zxhy1QV3.js} +4 -3
  16. package/dist/{benchmarks-DCLkQOmc.js.map → benchmarks-zxhy1QV3.js.map} +1 -1
  17. package/dist/campaign/index.d.ts +5 -5
  18. package/dist/campaign/index.js +4 -3
  19. package/dist/{campaign-lgObcHFC.js → campaign-DrS6_hLd.js} +19 -11
  20. package/dist/campaign-DrS6_hLd.js.map +1 -0
  21. package/dist/canonical-D011XM8r.js +86 -0
  22. package/dist/canonical-D011XM8r.js.map +1 -0
  23. package/dist/cli.js +10 -3
  24. package/dist/cli.js.map +1 -1
  25. package/dist/{client-C8L6h6Wf.d.ts → client-BohnDFBq.d.ts} +4 -4
  26. package/dist/{client-C8L6h6Wf.d.ts.map → client-BohnDFBq.d.ts.map} +1 -1
  27. package/dist/{completion-verifier-DSyRNVzU.d.ts → completion-verifier-IPoP4fQO.d.ts} +178 -4
  28. package/dist/completion-verifier-IPoP4fQO.d.ts.map +1 -0
  29. package/dist/contract/index.d.ts +10 -10
  30. package/dist/contract/index.js +9 -8
  31. package/dist/contract/index.js.map +1 -1
  32. package/dist/control.d.ts +2 -2
  33. package/dist/{cost-ledger-D-5_-dhi.js → cost-ledger-CZ9diLxY.js} +90 -45
  34. package/dist/cost-ledger-CZ9diLxY.js.map +1 -0
  35. package/dist/{cost-ledger-D2o6JOrL.d.ts → cost-ledger-DKgyIWRj.d.ts} +6 -2
  36. package/dist/cost-ledger-DKgyIWRj.d.ts.map +1 -0
  37. package/dist/default-registry-B8vf7Rmf.d.ts +118 -0
  38. package/dist/default-registry-B8vf7Rmf.d.ts.map +1 -0
  39. package/dist/default-registry-BgJJItGr.js +2364 -0
  40. package/dist/default-registry-BgJJItGr.js.map +1 -0
  41. package/dist/dspy-rlm-engine-DTkVyDX-.js +344 -0
  42. package/dist/dspy-rlm-engine-DTkVyDX-.js.map +1 -0
  43. package/dist/{eval-campaign-CHqfLnff.js → eval-campaign-BmptJj50.js} +2 -2
  44. package/dist/{eval-campaign-CHqfLnff.js.map → eval-campaign-BmptJj50.js.map} +1 -1
  45. package/dist/exact-types-MaaFcllV.d.ts +234 -0
  46. package/dist/exact-types-MaaFcllV.d.ts.map +1 -0
  47. package/dist/external-optimizer-contracts-BrxY2Sli.d.ts +32 -0
  48. package/dist/external-optimizer-contracts-BrxY2Sli.d.ts.map +1 -0
  49. package/dist/{extract-usage-p-56bh8q.js → extract-usage-DZs601Va.js} +2 -2
  50. package/dist/{extract-usage-p-56bh8q.js.map → extract-usage-DZs601Va.js.map} +1 -1
  51. package/dist/{feedback-trajectory-N_F0PwHz.d.ts → feedback-trajectory-BJUWOkJM.d.ts} +3 -2
  52. package/dist/feedback-trajectory-BJUWOkJM.d.ts.map +1 -0
  53. package/dist/fuzz.d.ts +1 -1
  54. package/dist/fuzz.js +1 -1
  55. package/dist/{hf-dataset-DBJXXoY1.js → hf-dataset-XggBupCr.js} +2 -2
  56. package/dist/{hf-dataset-DBJXXoY1.js.map → hf-dataset-XggBupCr.js.map} +1 -1
  57. package/dist/hosted/index.d.ts +3 -3
  58. package/dist/{index-C-Pr4OWg.d.ts → index-BTm_P9aC.d.ts} +12 -11
  59. package/dist/index-BTm_P9aC.d.ts.map +1 -0
  60. package/dist/{index-U3RHOShi.d.ts → index-CWOPCJiw.d.ts} +2 -2
  61. package/dist/{index-U3RHOShi.d.ts.map → index-CWOPCJiw.d.ts.map} +1 -1
  62. package/dist/{index-DRNl6g_N.d.ts → index-CtR1xh4V.d.ts} +3 -3
  63. package/dist/{index-DRNl6g_N.d.ts.map → index-CtR1xh4V.d.ts.map} +1 -1
  64. package/dist/index-DEb46kc6.d.ts.map +1 -1
  65. package/dist/{index-BnP1QJUv.d.ts → index-_66rVpwN.d.ts} +5 -5
  66. package/dist/{index-BnP1QJUv.d.ts.map → index-_66rVpwN.d.ts.map} +1 -1
  67. package/dist/index.d.ts +35 -55
  68. package/dist/index.d.ts.map +1 -1
  69. package/dist/index.js +55 -514
  70. package/dist/index.js.map +1 -1
  71. package/dist/{insight-report-B9ooYH_g.d.ts → insight-report-Bu5Wi9tG.d.ts} +4 -4
  72. package/dist/{insight-report-B9ooYH_g.d.ts.map → insight-report-Bu5Wi9tG.d.ts.map} +1 -1
  73. package/dist/{integrity-CKxosZ5Z.d.ts → integrity-COTh3DTH.d.ts} +2 -2
  74. package/dist/{integrity-CKxosZ5Z.d.ts.map → integrity-COTh3DTH.d.ts.map} +1 -1
  75. package/dist/kind-factory-CFxA0JQX.js +2133 -0
  76. package/dist/kind-factory-CFxA0JQX.js.map +1 -0
  77. package/dist/ledger-core/index.js +2 -1
  78. package/dist/{ledger-core-t6sItivm.js → ledger-core-Dxz0Rkwa.js} +210 -99
  79. package/dist/ledger-core-Dxz0Rkwa.js.map +1 -0
  80. package/dist/{llm-client-DKB25jV8.js → llm-client-bkztEfIx.js} +5 -5
  81. package/dist/llm-client-bkztEfIx.js.map +1 -0
  82. package/dist/meta-eval/index.d.ts +2 -2
  83. package/dist/multishot/index.d.ts +2 -2
  84. package/dist/openapi.json +1 -1
  85. package/dist/{proposal-findings-DCawte-y.js → proposal-findings-2GIUo1et.js} +2 -68
  86. package/dist/proposal-findings-2GIUo1et.js.map +1 -0
  87. package/dist/{release-report-CofgVNZt.d.ts → release-report-fZarvIm-.d.ts} +3 -3
  88. package/dist/{release-report-CofgVNZt.d.ts.map → release-report-fZarvIm-.d.ts.map} +1 -1
  89. package/dist/{replay-K8FaC0CB.d.ts → replay-DjG4IG60.d.ts} +34 -143
  90. package/dist/replay-DjG4IG60.d.ts.map +1 -0
  91. package/dist/{replay-Bju0T8Ls.js → replay-SA4OB7O7.js} +48 -136
  92. package/dist/replay-SA4OB7O7.js.map +1 -0
  93. package/dist/reporting.d.ts +4 -4
  94. package/dist/{researcher-Da0Wj-bt.d.ts → researcher-BxhtGfKa.d.ts} +5 -5
  95. package/dist/{researcher-Da0Wj-bt.d.ts.map → researcher-BxhtGfKa.d.ts.map} +1 -1
  96. package/dist/{reward-hacking-CQ3hTCO3.d.ts → reward-hacking-CqSLiV51.d.ts} +2 -2
  97. package/dist/{reward-hacking-CQ3hTCO3.d.ts.map → reward-hacking-CqSLiV51.d.ts.map} +1 -1
  98. package/dist/rl.d.ts +5 -5
  99. package/dist/rl.js +1 -1
  100. package/dist/rollout/index.d.ts +1 -1
  101. package/dist/rollout/index.js +2 -2
  102. package/dist/{rollout-DQFl0UXA.js → rollout-8nj3mYvx.js} +2 -2
  103. package/dist/{rollout-DQFl0UXA.js.map → rollout-8nj3mYvx.js.map} +1 -1
  104. package/dist/{rubric-predictive-validity-C4sztLR3.d.ts → rubric-predictive-validity-DQBQj6uV.d.ts} +2 -2
  105. package/dist/{rubric-predictive-validity-C4sztLR3.d.ts.map → rubric-predictive-validity-DQBQj6uV.d.ts.map} +1 -1
  106. package/dist/{run-evidence-BDIircdA.d.ts → run-evidence-C4RcRQT5.d.ts} +3 -3
  107. package/dist/{run-evidence-BDIircdA.d.ts.map → run-evidence-C4RcRQT5.d.ts.map} +1 -1
  108. package/dist/{run-record-BPCa2rQ8.d.ts → run-record-CztDMXVF.d.ts} +2 -2
  109. package/dist/{run-record-BPCa2rQ8.d.ts.map → run-record-CztDMXVF.d.ts.map} +1 -1
  110. package/dist/{semantic-concept-judge-Bz64IckK.js → semantic-concept-judge-BuIJ9IfB.js} +49 -6
  111. package/dist/semantic-concept-judge-BuIJ9IfB.js.map +1 -0
  112. package/dist/{server-KjXZZUDX.js → server-DaCpLfi0.js} +3 -3
  113. package/dist/{server-KjXZZUDX.js.map → server-DaCpLfi0.js.map} +1 -1
  114. package/dist/single-run-lock-BTTtPZ9N.js +989 -0
  115. package/dist/single-run-lock-BTTtPZ9N.js.map +1 -0
  116. package/dist/{skill-usage-CFDLLlhF.d.ts → skill-usage-B-BFS8M2.d.ts} +65 -40
  117. package/dist/skill-usage-B-BFS8M2.d.ts.map +1 -0
  118. package/dist/{skillopt-optimization-method-f4o9sUT4.js → skillopt-optimization-method-BbGnCC53.js} +20 -979
  119. package/dist/skillopt-optimization-method-BbGnCC53.js.map +1 -0
  120. package/dist/{skillopt-optimization-method-BpbnlvAZ.d.ts → skillopt-optimization-method-_s0Tub7Y.d.ts} +11 -39
  121. package/dist/skillopt-optimization-method-_s0Tub7Y.d.ts.map +1 -0
  122. package/dist/{statistics-_7P642CN.d.ts → statistics-B5d0Zd-z.d.ts} +2 -2
  123. package/dist/{statistics-_7P642CN.d.ts.map → statistics-B5d0Zd-z.d.ts.map} +1 -1
  124. package/dist/store-otlp-DX4fGIcf.js +757 -0
  125. package/dist/store-otlp-DX4fGIcf.js.map +1 -0
  126. package/dist/{summary-report-DHipz9Kx.d.ts → summary-report-Cg7BifAM.d.ts} +3 -3
  127. package/dist/{summary-report-DHipz9Kx.d.ts.map → summary-report-Cg7BifAM.d.ts.map} +1 -1
  128. package/dist/tool-groups-CdYq22lX.d.ts +258 -0
  129. package/dist/tool-groups-CdYq22lX.d.ts.map +1 -0
  130. package/dist/traces.d.ts +7 -6
  131. package/dist/traces.js +5 -4
  132. package/dist/{types-CTGbIm57.d.ts → types-BBFNHxSK.d.ts} +5 -5
  133. package/dist/{types-CTGbIm57.d.ts.map → types-BBFNHxSK.d.ts.map} +1 -1
  134. package/dist/{types-CKswbJGO.d.ts → types-DoEYskCd.d.ts} +5 -5
  135. package/dist/{types-CKswbJGO.d.ts.map → types-DoEYskCd.d.ts.map} +1 -1
  136. package/dist/{types-CTvKfr5F.d.ts → types-uPrS6mD-.d.ts} +2 -2
  137. package/dist/{types-CTvKfr5F.d.ts.map → types-uPrS6mD-.d.ts.map} +1 -1
  138. package/dist/usage-receipt-CgxMEBZq.js +134 -0
  139. package/dist/usage-receipt-CgxMEBZq.js.map +1 -0
  140. package/dist/wire/index.d.ts +3 -3
  141. package/dist/wire/index.js +1 -1
  142. package/docs/trace-analysis.md +191 -385
  143. package/package.json +5 -4
  144. package/dist/analyze-runs-PVtnfjvA.d.ts +0 -72
  145. package/dist/analyze-runs-PVtnfjvA.d.ts.map +0 -1
  146. package/dist/benchmark-CHX4orG7.d.ts.map +0 -1
  147. package/dist/benchmark-YDrpumqB.js.map +0 -1
  148. package/dist/campaign-lgObcHFC.js.map +0 -1
  149. package/dist/completion-verifier-DSyRNVzU.d.ts.map +0 -1
  150. package/dist/concurrency-MUjT7VjM.js +0 -109
  151. package/dist/concurrency-MUjT7VjM.js.map +0 -1
  152. package/dist/cost-ledger-D-5_-dhi.js.map +0 -1
  153. package/dist/cost-ledger-D2o6JOrL.d.ts.map +0 -1
  154. package/dist/default-registry-CLXbRt0f.js +0 -2594
  155. package/dist/default-registry-CLXbRt0f.js.map +0 -1
  156. package/dist/default-registry-Dc5D_Loc.d.ts +0 -202
  157. package/dist/default-registry-Dc5D_Loc.d.ts.map +0 -1
  158. package/dist/feedback-trajectory-N_F0PwHz.d.ts.map +0 -1
  159. package/dist/index-C-Pr4OWg.d.ts.map +0 -1
  160. package/dist/ledger-core-t6sItivm.js.map +0 -1
  161. package/dist/llm-client-DKB25jV8.js.map +0 -1
  162. package/dist/proposal-findings-DCawte-y.js.map +0 -1
  163. package/dist/registry-BdM7SuTr.d.ts +0 -124
  164. package/dist/registry-BdM7SuTr.d.ts.map +0 -1
  165. package/dist/replay-Bju0T8Ls.js.map +0 -1
  166. package/dist/replay-K8FaC0CB.d.ts.map +0 -1
  167. package/dist/semantic-concept-judge-Bz64IckK.js.map +0 -1
  168. package/dist/skill-usage-CFDLLlhF.d.ts.map +0 -1
  169. package/dist/skillopt-optimization-method-BpbnlvAZ.d.ts.map +0 -1
  170. package/dist/skillopt-optimization-method-f4o9sUT4.js.map +0 -1
  171. package/dist/tools-DZk2Jn64.js +0 -1876
  172. package/dist/tools-DZk2Jn64.js.map +0 -1
package/CHANGELOG.md CHANGED
@@ -4,6 +4,84 @@ All notable changes to `@tangle-network/agent-eval` and its sibling `agent-eval-
4
4
 
5
5
  ---
6
6
 
7
+ ## [0.139.0] - 2026-07-31 - recursive RLM trace analysts and caller failure reasons
8
+
9
+ ### Added
10
+
11
+ - Trace analysts run official DSPy RLMs; the Ax stack is retired (#495).
12
+ The published CodeTraceBench evidence remains bound to the retired direct runner via the evidence digests; a fresh certified run must replace it before any accuracy number is attributed to the new engine.
13
+ - `CostLedger.reconcile` accepts a caller-supplied failure reason: `reconcile(callId, observed, { error })` settles a failed receipt carrying that reason, and supplying a reason implies failure.
14
+ 0.138.0 had narrowed `CostReceipt.error` to the ledger's own `'paid-call-failed'`, silently discarding caller reasons — a crash orphan settled as a successful $0 call.
15
+ The receipt schema accepts any non-empty reason again, so ledgers persisted before 0.138.0 parse.
16
+
17
+ ## [0.138.0] - 2026-07-30 - exact analyst runs with sealed receipts
18
+
19
+ ### Fixed
20
+
21
+ - A release version bump no longer invalidates the published analyst benchmark evidence.
22
+ The dependency-lock pin now tracks the current lockfiles, the evidence keeps its own creation-time digest and package version, and a test proves the two locks differ by the version stamp alone — a real dependency change still forces a new benchmark run or explicit retirement of the evidence.
23
+
24
+ ### Added
25
+
26
+ - `AnalystRegistry.runExact()` requires ordered analyst ids and an explicit value for every run-policy field.
27
+ It never inherits registry insertion order or the constructor's default budget.
28
+ Disabled budgets, timeouts, cancellation, cost attribution, tags, and prior findings use explicit `null` values.
29
+ Exact runs bind analyst, ledger, hook, chat, and policy identities with non-secret configuration digests.
30
+ They use the registry's shared serial execution path and persist canonical equal or weighted allocations instead of adding a second scheduler or allocator.
31
+ Exact receipts explicitly distinguish complete execution from a failed ordered prefix.
32
+ `ExactAnalystRunExecutionError.result` is always a canonical immutable failed receipt with valid completed work and accounting.
33
+ - `defineTraceAnalyst()` accepts canonical `executionConfig` and returns an exact-capable custom analyst when it is supplied, while preserving its existing minimal form.
34
+ - **Breaking:** `AnalystBenchmarkCase` now requires `clusterId` and `labelState`.
35
+ A case must identify its independent source unit and state whether labels prove an issue, prove no issue, or leave the outcome unknown.
36
+ The benchmark no longer guesses either field from an empty issue list.
37
+ - `agent-eval analyst-benchmark` compares an empty baseline with a real-model AgentRx or CodeTraceBench analyst through any OpenAI-compatible endpoint.
38
+ It requires an explicit case limit and immutable dataset revision, validates labeled spans before paid work, uses benchmark-specific output adapters, and writes complete JSON plus Markdown results.
39
+ CodeTraceBench cases also retain hashed final verification artifacts, parse known upstream result formats, and mark missing outcomes unavailable.
40
+ Limited hash samples report source-versus-selected class, agent, model, difficulty, and solved distributions without claiming representativeness.
41
+ The published all-row score is retained, while a calibrated view measures solved label-empty trajectories as trusted negatives and keeps failed or unknown label-empty trajectories unlabeled.
42
+ Interrupted runs persist hash-chained observations and resume only when public inputs, model settings, local paths, and endpoint still match.
43
+ Reports include pooled and per-case step-localization metrics, exact source-quote coverage, final-result availability, and imported runtime duration.
44
+ External runner failures retain reported token, cost, duration, and metadata instead of becoming telemetry gaps.
45
+ Public model runs use one structured model call over a bounded trace projection.
46
+ A durable run-wide cost ledger enforces `--max-cost-usd` across concurrency and resume.
47
+ Paid responses are cached under deterministic call ids before settlement, so resume neither loses a completed response nor creates a second reservation after interruption.
48
+ Completed results retain a digest of every behavior-defining source file and are read through one strict recursive schema.
49
+ The repository includes one pinned 32-case CodeTraceBench input, two complete 64-call GLM-5.2 Agent Eval runs, and a failure-inclusive run of pinned CodeTracer on the same trajectories.
50
+ Exact source, input, result, resume, usage, cost, and secret-scan checks are committed with the results.
51
+ - The Python package gains a `dspy` extra: `agent-eval-rpc[dspy]` runs official `dspy.RLM` in a sandboxed Deno/Pyodide child process with seven allowlisted trace tools and strict JSON I/O.
52
+
53
+ ### Changed
54
+
55
+ - **Breaking:** trace analysts are recursive research programs run through an explicit analysis engine.
56
+ `analyzeTraces` requires an `engine` (the DSPy RLM engine is the primary implementation) and reports engine iterations; `maxTurns`, `maxSubqueries`, `onTurn`, and `AnalyzeTracesTurnSnapshot` are gone.
57
+ `callLlmJson` remains only as the one-shot `direct` benchmark baseline.
58
+ - **Breaking:** `defineTraceAnalyst()` returns an inert `TraceAnalystDefinition` for registration instead of a registrable analyst, and no longer takes a `cost` declaration — analyst cost is always metered LLM usage.
59
+ `createTraceAnalystKind` is now `createTraceAnalyst`; `TraceAnalystKindSpec` and `CreateTraceAnalystKindOpts` are replaced by `TraceAnalystDefinition`.
60
+ - **Breaking:** `BuildTraceAnalystSurfaceDispatchOptions.analyze` receives `instructions` instead of `actorDescription`.
61
+ - **Breaking:** `SteeringOptimizerBackend` narrows to `'pairwise'`.
62
+ - Citation verification is store-backed: cited trace and span ids must resolve in the trace analysis store, and encoded or foreign ids are rejected.
63
+ - External-optimizer subprocess calls fail on HTTP 200 responses with zero input and output usage, and on output over-reservation after recording the actual charge.
64
+ - Relative external-optimizer runner commands resolve against the caller's working directory instead of the child's temporary directory.
65
+
66
+ ### Removed
67
+
68
+ - **Breaking:** the Ax analyst stack and the `@ax-llm/ax` dependency: `createAnalystAi`, `CreateAnalystAiConfig`, `structureFindings`, `StructureFindingsOptions`, `StructureFindingsResult`, `AxGepaSteeringOptimizer`, and `AxSteeringOptimizerConfig`.
69
+ - **Breaking:** `createPublicBenchmarkModelRunner`; the analyst benchmark CLI defaults to the DSPy RLM runner and keeps the one-shot runner as the explicit `direct` baseline.
70
+ - `buildTraceAnalystTools`; trace tools are built from the transport-neutral descriptors.
71
+
72
+ ### Fixed
73
+
74
+ - Canonical `trace://<trace>/span/<span>` evidence is classified as span evidence instead of artifact evidence.
75
+ - Public model output selects positive integer assistant step ids.
76
+ The runner builds canonical trace URIs and exact action excerpts from those spans, and rejects missing, non-assistant, or empty steps.
77
+ - Capped concurrent paid calls wait for active reservations to settle when their final spend may still fit.
78
+ They fail immediately only when committed spend plus the next enforced maximum exceeds the run limit.
79
+ - Releasing a single-run lock now removes its process exit listener instead of leaking one callback per completed campaign.
80
+ - Phoenix evaluator tests use OpenTelemetry Core 2.10 instead of the vulnerable 1.x transitive dependency.
81
+ - CodeTracer prediction adapters accept the published schema and both flat and grouped step-label outputs emitted by CodeTracer 0.2.
82
+ - The CodeTraceBench model prompt now matches the public incorrect-step task by scoring wrong actions that are later recovered, instead of treating final task success as proof that earlier steps were correct.
83
+ - JSON-text finding rows reach the existing per-row schema repair instead of failing the entire trace analyst response.
84
+
7
85
  ## [0.137.0] - 2026-07-29 - trace analyst measurement and review integrity
8
86
 
9
87
  ### Added
package/README.md CHANGED
@@ -357,6 +357,40 @@ pnpm tsx examples/selfimprove-quickstart/index.ts
357
357
  You do not need a runnable agent to analyze data you already captured.
358
358
  Use `analyzeRuns()` for `RunRecord[]`.
359
359
  For traces, run a registry of built-in or custom analysts, measure it on labeled issues and exact span locations, then turn only reviewed findings into eval data.
360
+ For a public quality check, convert CodeTraceBench with `traces import-codetracebench`, then run `agent-eval analyst-benchmark` against pinned labels.
361
+ The command compares an empty baseline with the official DSPy `RLM` trace analyst and records its trace reads, model calls, tokens, cost, runtime, and cited findings.
362
+
363
+ Use `AnalystRegistry.runExact()` when the caller, rather than registry defaults, must own every execution choice.
364
+ The ordered `analystIds` array is the execution order, and `null` explicitly disables optional budget, timeout, cancellation, cost, tag, or prior-finding channels.
365
+ Exact runs are serial; callers that need recursive or concurrent scheduling compose them through their runtime rather than adding a second scheduler here.
366
+
367
+ ```ts
368
+ const result = await registry.runExact('analysis-1', inputs, {
369
+ analystIds: ['failure-mode', 'improvement'],
370
+ budget: { kind: 'equal', totalUsd: 2 },
371
+ totalTimeoutMs: 30_000,
372
+ signal: null,
373
+ costLedger: null,
374
+ costLedgerIdentity: null,
375
+ costPhase: null,
376
+ tags: null,
377
+ priorFindings: null,
378
+ chainFindings: true,
379
+ missingInputMode: 'abort',
380
+ applyRegistryHooks: false,
381
+ useRegistryChat: false,
382
+ })
383
+ ```
384
+
385
+ Custom analysts passed to `runExact()` declare canonical `executionConfig`.
386
+ The same `defineTraceAnalyst()` helper returns an exact-capable analyst when that field is present.
387
+ Built-in analysts already declare it.
388
+ Trace analysts selected by `runExact()` also require `aiIdentity`, using the same non-secret `id`, `version`, and canonical `config` shape as cost ledgers, registry hooks, and registry chat clients.
389
+ Exact lifecycle hooks receive frozen snapshots for observation; they cannot rewrite the planned context.
390
+ Persisted results store configuration digests, not raw configuration.
391
+ The persisted plan records the exact equal or weighted allocation for every routed analyst, and archival validates summaries against that same plan.
392
+ Every exact receipt says whether it is `complete` or `failed`; a complete receipt must cover the full plan, while a failed receipt may contain only the executed prefix.
393
+ Any failure after an exact run starts rejects with `ExactAnalystRunExecutionError`; its immutable failed receipt preserves valid completed summaries, findings, usage, and cost.
360
394
 
361
395
  See [concepts](./docs/concepts.md), [customer paths](./docs/customer-journeys.md), and [trace analysis](./docs/trace-analysis.md).
362
396