evalrx 0.1.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (339) hide show
  1. evalrx/__init__.py +139 -0
  2. evalrx/agent_assets/__init__.py +2 -0
  3. evalrx/agent_assets/skills/README.md +28 -0
  4. evalrx/agent_assets/skills/eval-chart-style/SKILL.md +172 -0
  5. evalrx/agent_assets/skills/evalrx-report-ui/SKILL.md +116 -0
  6. evalrx/agent_assets/skills/nature-figure/LICENSE +201 -0
  7. evalrx/agent_assets/skills/nature-figure/README.md +412 -0
  8. evalrx/agent_assets/skills/nature-figure/SKILL.md +60 -0
  9. evalrx/agent_assets/skills/nature-figure/manifest.yaml +59 -0
  10. evalrx/agent_assets/skills/nature-figure/references/api.md +436 -0
  11. evalrx/agent_assets/skills/nature-figure/references/backend-selection.md +100 -0
  12. evalrx/agent_assets/skills/nature-figure/references/chart-types.md +281 -0
  13. evalrx/agent_assets/skills/nature-figure/references/common-patterns.md +350 -0
  14. evalrx/agent_assets/skills/nature-figure/references/demos.md +65 -0
  15. evalrx/agent_assets/skills/nature-figure/references/design-theory.md +439 -0
  16. evalrx/agent_assets/skills/nature-figure/references/figure-contract.md +93 -0
  17. evalrx/agent_assets/skills/nature-figure/references/figure-legend-conventions.md +71 -0
  18. evalrx/agent_assets/skills/nature-figure/references/nature-2026-observations.md +112 -0
  19. evalrx/agent_assets/skills/nature-figure/references/qa-contract.md +119 -0
  20. evalrx/agent_assets/skills/nature-figure/references/r-template-index.md +66 -0
  21. evalrx/agent_assets/skills/nature-figure/references/r-workflow.md +161 -0
  22. evalrx/agent_assets/skills/nature-figure/references/tutorials.md +251 -0
  23. evalrx/agent_assets/skills/nature-figure/static/core/contract.md +29 -0
  24. evalrx/agent_assets/skills/nature-figure/static/core/stance.md +37 -0
  25. evalrx/agent_assets/skills/nature-figure/static/fragments/backend/python.md +37 -0
  26. evalrx/agent_assets/skills/nature-figure/static/fragments/backend/r.md +44 -0
  27. evalrx/agent_assets/skills/outcome-driver-analysis/SKILL.md +213 -0
  28. evalrx/agent_assets/skills/outcome-driver-analysis/assets/analysis_report_template.md +53 -0
  29. evalrx/agent_assets/skills/outcome-driver-analysis/references/model_selection.md +72 -0
  30. evalrx/agent_assets/skills/outcome-driver-analysis/scripts/explanatory_var_eda.R +130 -0
  31. evalrx/agent_assets/skills/outcome-driver-analysis/scripts/explanatory_var_eda.py +150 -0
  32. evalrx/agent_assets/skills/outcome-driver-analysis/scripts/fit_outcome_model.R +181 -0
  33. evalrx/agent_assets/skills/outcome-driver-analysis/scripts/fit_outcome_model.py +186 -0
  34. evalrx/agent_assets/skills/outcome-driver-analysis/scripts/univariate_eda.R +149 -0
  35. evalrx/agent_assets/skills/outcome-driver-analysis/scripts/univariate_eda.py +177 -0
  36. evalrx/agent_assets/skills.py +27 -0
  37. evalrx/agent_runtime/__init__.py +78 -0
  38. evalrx/agent_runtime/_docker_runner.py +89 -0
  39. evalrx/agent_runtime/cli_runtime.py +103 -0
  40. evalrx/agent_runtime/cli_transcript.py +138 -0
  41. evalrx/agent_runtime/cli_types.py +68 -0
  42. evalrx/agent_runtime/codegen/__init__.py +5 -0
  43. evalrx/agent_runtime/codegen/runner.py +94 -0
  44. evalrx/agent_runtime/experiment_harness.py +117 -0
  45. evalrx/agent_runtime/factory.py +102 -0
  46. evalrx/agent_runtime/json_shape.py +44 -0
  47. evalrx/agent_runtime/judges/__init__.py +28 -0
  48. evalrx/agent_runtime/judges/agy.py +179 -0
  49. evalrx/agent_runtime/judges/autodetect.py +135 -0
  50. evalrx/agent_runtime/judges/claude.py +159 -0
  51. evalrx/agent_runtime/judges/codex.py +120 -0
  52. evalrx/agent_runtime/providers/__init__.py +21 -0
  53. evalrx/agent_runtime/providers/antigravity.py +31 -0
  54. evalrx/agent_runtime/providers/base.py +145 -0
  55. evalrx/agent_runtime/providers/claude_code.py +49 -0
  56. evalrx/agent_runtime/providers/codex.py +37 -0
  57. evalrx/agent_runtime/providers/gemini_cli.py +26 -0
  58. evalrx/agent_runtime/providers/kimi_cli.py +27 -0
  59. evalrx/agent_runtime/providers/opencode.py +27 -0
  60. evalrx/agent_runtime/providers/registry.py +58 -0
  61. evalrx/agent_runtime/sandbox.py +517 -0
  62. evalrx/agent_runtime/skill_audit.py +143 -0
  63. evalrx/agent_runtime/skills/__init__.py +19 -0
  64. evalrx/agent_runtime/skills/installer.py +68 -0
  65. evalrx/agent_runtime/skills/prompt_policy.py +86 -0
  66. evalrx/agent_runtime/skills/resolver.py +19 -0
  67. evalrx/analysis/__init__.py +132 -0
  68. evalrx/analysis/adjudicate.py +154 -0
  69. evalrx/analysis/analysis_module.py +361 -0
  70. evalrx/analysis/api.py +171 -0
  71. evalrx/analysis/case_studio.py +651 -0
  72. evalrx/analysis/cli.py +114 -0
  73. evalrx/analysis/dashboard.py +350 -0
  74. evalrx/analysis/eval_case_matrix.py +118 -0
  75. evalrx/analysis/eval_viz_theme.py +833 -0
  76. evalrx/analysis/explore_run.py +333 -0
  77. evalrx/analysis/explorer.py +1276 -0
  78. evalrx/analysis/failure_modes.py +607 -0
  79. evalrx/analysis/fused_pipeline.py +489 -0
  80. evalrx/analysis/holdout.py +300 -0
  81. evalrx/analysis/hypothesis_agent.py +230 -0
  82. evalrx/analysis/narration.py +177 -0
  83. evalrx/analysis/operationalize.py +442 -0
  84. evalrx/analysis/plain_language.py +42 -0
  85. evalrx/analysis/planner.py +283 -0
  86. evalrx/analysis/probe_search.py +203 -0
  87. evalrx/analysis/profile.py +268 -0
  88. evalrx/analysis/prompts/__init__.py +0 -0
  89. evalrx/analysis/prompts/explorer.py +417 -0
  90. evalrx/analysis/prompts/failure_modes.py +33 -0
  91. evalrx/analysis/prompts/holdout.py +27 -0
  92. evalrx/analysis/prompts/hypothesis_agent.py +78 -0
  93. evalrx/analysis/prompts/run_codebase.py +47 -0
  94. evalrx/analysis/prompts/stats_agent.py +72 -0
  95. evalrx/analysis/prompts/stats_tool_generator.py +43 -0
  96. evalrx/analysis/result_marker.py +47 -0
  97. evalrx/analysis/run_codebase.py +242 -0
  98. evalrx/analysis/run_view.py +205 -0
  99. evalrx/analysis/stage_views.py +93 -0
  100. evalrx/analysis/stats_agent.py +944 -0
  101. evalrx/analysis/stats_tool_agent.py +261 -0
  102. evalrx/analysis/stats_tool_generator.py +415 -0
  103. evalrx/analysis/stats_tools.py +1153 -0
  104. evalrx/analysis/trajectory_records.py +193 -0
  105. evalrx/analysis/workbench.py +431 -0
  106. evalrx/analyzers/__init__.py +42 -0
  107. evalrx/analyzers/agent/__init__.py +25 -0
  108. evalrx/analyzers/agent/counterfactual.py +84 -0
  109. evalrx/analyzers/agent/first_error_judge.py +96 -0
  110. evalrx/analyzers/agent/ignored_obs.py +81 -0
  111. evalrx/analyzers/agent/loop_detect.py +79 -0
  112. evalrx/analyzers/agent/reliability.py +165 -0
  113. evalrx/analyzers/agent/tool_shap.py +225 -0
  114. evalrx/analyzers/agent/trajectory_rubric.py +168 -0
  115. evalrx/analyzers/attention/__init__.py +19 -0
  116. evalrx/analyzers/attention/relative_attn.py +610 -0
  117. evalrx/analyzers/attention/rollout.py +73 -0
  118. evalrx/analyzers/attention/sink.py +56 -0
  119. evalrx/analyzers/attention/summary.py +190 -0
  120. evalrx/analyzers/attribution/__init__.py +6 -0
  121. evalrx/analyzers/attribution/generic_attn.py +31 -0
  122. evalrx/analyzers/attribution/gradcam.py +30 -0
  123. evalrx/analyzers/base.py +12 -0
  124. evalrx/analyzers/geometry/__init__.py +6 -0
  125. evalrx/analyzers/geometry/cka.py +70 -0
  126. evalrx/analyzers/geometry/linear_probe.py +157 -0
  127. evalrx/analyzers/hallucination/__init__.py +9 -0
  128. evalrx/analyzers/hallucination/chair.py +78 -0
  129. evalrx/analyzers/hallucination/opera.py +29 -0
  130. evalrx/analyzers/hallucination/pope.py +119 -0
  131. evalrx/analyzers/hallucination/selfcheck.py +155 -0
  132. evalrx/analyzers/hallucination/vcd.py +29 -0
  133. evalrx/analyzers/lens/__init__.py +7 -0
  134. evalrx/analyzers/lens/layer_contrast.py +133 -0
  135. evalrx/analyzers/lens/logit_lens.py +138 -0
  136. evalrx/analyzers/lens/tuned_lens.py +30 -0
  137. evalrx/analyzers/patching/__init__.py +5 -0
  138. evalrx/analyzers/patching/causal_trace.py +30 -0
  139. evalrx/analyzers/perturbation/__init__.py +23 -0
  140. evalrx/analyzers/perturbation/_shapley.py +54 -0
  141. evalrx/analyzers/perturbation/context_shap.py +174 -0
  142. evalrx/analyzers/perturbation/cot_faithfulness.py +239 -0
  143. evalrx/analyzers/perturbation/format_sensitivity.py +237 -0
  144. evalrx/analyzers/perturbation/mm_shap.py +146 -0
  145. evalrx/analyzers/perturbation/modality_ablation.py +196 -0
  146. evalrx/analyzers/perturbation/perturbation_battery.py +274 -0
  147. evalrx/analyzers/perturbation/prompt_contrast.py +265 -0
  148. evalrx/analyzers/perturbation/rise.py +94 -0
  149. evalrx/analyzers/perturbation/vl_shap.py +102 -0
  150. evalrx/analyzers/reasoning/__init__.py +33 -0
  151. evalrx/analyzers/reasoning/_text.py +328 -0
  152. evalrx/analyzers/reasoning/answer_extraction_audit.py +327 -0
  153. evalrx/analyzers/reasoning/arith_audit.py +226 -0
  154. evalrx/analyzers/reasoning/contamination.py +214 -0
  155. evalrx/analyzers/reasoning/knowledge_split.py +253 -0
  156. evalrx/analyzers/reasoning/self_repair.py +246 -0
  157. evalrx/analyzers/reasoning/step_rollout_value.py +216 -0
  158. evalrx/analyzers/reasoning/termination_audit.py +258 -0
  159. evalrx/analyzers/uncertainty/__init__.py +18 -0
  160. evalrx/analyzers/uncertainty/calibration.py +174 -0
  161. evalrx/analyzers/uncertainty/coverage_gap.py +199 -0
  162. evalrx/analyzers/uncertainty/entropy.py +90 -0
  163. evalrx/analyzers/uncertainty/logprob_entropy.py +69 -0
  164. evalrx/analyzers/uncertainty/self_consistency.py +204 -0
  165. evalrx/analyzers/uncertainty/verbalized_conf.py +64 -0
  166. evalrx/cli.py +411 -0
  167. evalrx/config.py +77 -0
  168. evalrx/contract/__init__.py +179 -0
  169. evalrx/contract/common.py +452 -0
  170. evalrx/contract/emit.py +948 -0
  171. evalrx/contract/export.py +237 -0
  172. evalrx/contract/m1.py +325 -0
  173. evalrx/contract/m2.py +317 -0
  174. evalrx/contract/m3.py +165 -0
  175. evalrx/contract/m4.py +130 -0
  176. evalrx/contract/m5.py +292 -0
  177. evalrx/contract/methodology.py +76 -0
  178. evalrx/contract/pre_m1.py +58 -0
  179. evalrx/contract/typescript.py +140 -0
  180. evalrx/core/__init__.py +85 -0
  181. evalrx/core/analyzer.py +174 -0
  182. evalrx/core/capability.py +54 -0
  183. evalrx/core/case.py +443 -0
  184. evalrx/core/experiment.py +106 -0
  185. evalrx/core/model.py +198 -0
  186. evalrx/core/pipeline.py +42 -0
  187. evalrx/core/registry.py +142 -0
  188. evalrx/core/result.py +64 -0
  189. evalrx/core/spec.py +173 -0
  190. evalrx/core/tokentype.py +165 -0
  191. evalrx/core/tool.py +92 -0
  192. evalrx/datasets/__init__.py +41 -0
  193. evalrx/datasets/base.py +68 -0
  194. evalrx/datasets/gui_os.py +52 -0
  195. evalrx/datasets/llm_qa.py +57 -0
  196. evalrx/datasets/pure_qa.py +12 -0
  197. evalrx/datasets/vlm_qa.py +695 -0
  198. evalrx/datasets/web_search_qa.py +52 -0
  199. evalrx/eval_agent/__init__.py +341 -0
  200. evalrx/eval_agent/_tools.py +81 -0
  201. evalrx/eval_agent/ab_runner.py +50 -0
  202. evalrx/eval_agent/agentic/__init__.py +43 -0
  203. evalrx/eval_agent/agentic/actions.py +216 -0
  204. evalrx/eval_agent/agentic/board.py +107 -0
  205. evalrx/eval_agent/agentic/loop.py +190 -0
  206. evalrx/eval_agent/agentic/tools.py +538 -0
  207. evalrx/eval_agent/checkpoint.py +57 -0
  208. evalrx/eval_agent/cli_agent.py +59 -0
  209. evalrx/eval_agent/cli_skills.py +5 -0
  210. evalrx/eval_agent/evolution.py +396 -0
  211. evalrx/eval_agent/git_manager.py +215 -0
  212. evalrx/eval_agent/hypothesis.py +172 -0
  213. evalrx/eval_agent/label_quarantine.py +209 -0
  214. evalrx/eval_agent/legacy.py +530 -0
  215. evalrx/eval_agent/log_schema.py +497 -0
  216. evalrx/eval_agent/loop.py +2159 -0
  217. evalrx/eval_agent/loop_reports.py +116 -0
  218. evalrx/eval_agent/model_instrumentation.py +282 -0
  219. evalrx/eval_agent/narration.py +193 -0
  220. evalrx/eval_agent/nl_runner.py +460 -0
  221. evalrx/eval_agent/orchestrator.py +61 -0
  222. evalrx/eval_agent/preregister.py +93 -0
  223. evalrx/eval_agent/prompts/__init__.py +1 -0
  224. evalrx/eval_agent/prompts/agentic.py +46 -0
  225. evalrx/eval_agent/prompts/case_discovery.py +25 -0
  226. evalrx/eval_agent/prompts/diagnosis.py +125 -0
  227. evalrx/eval_agent/prompts/experiment_writer.py +265 -0
  228. evalrx/eval_agent/prompts/explore_step.py +37 -0
  229. evalrx/eval_agent/prompts/fix_agent.py +257 -0
  230. evalrx/eval_agent/prompts/hypothesis_tester.py +15 -0
  231. evalrx/eval_agent/prompts/nl_runner.py +38 -0
  232. evalrx/eval_agent/prompts/probe_agent.py +25 -0
  233. evalrx/eval_agent/prompts/probe_candidate_generator.py +14 -0
  234. evalrx/eval_agent/prompts/probe_generator.py +35 -0
  235. evalrx/eval_agent/prompts/whitebox_probe_generator.py +38 -0
  236. evalrx/eval_agent/report.py +58 -0
  237. evalrx/eval_agent/run_context.py +354 -0
  238. evalrx/eval_agent/run_log.schema.json +1215 -0
  239. evalrx/eval_agent/run_logger_v2.py +1764 -0
  240. evalrx/eval_agent/run_metadata.py +208 -0
  241. evalrx/eval_agent/stages/__init__.py +56 -0
  242. evalrx/eval_agent/stages/case_discovery.py +293 -0
  243. evalrx/eval_agent/stages/diagnosis.py +1017 -0
  244. evalrx/eval_agent/stages/experiment_writer.py +1634 -0
  245. evalrx/eval_agent/stages/fix_agent.py +3916 -0
  246. evalrx/eval_agent/stages/fix_internals.py +499 -0
  247. evalrx/eval_agent/stages/fix_pipeline.py +725 -0
  248. evalrx/eval_agent/stages/fix_tiers.py +187 -0
  249. evalrx/eval_agent/stages/fix_tools.py +1034 -0
  250. evalrx/eval_agent/stages/hypothesis_tester.py +1014 -0
  251. evalrx/eval_agent/stages/probe.py +439 -0
  252. evalrx/eval_agent/stages/probe_agent.py +1079 -0
  253. evalrx/eval_agent/stages/probe_candidate_generator.py +128 -0
  254. evalrx/eval_agent/stages/probe_generator.py +326 -0
  255. evalrx/eval_agent/stages/probe_search_agent.py +106 -0
  256. evalrx/eval_agent/stages/protocol.py +112 -0
  257. evalrx/eval_agent/stages/repair_catalog.py +273 -0
  258. evalrx/eval_agent/stages/surgery.py +524 -0
  259. evalrx/eval_agent/stages/whitebox_probe_generator.py +351 -0
  260. evalrx/eval_agent/store.py +231 -0
  261. evalrx/logging_utils.py +112 -0
  262. evalrx/models/__init__.py +161 -0
  263. evalrx/models/_discover.py +101 -0
  264. evalrx/models/agent.py +380 -0
  265. evalrx/models/backends/__init__.py +58 -0
  266. evalrx/models/backends/api.py +169 -0
  267. evalrx/models/backends/base.py +57 -0
  268. evalrx/models/backends/gemini_compat.py +579 -0
  269. evalrx/models/backends/hf_local.py +2074 -0
  270. evalrx/models/backends/openai_compat.py +301 -0
  271. evalrx/models/backends/vllm_offline.py +116 -0
  272. evalrx/models/base.py +24 -0
  273. evalrx/models/blackbox/__init__.py +4 -0
  274. evalrx/models/blackbox/agent.py +31 -0
  275. evalrx/models/blackbox/base.py +29 -0
  276. evalrx/models/blackbox/gemini.py +279 -0
  277. evalrx/models/blackbox/llm_api.py +17 -0
  278. evalrx/models/blackbox/vlm_api.py +17 -0
  279. evalrx/models/compose.py +66 -0
  280. evalrx/models/inference.py +88 -0
  281. evalrx/models/paper_methods/__init__.py +8 -0
  282. evalrx/models/paper_methods/aad.py +53 -0
  283. evalrx/models/paper_methods/ifcd.py +204 -0
  284. evalrx/models/paper_methods/pai.py +164 -0
  285. evalrx/models/paper_methods/tcd.py +202 -0
  286. evalrx/models/paper_methods/vcd.py +45 -0
  287. evalrx/models/paper_methods/vicrop.py +137 -0
  288. evalrx/models/toolcodec.py +143 -0
  289. evalrx/models/tools/__init__.py +20 -0
  290. evalrx/models/tools/perception.py +300 -0
  291. evalrx/models/tools/visual.py +174 -0
  292. evalrx/models/whitebox/__init__.py +26 -0
  293. evalrx/models/whitebox/agent.py +31 -0
  294. evalrx/models/whitebox/base.py +24 -0
  295. evalrx/models/whitebox/qwen.py +61 -0
  296. evalrx/models/whitebox/qwen2_5_omni.py +29 -0
  297. evalrx/models/whitebox/qwen2_audio.py +25 -0
  298. evalrx/models/whitebox/qwen_omni.py +53 -0
  299. evalrx/models/whitebox/qwen_vl.py +62 -0
  300. evalrx/observability/__init__.py +21 -0
  301. evalrx/observability/envelope.py +122 -0
  302. evalrx/observability/outbox.py +111 -0
  303. evalrx/observability/tracer.py +882 -0
  304. evalrx/reporting/__init__.py +28 -0
  305. evalrx/reporting/case_study.py +947 -0
  306. evalrx/reporting/compiler.py +587 -0
  307. evalrx/reporting/dynamic.py +1882 -0
  308. evalrx/reporting/html_report.py +2225 -0
  309. evalrx/reporting/langfuse_exporter.py +38 -0
  310. evalrx/reporting/langfuse_source.py +155 -0
  311. evalrx/reporting/model.py +151 -0
  312. evalrx/reporting/run_events.py +184 -0
  313. evalrx/reporting/server.py +557 -0
  314. evalrx/reporting/stages.py +58 -0
  315. evalrx/reporting/static_export.py +142 -0
  316. evalrx/reporting/web_dist/index.html +146 -0
  317. evalrx/specs.py +727 -0
  318. evalrx/stats/__init__.py +47 -0
  319. evalrx/stats/api.py +192 -0
  320. evalrx/stats/bootstrap.py +86 -0
  321. evalrx/stats/ebh.py +27 -0
  322. evalrx/stats/evalue.py +98 -0
  323. evalrx/stats/friedman.py +138 -0
  324. evalrx/stats/mcnemar.py +40 -0
  325. evalrx/stats/multiplicity.py +159 -0
  326. evalrx/stats/subset_sampling.py +55 -0
  327. evalrx/term_links.py +43 -0
  328. evalrx/viz/__init__.py +7 -0
  329. evalrx/viz/labels.py +77 -0
  330. evalrx/viz/prompts.py +39 -0
  331. evalrx/viz/renderer.py +590 -0
  332. evalrx/viz/schema.py +36 -0
  333. evalrx/viz/style.py +134 -0
  334. evalrx-0.1.2.dist-info/METADATA +532 -0
  335. evalrx-0.1.2.dist-info/RECORD +339 -0
  336. evalrx-0.1.2.dist-info/WHEEL +5 -0
  337. evalrx-0.1.2.dist-info/entry_points.txt +3 -0
  338. evalrx-0.1.2.dist-info/licenses/LICENSE +121 -0
  339. evalrx-0.1.2.dist-info/top_level.txt +1 -0
@@ -0,0 +1,1764 @@
1
+ """RunLoggerV2 — the diagnose-loop logger.
2
+
3
+ The only logger: an earlier flat ``run_log.jsonl`` design this one replaced
4
+ coexisted with it for one migration window and has since been removed — see
5
+ ``RUN_LOGGER_V2.md`` (next to this file) for the full design rationale,
6
+ layout, and trade-offs; the four rules that shaped it:
7
+
8
+ 1. Few files. One JSON document per pipeline stage, not a scattered pile of
9
+ ``prompts/*.txt`` + ``artifacts/*.json`` + ``experiments/*.py`` + ...
10
+ 2. Same-type logging in one JSON. Every event of a given kind (all ``probe``
11
+ events, all ``model_call`` events, ...) lives in ONE array, in ONE file —
12
+ not one small file per call.
13
+ 3. M1..M5 each get their own folder. A reader who only cares about M3 opens
14
+ exactly one folder.
15
+ 4. Nothing but JSON, except real binary artifacts (images, audio, tensors). Code,
16
+ stdout, prompts, markdown summaries — all of that is now a STRING VALUE
17
+ inside the JSON, not a sibling ``.py``/``.txt``/``.md`` file.
18
+
19
+ Every ``log_*`` method keeps the same name, signature, and call-site
20
+ behavior for return values (e.g. ``log_probe``'s ``list[Path]``) the flat
21
+ JSONL logger it replaced had, so ``VLDiagnoseLoop(run_logger=...)`` /
22
+ ``ProbeAgent(run_logger=...)`` / ``AutoDiagnoseLoop(run_logger=...)`` never
23
+ needed to change in ``loop.py``, ``probe_agent.py``, or any ``stages/*.py``
24
+ across the switch.
25
+
26
+ Scope, by design:
27
+ - RunContext integration uses an external ephemeral runtime tree. Generated
28
+ text/code is captured into stage JSON and the runtime tree is removed at
29
+ finalization instead of becoming a forest of trial files.
30
+ - No human-readable Markdown summaries (``record.md``, ``outcome.md``) —
31
+ the same information is in the JSON for a renderer to build one from.
32
+ - Verbose console narration is a plain one-line-per-event summary, not a
33
+ multi-line stage narration.
34
+ Native Langfuse/OpenTelemetry mirroring (:class:`DiagnosticTracer`) IS kept,
35
+ reused unchanged — it is orthogonal to file layout.
36
+ """
37
+
38
+ from __future__ import annotations
39
+
40
+ import json
41
+ import os
42
+ import re
43
+ import tempfile
44
+ import threading
45
+ import uuid
46
+ import warnings
47
+ from datetime import datetime, timezone
48
+ from pathlib import Path
49
+ from typing import TYPE_CHECKING, Any
50
+
51
+ from evalrx.eval_agent.hypothesis import hypothesis_id
52
+ from evalrx.eval_agent.log_schema import RUN_LOG_SCHEMA_VERSION
53
+
54
+ if TYPE_CHECKING:
55
+ from evalrx.analysis.analysis_module import AnalysisReport
56
+ from evalrx.core.result import Result
57
+ from evalrx.eval_agent.hypothesis import Hypothesis
58
+ from evalrx.eval_agent.loop_reports import AutoDiagnoseReport
59
+ from evalrx.eval_agent.stages.diagnosis import DiagnosisResult
60
+ from evalrx.eval_agent.stages.surgery import InterventionResult
61
+
62
+ RUN_LOGGER_V2_VERSION = 1
63
+
64
+ _STAGES = ("M1", "M2", "M3", "M4", "M5")
65
+
66
+ #: A tag string not matching this falls back to _STAGE_ALIASES, then to the
67
+ #: run-level "unrouted" bucket (never silently dropped — see _resolve_stage).
68
+ _STAGE_RE = re.compile(r"m([1-5])", re.IGNORECASE)
69
+
70
+ #: Tags used somewhere in the codebase that carry no "m<N>" substring at all
71
+ #: (checked against every literal `module=`/`stage=` value passed to a
72
+ #: log_* method as of this writing — see RUN_LOGGER_V2.md's "routing" table).
73
+ _STAGE_ALIASES: dict[str, str] = {
74
+ "fix_pipeline": "M5",
75
+ "fix": "M5",
76
+ "explore": "M2",
77
+ }
78
+
79
+ #: Extensions treated as genuine binary media — the one thing rule 4 still
80
+ #: allows as a separate file. Everything else becomes a JSON string value.
81
+ _MEDIA_EXTS = frozenset({
82
+ ".npy", ".png", ".jpg", ".jpeg", ".gif", ".webp", ".bmp",
83
+ ".wav", ".mp3", ".flac", ".ogg", ".mp4", ".avi", ".mov",
84
+ })
85
+
86
+ #: Text/code file suffixes worth inlining from a sandbox workspace snapshot
87
+ #: (skip weights/binaries — those go through the media/artifact path instead).
88
+ _INLINE_SUFFIXES = frozenset(
89
+ {".py", ".json", ".jsonl", ".md", ".txt", ".yaml", ".yml", ".csv", ".log", ".toml"}
90
+ )
91
+ _INLINE_MAX_BYTES = 2_000_000 # skip (note-only) any single file larger than this
92
+
93
+
94
+ def _resolve_stage(tag: "str | None") -> "str | None":
95
+ """"m1_probe" / "M4_SURGERY" / "codegen_m2_stats" / "fix_pipeline" -> "M1".."M5".
96
+
97
+ Returns ``None`` when *tag* matches nothing — the caller must not drop
98
+ the event in that case; route it to the run-level "unrouted" bucket
99
+ instead (see ``RunLoggerV2._route``). A tag is never assumed unroutable
100
+ without trying both the regex AND the alias table.
101
+ """
102
+ if not tag:
103
+ return None
104
+ m = _STAGE_RE.search(tag)
105
+ if m:
106
+ return f"M{m.group(1)}"
107
+ return _STAGE_ALIASES.get(tag.strip().lower())
108
+
109
+
110
+ def _atomic_write_json(path: Path, obj: Any) -> None:
111
+ """Write *obj* as JSON to *path* such that a reader never sees a partial file.
112
+
113
+ Writes to a sibling temp file first, then ``os.replace`` (atomic on the
114
+ same filesystem) — a crash mid-write leaves the OLD complete file in
115
+ place, never a truncated one. This runs on every single logged event
116
+ (see the design doc's "durability" section for the cost trade-off that
117
+ was chosen deliberately here, not overlooked).
118
+ """
119
+ path.parent.mkdir(parents=True, exist_ok=True)
120
+ tmp = path.with_name(f".{path.name}.tmp{os.getpid()}")
121
+ tmp.write_text(json.dumps(obj, indent=2, default=str, ensure_ascii=False), encoding="utf-8")
122
+ os.replace(tmp, path)
123
+
124
+
125
+ def _inline_workspace(
126
+ workdir: "str | Path", media_dir: Path, *, run_dir: "Path | None" = None,
127
+ max_bytes: "int | None" = _INLINE_MAX_BYTES,
128
+ preserve_all: bool = False,
129
+ ) -> "dict[str, Any] | None":
130
+ """Read a sandbox working directory into a JSON-safe dict, inlining text.
131
+
132
+ Returns ``{"files": {relative_path: content_or_note}, "media": [rel_paths],
133
+ "skipped": n}`` or ``None`` when *workdir* does not exist. Text-like files
134
+ (see ``_INLINE_SUFFIXES``) are read and inlined verbatim; recognised media
135
+ extensions are COPIED into *media_dir* (a real binary artifact, rule 4's
136
+ one exception) with a path reference left in ``"media"``; anything else is
137
+ skipped with a one-line note so its existence is still visible.
138
+ """
139
+ import hashlib
140
+ import shutil
141
+
142
+ src = Path(workdir)
143
+ if not src.exists() or not src.is_dir():
144
+ return None
145
+ files: dict[str, Any] = {}
146
+ media: list[str] = []
147
+ media_files: dict[str, str] = {}
148
+ skipped = 0
149
+ for f in sorted(src.rglob("*")):
150
+ if not f.is_file():
151
+ continue
152
+ rel = str(f.relative_to(src))
153
+ suffix = f.suffix.lower()
154
+ inline_content = None
155
+ binary = suffix in _MEDIA_EXTS
156
+ if preserve_all and suffix not in _INLINE_SUFFIXES and not binary:
157
+ try:
158
+ inline_content = f.read_text(encoding="utf-8")
159
+ if "\x00" in inline_content:
160
+ binary = True
161
+ except UnicodeError:
162
+ binary = True
163
+ if binary:
164
+ media_dir.mkdir(parents=True, exist_ok=True)
165
+ # Path + content identify a snapshot: trials cannot collide, and
166
+ # later writes to the same source cannot replace earlier evidence.
167
+ hasher = hashlib.sha256(str(f.resolve()).encode("utf-8"))
168
+ with f.open("rb") as stream:
169
+ for chunk in iter(lambda: stream.read(1024 * 1024), b""):
170
+ hasher.update(chunk)
171
+ digest = hasher.hexdigest()[:12]
172
+ dest = media_dir / f"{f.stem}_{digest}{suffix}"
173
+ try:
174
+ shutil.copy2(f, dest)
175
+ try:
176
+ media.append(str(dest.relative_to(run_dir)) if run_dir else str(dest))
177
+ except ValueError:
178
+ media.append(str(dest))
179
+ media_files[rel] = media[-1]
180
+ except Exception: # noqa: BLE001
181
+ if preserve_all:
182
+ raise # do not let finalization delete unarchived evidence
183
+ skipped += 1
184
+ continue
185
+ if suffix not in _INLINE_SUFFIXES and inline_content is None:
186
+ files[rel] = f"<skipped: {suffix or 'no extension'}, not a recognised text type>"
187
+ skipped += 1
188
+ continue
189
+ try:
190
+ if max_bytes is not None and f.stat().st_size > max_bytes:
191
+ files[rel] = f"<skipped: {f.stat().st_size} bytes, over the inline cap>"
192
+ skipped += 1
193
+ continue
194
+ files[rel] = inline_content if inline_content is not None else f.read_text(encoding="utf-8", errors="replace")
195
+ except Exception as exc: # noqa: BLE001
196
+ if preserve_all:
197
+ raise
198
+ files[rel] = f"<could not read: {exc}>"
199
+ skipped += 1
200
+ return {"files": files, "media": media, "media_files": media_files, "skipped": skipped}
201
+
202
+
203
+ # Pure, stateless content-shaping helpers with no file-writing side effects.
204
+
205
+
206
+ def _case_snapshot(case: Any) -> "dict[str, Any]":
207
+ """Make a small, renderer-safe baseline record for an evidence example."""
208
+ if hasattr(case, "to_dict"):
209
+ value = case.to_dict()
210
+ elif isinstance(case, dict):
211
+ value = dict(case)
212
+ else:
213
+ value = {"id": str(getattr(case, "id", ""))}
214
+ inputs = value.get("inputs") if isinstance(value.get("inputs"), dict) else {}
215
+ return {
216
+ "id": str(value.get("id") or value.get("case_id") or ""),
217
+ "input": inputs.get("prompt") or value.get("prompt") or value.get("instruction") or "",
218
+ "baseline_output": value.get("observed", value.get("output")),
219
+ "expected": value.get("expected"),
220
+ "outcome": value.get("label") or value.get("status") or "unknown",
221
+ }
222
+
223
+
224
+ def _iter_cases(cases: Any) -> "list[Any]":
225
+ """Accept CaseBatch, a plain sequence, or a generator without assumptions."""
226
+ if cases is None:
227
+ return []
228
+ value = getattr(cases, "cases", cases)
229
+ try:
230
+ return list(value)
231
+ except TypeError:
232
+ return []
233
+
234
+
235
+ def _probe_examples(results: "dict[str, Any]", cases: Any) -> "list[dict[str, Any]]":
236
+ """Persist two real, bounded M1 walkthroughs beside aggregate findings.
237
+
238
+ A probe only becomes a before/after comparison when its analyzer explicitly
239
+ records both outputs. Otherwise this records an honest *baseline case +
240
+ check result* example; downstream UI must not call it an intervention.
241
+ """
242
+ snapshots: dict[str, dict[str, Any]] = {}
243
+ for case in _iter_cases(cases):
244
+ snapshot = _case_snapshot(case)
245
+ if snapshot["id"]:
246
+ snapshots[snapshot["id"]] = snapshot
247
+ output: list[dict[str, Any]] = []
248
+ used: set[str] = set()
249
+ for name, result in results.items():
250
+ findings = getattr(result, "findings", {}) or {}
251
+ rows = findings.get("per_case") or []
252
+ if not isinstance(rows, list):
253
+ continue
254
+ rows = sorted(
255
+ (row for row in rows if isinstance(row, dict)),
256
+ key=lambda row: 0 if str(snapshots.get(str(row.get("sample_id") or row.get("case_id") or ""), {}).get("outcome", "")).lower() == "fail" else 1,
257
+ )
258
+ for row in rows:
259
+ case_id = str(row.get("sample_id") or row.get("case_id") or "")
260
+ snapshot = snapshots.get(case_id)
261
+ if not snapshot or case_id in used:
262
+ continue
263
+ checked = {
264
+ str(key).replace("_", " "): value for key, value in row.items()
265
+ if key not in {"sample_id", "case_id"} and isinstance(value, (str, int, float, bool))
266
+ }
267
+ if not checked:
268
+ continue
269
+ used.add(case_id)
270
+ output.append({
271
+ "id": f"m1-{name}-{case_id}", "kind": "case_measurement", "case_id": case_id,
272
+ "probe_title": str(name).replace("_", " ").title(),
273
+ **snapshot, "check_result": checked,
274
+ "plain_reading": "This one case illustrates the recorded check. The aggregate M1 result uses all measured cases.",
275
+ "evidence_scope": "one recorded case within M1",
276
+ })
277
+ break
278
+ if len(output) >= 2:
279
+ break
280
+ return output
281
+
282
+
283
+ def _artifact_to_numpy(artifact: Any) -> "Any | None":
284
+ """Convert *artifact* to a numpy array, or return None if not possible.
285
+
286
+ Handles: torch.Tensor, list[torch.Tensor] (e.g. per-layer attentions),
287
+ and numpy arrays. A list of tensors is stacked along a new first axis so
288
+ that ``attentions`` (list of ``(heads, seq, seq)``) becomes
289
+ ``(layers, heads, seq, seq)`` — a single array that retains all the data.
290
+ """
291
+ try:
292
+ import numpy as np
293
+ except ImportError:
294
+ return None
295
+
296
+ if hasattr(artifact, "detach"): # torch.Tensor
297
+ return artifact.detach().cpu().float().numpy()
298
+ if isinstance(artifact, np.ndarray):
299
+ return artifact
300
+ if isinstance(artifact, list) and artifact and hasattr(artifact[0], "detach"):
301
+ try:
302
+ import torch
303
+ return torch.stack(artifact).detach().cpu().float().numpy()
304
+ except Exception: # noqa: BLE001
305
+ return None
306
+ return None
307
+
308
+
309
+ def _save_artifact_figure(artifact_dir: Path, stem: str, arr: Any) -> None:
310
+ """Save a matplotlib figure of *arr* when the shape and stem are recognised.
311
+
312
+ Dispatch table (first match wins):
313
+ - 4-D + ``attn`` in stem → mean over (layers, heads) → 2-D heatmap
314
+ - 3-D + ``attn`` in stem → mean over heads → 2-D heatmap
315
+ - 2-D + heatmap keyword → direct heatmap (viridis)
316
+ - 1-D + curve keyword → line plot
317
+ Skips silently when matplotlib is unavailable or the shape is unrecognised.
318
+ """
319
+ try:
320
+ import matplotlib.pyplot as plt
321
+ plt.ioff()
322
+ except ImportError:
323
+ return
324
+
325
+ key = stem.lower()
326
+ # Skip logit arrays — (seq, vocab) shape is too large for a useful figure
327
+ if "logit" in key:
328
+ return
329
+
330
+ _is_attn = any(k in key for k in ("attn", "attention"))
331
+
332
+ fig = None
333
+ try:
334
+ ndim = arr.ndim
335
+ if ndim == 4 and _is_attn:
336
+ mat = arr.mean(axis=(0, 1)) # (layers, heads, seq, seq) → (seq, seq)
337
+ n_layers, n_heads = arr.shape[0], arr.shape[1]
338
+ fig, ax = plt.subplots(figsize=(8, 7))
339
+ im = ax.imshow(mat, cmap="viridis", aspect="auto", vmin=0)
340
+ ax.set_title(f"{stem} (mean over {n_layers}L × {n_heads}H)")
341
+ plt.colorbar(im, ax=ax, fraction=0.046, pad=0.04)
342
+ plt.tight_layout()
343
+ elif ndim == 3 and _is_attn:
344
+ mat = arr.mean(axis=0) # (heads, seq, seq) → (seq, seq)
345
+ n_heads = arr.shape[0]
346
+ fig, ax = plt.subplots(figsize=(8, 7))
347
+ im = ax.imshow(mat, cmap="viridis", aspect="auto", vmin=0)
348
+ ax.set_title(f"{stem} (mean over {n_heads} heads)")
349
+ plt.colorbar(im, ax=ax, fraction=0.046, pad=0.04)
350
+ plt.tight_layout()
351
+ elif ndim == 2 and "diff" in key:
352
+ # Signed difference map (e.g. FAIL-mean minus PASS-mean attention):
353
+ # diverging colormap with symmetric limits so the sign is readable.
354
+ bound = float(max(abs(arr.min()), abs(arr.max()))) or 1.0
355
+ fig, ax = plt.subplots(figsize=(8, 7))
356
+ im = ax.imshow(arr, cmap="coolwarm", aspect="auto", vmin=-bound, vmax=bound)
357
+ ax.set_title(stem)
358
+ plt.colorbar(im, ax=ax, fraction=0.046, pad=0.04)
359
+ plt.tight_layout()
360
+ elif ndim == 2 and (_is_attn or any(k in key for k in ("rollout", "spatial", "map"))):
361
+ fig, ax = plt.subplots(figsize=(8, 7))
362
+ im = ax.imshow(arr, cmap="viridis", aspect="auto")
363
+ ax.set_title(stem)
364
+ plt.colorbar(im, ax=ax, fraction=0.046, pad=0.04)
365
+ plt.tight_layout()
366
+ elif ndim == 1 and any(k in key for k in ("entropy", "score", "prob", "weight", "rollout")):
367
+ fig, ax = plt.subplots(figsize=(8, 3))
368
+ ax.plot(arr)
369
+ ax.set_xlabel("position")
370
+ ax.set_ylabel(stem)
371
+ ax.set_title(stem)
372
+ plt.tight_layout()
373
+
374
+ if fig is not None:
375
+ fig.savefig(artifact_dir / f"{stem}.png", dpi=100, bbox_inches="tight")
376
+ except Exception: # noqa: BLE001
377
+ pass
378
+ finally:
379
+ if fig is not None:
380
+ plt.close(fig)
381
+
382
+
383
+ class _V2JsonFormatter:
384
+ """Renders one plain one-line console summary per event, for ``verbose=True``.
385
+
386
+ Intentionally simple — file layout was this module's ask, not console UX;
387
+ see the design doc.
388
+ """
389
+
390
+ @staticmethod
391
+ def line(stage: "str | None", event: str, payload: "dict[str, Any]") -> str:
392
+ where = f"[{stage}]" if stage else "[run]"
393
+ cycle = payload.get("cycle")
394
+ tail = f" cycle={cycle}" if cycle is not None else ""
395
+ return f"{where} {event}{tail}"
396
+
397
+
398
+ class RunLoggerV2:
399
+ """A tidy, from-scratch M1..M5 logger. See the module docstring + design doc.
400
+
401
+ Args:
402
+ run_dir: Directory to write into. Created if missing. Defaults to
403
+ ``runs_v2/<YYYYMMDD_HHMMSS>/`` relative to cwd.
404
+ verbose: Print a one-line raw summary of every event to stdout
405
+ (``[M1] probe cycle=0``). Ignored when *narrate* is set.
406
+ narrate: Print live, aligned M1-M5 narration instead — the same
407
+ visual style as ``evalrx explore``'s terminal output (see
408
+ :mod:`evalrx.eval_agent.narration.LoopNarrator`), built
409
+ from real per-stage counts instead of a raw event dump.
410
+ trace_id: Ties every event to one Langfuse trace; auto-generated if
411
+ omitted.
412
+ observability_mode: Forwarded to :class:`DiagnosticTracer` unchanged.
413
+
414
+ Layout written under *run_dir*::
415
+
416
+ run.json run-wide: run_start, cases, report_published,
417
+ loop_end, agent_decisions, agent_tool_calls,
418
+ unrouted (see _resolve_stage)
419
+ M1/log.json probe, model_calls, tool_codegen, tool_registry,
420
+ stage_skipped — all M1-tagged events
421
+ M2/log.json analysis, explore, ...
422
+ M3/log.json diagnosis, ...
423
+ M4/log.json surgery (hypothesis-verification kind), ...
424
+ M5/log.json surgery (intervention kind), experiment, fix, ...
425
+ M*/artifacts/ binary media for that stage only (rule 4's exception)
426
+ media/ case-level baseline media (images/audio referenced
427
+ by FailureCase.inputs)
428
+ artifacts/ run-global named JSON (save_artifact_json)
429
+ """
430
+
431
+ def __init__(
432
+ self,
433
+ run_dir: "str | Path | None" = None,
434
+ *,
435
+ verbose: bool = False,
436
+ narrate: bool = False,
437
+ trace_id: "str | None" = None,
438
+ observability_mode: "str | None" = None,
439
+ context: "Any | None" = None,
440
+ ) -> None:
441
+ if run_dir is None:
442
+ run_dir = Path("runs_v2") / datetime.now().strftime("%Y%m%d_%H%M%S")
443
+ self.run_dir = Path(run_dir)
444
+ self.run_dir.mkdir(parents=True, exist_ok=True)
445
+ self.run_json_path = self.run_dir / "run.json"
446
+
447
+ self.trace_id: str = trace_id or str(uuid.uuid4())
448
+ self.current_cycle: int = -1
449
+ # Live M1-M5 terminal narration (see evalrx.eval_agent.narration) --
450
+ # opt-in, takes over from the plain `verbose` one-liner below rather
451
+ # than stacking with it, so a run never prints each event twice.
452
+ self._narrator = None
453
+ if narrate:
454
+ from evalrx.eval_agent.narration import LoopNarrator
455
+
456
+ self._narrator = LoopNarrator()
457
+ self.verbose = verbose
458
+ self._context = context
459
+ self._closed = False
460
+ # Producers use this capability flag to keep text/code in log events
461
+ # instead of writing sibling files into trial directories.
462
+ self.inline_text_artifacts = True
463
+ self.preserve_full_model_io = True
464
+
465
+ # One in-memory doc per stage + one run-level doc. Every log_* method
466
+ # appends to the relevant bucket(s), then atomically rewrites exactly
467
+ # the doc(s) it touched — see _atomic_write_json.
468
+ self._lock = threading.RLock()
469
+ self._event_seq = 0
470
+ self._validate_events = bool(os.environ.get("EVALRX_VALIDATE_LOG"))
471
+ self._event_validator = None
472
+ self._run_doc: dict[str, Any] = {
473
+ "trace_id": self.trace_id,
474
+ "run_start": None,
475
+ "cases": [],
476
+ "report_published": [],
477
+ "diagnose_reports": [],
478
+ "manifest": None,
479
+ "loop_end": [],
480
+ "agent_decisions": [],
481
+ "agent_tool_calls": [],
482
+ "unrouted": [],
483
+ }
484
+ self._stage_docs: dict[str, dict[str, Any]] = {s: {} for s in _STAGES}
485
+ self._logged_case_ids: set[str] = set()
486
+ self._model_call_seq = 0
487
+ self._codegen_seq = 0
488
+ # See log_model_call / log_probe: calls are recorded immediately into
489
+ # M1's doc AND buffered here so log_probe can replay them into
490
+ # Langfuse nested under the right probe span once it exists.
491
+ self._pending_model_calls: "dict[int, list[dict[str, Any]]]" = {}
492
+
493
+ from evalrx.observability.tracer import DiagnosticTracer
494
+ # The SQLite delivery queue is runtime state, not part of the tidy run
495
+ # artifact. Keep it outside the run tree; langfuse_trace.json remains
496
+ # the durable, portable JSON trace bundled with the run.
497
+ outbox_dir = Path(tempfile.gettempdir()) / "evalrx-v2-outbox"
498
+ self._outbox_path = outbox_dir / f"{self.trace_id}.sqlite3"
499
+ self.tracer = DiagnosticTracer(
500
+ run_dir=self.run_dir, mode=observability_mode, auto_sync=True,
501
+ outbox_path=self._outbox_path,
502
+ )
503
+ self.tracer.trace_id = self.trace_id
504
+
505
+ self._flush_run()
506
+
507
+ # ------------------------------------------------------------------
508
+ # Internal: doc access, routing, durability
509
+ # ------------------------------------------------------------------
510
+
511
+ def _stage_dir(self, stage: str) -> Path:
512
+ d = self.run_dir / stage
513
+ d.mkdir(parents=True, exist_ok=True)
514
+ return d
515
+
516
+ def _stage_artifacts_dir(self, stage: str) -> Path:
517
+ d = self._stage_dir(stage) / "artifacts"
518
+ d.mkdir(parents=True, exist_ok=True)
519
+ return d
520
+
521
+ def _flush_run(self) -> None:
522
+ _atomic_write_json(self.run_json_path, self._run_doc)
523
+
524
+ def _flush_stage(self, stage: str) -> None:
525
+ _atomic_write_json(self._stage_dir(stage) / "log.json", self._stage_docs[stage])
526
+
527
+ def _bucket(self, stage: str, key: str) -> list:
528
+ return self._stage_docs[stage].setdefault(key, [])
529
+
530
+ def _stamp_event(self, key: str, record: dict[str, Any], stage: str) -> None:
531
+ """Assign durable event identity under ``_lock``; no telemetry side effects."""
532
+ event = {
533
+ "cases": "case_record", "model_calls": "model_call",
534
+ "diagnose_reports": "diagnose_report", "agent_decisions": "agent_decision",
535
+ "agent_tool_calls": "agent_tool",
536
+ }.get(key, key)
537
+ self._event_seq += 1
538
+ cycle = record.get("cycle", -1)
539
+ span = {
540
+ "run_start": "run_start", "probe": f"c{cycle}.m1",
541
+ "analysis": f"c{cycle}.m2", "diagnosis": f"c{cycle}.m3",
542
+ "explore": f"c{cycle}.explore", "surgery": f"c{cycle}.{stage.lower()}",
543
+ "fix": "fix", "agent_decision": f"s{record.get('step')}.decision",
544
+ "agent_tool": f"s{record.get('step')}.tool",
545
+ "stage_skipped": f"{stage.lower()}.skipped",
546
+ }.get(event, f"{stage.lower()}.{event}.{self._event_seq}")
547
+ record.update(event=event, schema_version=RUN_LOG_SCHEMA_VERSION,
548
+ trace_id=self.trace_id, event_seq=self._event_seq,
549
+ stage=stage, span_id=span)
550
+ record.setdefault("ts", self._ts())
551
+ if self._validate_events:
552
+ try:
553
+ from evalrx.eval_agent.log_schema import _validator, build_schema
554
+
555
+ if self._event_validator is None:
556
+ self._event_validator = _validator(build_schema())
557
+ self._event_validator.validate(record)
558
+ except ImportError:
559
+ pass
560
+ except Exception as exc: # warn-only
561
+ warnings.warn(f"RunLoggerV2: event {event!r} violates log schema: {exc}")
562
+
563
+ def _append_stage(self, tag: "str | None", key: str, record: "dict[str, Any]") -> str:
564
+ """Route *record* by *tag* into the right stage bucket; flush; return the stage."""
565
+ stage = _resolve_stage(tag)
566
+ with self._lock:
567
+ if stage is None:
568
+ warnings.warn(
569
+ f"RunLoggerV2: could not route event {key!r} (tag={tag!r}) to a "
570
+ "stage — filed under run.json['unrouted'] instead of being lost.",
571
+ stacklevel=3,
572
+ )
573
+ record = {"key": key, "tag": tag, **record}
574
+ self._stamp_event("unrouted", record, "RUN")
575
+ self._run_doc["unrouted"].append(record)
576
+ self._flush_run()
577
+ return "unrouted"
578
+ self._stamp_event(key, record, stage)
579
+ self._bucket(stage, key).append(record)
580
+ self._flush_stage(stage)
581
+ if self._narrator is not None:
582
+ self._narrator.on_event(stage, key, record)
583
+ elif self.verbose:
584
+ print(_V2JsonFormatter.line(stage, key, record))
585
+ return stage
586
+
587
+ def _append_run(self, key: str, record: "dict[str, Any]") -> None:
588
+ """Append *record* to the (always list-valued) run.json bucket *key*.
589
+
590
+ ``run_start`` is the one run.json field that isn't a list — it's set
591
+ directly by ``log_run_start``, never through here.
592
+ """
593
+ with self._lock:
594
+ self._stamp_event(key, record, "RUN")
595
+ self._run_doc[key].append(record)
596
+ self._flush_run()
597
+ if self._narrator is not None:
598
+ self._narrator.on_run_event(key, record)
599
+ elif self.verbose:
600
+ print(_V2JsonFormatter.line(None, key, record))
601
+
602
+ @property
603
+ def managed_json_paths(self) -> "tuple[Path, ...]":
604
+ """Atomic JSON documents that may be rewritten while quarantine runs."""
605
+ return (self.run_json_path, *(self.run_dir / s / "log.json" for s in _STAGES))
606
+
607
+ @staticmethod
608
+ def _ts() -> str:
609
+ return datetime.now(timezone.utc).isoformat(timespec="microseconds")
610
+
611
+ def _save_media(self, stage: str, stem: str, artifact: Any) -> "str | None":
612
+ """Save a numeric artifact (tensor/array) + a rendered figure, if any.
613
+
614
+ Writes under this stage's ``artifacts/`` dir. Returns the ``.npy``
615
+ path (run-relative) or ``None`` when *artifact* isn't a recognised
616
+ numeric type.
617
+ """
618
+ try:
619
+ import numpy as np
620
+
621
+ arr = _artifact_to_numpy(artifact)
622
+ if arr is not None:
623
+ art_dir = self._stage_artifacts_dir(stage)
624
+ path = art_dir / f"{stem}.npy"
625
+ np.save(path, arr)
626
+ _save_artifact_figure(art_dir, stem, arr)
627
+ return str(path.relative_to(self.run_dir))
628
+ return None
629
+ except Exception as exc: # noqa: BLE001
630
+ warnings.warn(f"RunLoggerV2: could not save artifact {stem!r}: {exc}")
631
+ return None
632
+
633
+ def _save_case_media(self, path: Path) -> "str | None":
634
+ """Copy external case media into ``media/`` (content-hash-deduped); return rel path."""
635
+ import hashlib
636
+ import shutil
637
+
638
+ media_dir = self.run_dir / "media"
639
+ media_dir.mkdir(parents=True, exist_ok=True)
640
+ digest = hashlib.sha256()
641
+ try:
642
+ with path.open("rb") as handle:
643
+ for chunk in iter(lambda: handle.read(1024 * 1024), b""):
644
+ digest.update(chunk)
645
+ except OSError:
646
+ return None
647
+ copied = media_dir / f"{digest.hexdigest()[:16]}_{path.name}"
648
+ if not copied.exists():
649
+ shutil.copy2(path, copied)
650
+ return str(copied.relative_to(self.run_dir))
651
+
652
+ def _portable_path(self, value: "str | Path") -> str:
653
+ path = Path(value)
654
+ try:
655
+ return str(path.resolve().relative_to(self.run_dir.resolve()))
656
+ except (OSError, ValueError):
657
+ return str(value)
658
+
659
+ # ------------------------------------------------------------------
660
+ # Run provenance
661
+ # ------------------------------------------------------------------
662
+
663
+ def log_run_start(self, config: "dict[str, Any] | None" = None) -> None:
664
+ import platform
665
+
666
+ entry: dict[str, Any] = {"ts": self._ts(), "trace_id": self.trace_id}
667
+ if config:
668
+ entry.update(config)
669
+ entry.setdefault("python_version", platform.python_version())
670
+ try:
671
+ from evalrx import __version__ as _ver # type: ignore
672
+ entry.setdefault("evalrx_version", _ver)
673
+ except Exception: # noqa: BLE001
674
+ pass
675
+ commit = self._git_commit()
676
+ if commit:
677
+ entry.setdefault("git_commit", commit)
678
+ with self._lock:
679
+ self._stamp_event("run_start", entry, "RUN")
680
+ self._run_doc["run_start"] = entry
681
+ self._flush_run()
682
+ if self._narrator is not None:
683
+ self._narrator.on_run_start(entry)
684
+ elif self.verbose:
685
+ print(_V2JsonFormatter.line(None, "run_start", entry))
686
+
687
+ model_name = str(entry.get("model") or "Target Model")
688
+ proto = entry.get("protocol") or {}
689
+ proto_desc = proto.get("description", "") if isinstance(proto, dict) else str(proto)
690
+ bench_name = str(entry.get("benchmark_name") or proto_desc or "Benchmark")
691
+ self.tracer.start_trace(
692
+ model=model_name, benchmark=bench_name,
693
+ n_cases=int(entry.get("n_cases", 0) or 0), metadata=entry,
694
+ )
695
+
696
+ @staticmethod
697
+ def _git_commit() -> "str | None":
698
+ import subprocess
699
+
700
+ try:
701
+ out = subprocess.run(
702
+ ["git", "rev-parse", "--short", "HEAD"],
703
+ capture_output=True, text=True, timeout=3, check=False,
704
+ )
705
+ commit = out.stdout.strip()
706
+ if commit:
707
+ return commit
708
+ except Exception: # noqa: BLE001
709
+ pass
710
+ return os.environ.get("EVALRX_GIT_COMMIT") or None
711
+
712
+ def log_cases(self, cases: "Any", *, split: "str | None" = None) -> None:
713
+ """Persist complete case I/O; media is copied into ``media/`` (rule 4's exception).
714
+
715
+ ``split`` is the partition the loop assigned (``explore`` / ``confirm`` /
716
+ ``test``), recorded on the row so a report can group the cases the way
717
+ the run actually used them.
718
+ """
719
+ for case in cases:
720
+ case_id = str(getattr(case, "id", "") or "")
721
+ if not case_id or case_id in self._logged_case_ids:
722
+ continue
723
+ if hasattr(case, "to_dict"):
724
+ payload = case.to_dict()
725
+ elif isinstance(case, dict):
726
+ payload = dict(case)
727
+ else:
728
+ payload = {"id": case_id, "value": str(case)}
729
+ payload = json.loads(json.dumps(payload, ensure_ascii=False, default=str))
730
+ media_paths: list[str] = []
731
+ inputs = getattr(case, "inputs", None)
732
+ for kind in ("image", "audio", "video"):
733
+ value = getattr(inputs, kind, None)
734
+ if not isinstance(value, (str, Path)):
735
+ continue
736
+ path = Path(value)
737
+ if not path.is_absolute():
738
+ run_relative = self.run_dir / path
739
+ path = run_relative if run_relative.is_file() else path.resolve()
740
+ if not path.is_file():
741
+ continue
742
+ saved = self._save_case_media(path)
743
+ if saved:
744
+ media_paths.append(saved)
745
+ record: dict[str, Any] = {
746
+ "ts": self._ts(), "case_id": case_id, "case": payload, "media_paths": media_paths,
747
+ }
748
+ if split:
749
+ record["split"] = str(split)
750
+ self._append_run("cases", record)
751
+ self._logged_case_ids.add(case_id)
752
+
753
+ def log_report_published(self, envelope: "dict[str, Any]") -> None:
754
+ generated = envelope.get("generated_by") or {}
755
+ self._append_run("report_published", {
756
+ "ts": self._ts(),
757
+ "report_schema_version": int(envelope.get("schema_version") or 1),
758
+ "catalog_version": str(envelope.get("catalog_version") or ""),
759
+ "json_render_version": str(envelope.get("json_render_version") or ""),
760
+ "source_event_seq": int(envelope.get("source_event_seq") or 0),
761
+ "sha256": str(envelope.get("sha256") or ""),
762
+ "generated_by": generated,
763
+ "report_paths": ["report/report_data.json", "report/report_spec.json"],
764
+ })
765
+
766
+ def log_diagnose_report(
767
+ self,
768
+ report: Any,
769
+ cases: "list[Any]",
770
+ *,
771
+ discovery: "list[dict[str, Any]] | None" = None,
772
+ ) -> None:
773
+ """Inline the standard post-diagnosis report into ``run.json`` as one
774
+ detailed machine-readable record; a UI renders prose from this data
775
+ rather than reading separate JSON/Markdown siblings."""
776
+ hyps_src = getattr(report, "all_hypotheses", None)
777
+ if hyps_src is None:
778
+ hyps_src = getattr(report, "final_hypotheses", [])
779
+ hypotheses = [
780
+ {
781
+ "statement": h.statement,
782
+ "plain_statement": getattr(h, "plain_statement", ""),
783
+ "failure_mode": h.predicted_failure_mode,
784
+ "status": h.status.value if h.status else None,
785
+ }
786
+ for h in hyps_src
787
+ ]
788
+ m4_results = [
789
+ {
790
+ "hypothesis": tr.hypothesis.statement,
791
+ "failure_mode": tr.hypothesis.predicted_failure_mode,
792
+ "status": tr.status.value,
793
+ "effect_size": tr.effect_size,
794
+ "confidence": tr.confidence,
795
+ "protocol_consistent": tr.is_consistent_with_protocol,
796
+ "verdict": tr.verdict,
797
+ "evidence": tr.evidence,
798
+ }
799
+ for tr in getattr(report, "all_test_results", [])
800
+ ]
801
+ self._append_run("diagnose_reports", {
802
+ "ts": self._ts(),
803
+ "cycles": report.cycles,
804
+ "stopped_by": getattr(report, "stopped_by", None),
805
+ "resolved": getattr(report, "resolved", None),
806
+ "n_cases": len(cases),
807
+ "n_hypotheses": len(hypotheses),
808
+ "n_verified": len(getattr(report, "verified_hypotheses", [])),
809
+ "hypotheses": hypotheses,
810
+ "m4_results": m4_results,
811
+ "discovery": list(discovery or []),
812
+ })
813
+
814
+ def log_manifest(self, *, run_id: str, config: "dict[str, Any]") -> None:
815
+ """Record final run provenance and a compact file index in ``run.json``."""
816
+ files = [
817
+ str(path.relative_to(self.run_dir))
818
+ for path in sorted(self.run_dir.rglob("*"))
819
+ if path.is_file() and not path.name.startswith(".")
820
+ ]
821
+ with self._lock:
822
+ self._run_doc["manifest"] = {
823
+ "ts": self._ts(), "run_id": run_id, "config": dict(config), "files": files,
824
+ }
825
+ self._flush_run()
826
+
827
+ def log_runtime_snapshot(self, root: Path) -> None:
828
+ """Preserve execution files, including discarded trials, before cleanup."""
829
+ snapshot = _inline_workspace(
830
+ root, self._stage_artifacts_dir("M5"), run_dir=self.run_dir,
831
+ max_bytes=None, preserve_all=True,
832
+ )
833
+ with self._lock:
834
+ self._run_doc["runtime_snapshot"] = snapshot
835
+ self._flush_run()
836
+
837
+ def log_model_exchange(
838
+ self,
839
+ stage: str,
840
+ *,
841
+ role: str,
842
+ operation: str,
843
+ inputs: Any,
844
+ output: Any = None,
845
+ error: "str | None" = None,
846
+ duration_sec: "float | None" = None,
847
+ metadata: "dict[str, Any] | None" = None,
848
+ cycle: "int | None" = None,
849
+ ) -> None:
850
+ """Persist one exact model/agent input-output exchange in its stage."""
851
+ def json_safe(value: Any) -> Any:
852
+ import dataclasses
853
+
854
+ if dataclasses.is_dataclass(value):
855
+ value = dataclasses.asdict(value)
856
+ elif hasattr(value, "to_dict") and callable(value.to_dict):
857
+ try:
858
+ value = value.to_dict()
859
+ except Exception: # noqa: BLE001
860
+ pass
861
+ return json.loads(json.dumps(value, ensure_ascii=False, default=str))
862
+
863
+ entry: dict[str, Any] = {
864
+ "ts": self._ts(),
865
+ "cycle": self.current_cycle if cycle is None else cycle,
866
+ "role": role,
867
+ "operation": operation, "inputs": json_safe(inputs), "output": json_safe(output),
868
+ "error": error, "metadata": json_safe(dict(metadata or {})),
869
+ }
870
+ if duration_sec is not None:
871
+ entry["duration_sec"] = round(duration_sec, 4)
872
+ self._append_stage(stage, "model_calls", entry)
873
+
874
+ def save_artifact_json(self, stem: str, obj: Any) -> "str | None":
875
+ """Write *obj* as JSON under the run-global ``artifacts/`` dir; return rel path."""
876
+ try:
877
+ d = self.run_dir / "artifacts"
878
+ d.mkdir(parents=True, exist_ok=True)
879
+ path = d / stem
880
+ path.write_text(json.dumps(obj, indent=2, default=str), encoding="utf-8")
881
+ return str(path.relative_to(self.run_dir))
882
+ except Exception as exc: # noqa: BLE001
883
+ warnings.warn(f"RunLoggerV2: could not save artifact {stem!r}: {exc}")
884
+ return None
885
+
886
+ # ------------------------------------------------------------------
887
+ # M1 — target-model calls (see model_instrumentation.InstrumentedModel)
888
+ # ------------------------------------------------------------------
889
+
890
+ def log_model_call(
891
+ self,
892
+ *,
893
+ cycle: int,
894
+ analyzer: str,
895
+ call_index: int,
896
+ method: str,
897
+ inputs: Any,
898
+ kwargs: "dict[str, Any]",
899
+ output: Any,
900
+ duration_sec: float,
901
+ error: "str | None",
902
+ case_id: "str | None" = None,
903
+ batch_case_ids: "list[str] | None" = None,
904
+ n_batch_cases: "int | None" = None,
905
+ ) -> None:
906
+ record: dict[str, Any] = {
907
+ "ts": self._ts(), "cycle": cycle, "analyzer": analyzer,
908
+ "call_index": call_index, "method": method,
909
+ "case_id": case_id, "batch_case_ids": batch_case_ids or [],
910
+ "n_batch_cases": n_batch_cases if n_batch_cases is not None else len(batch_case_ids or []),
911
+ "inputs": inputs, "kwargs": kwargs, "output": output,
912
+ "duration_sec": round(duration_sec, 4),
913
+ }
914
+ if error is not None:
915
+ record["error"] = error
916
+ with self._lock:
917
+ self._model_call_seq += 1
918
+ record["seq"] = self._model_call_seq
919
+ self._stamp_event("model_calls", record, "M1")
920
+ self._bucket("M1", "model_calls").append(record)
921
+ self._flush_stage("M1")
922
+ self._pending_model_calls.setdefault(cycle, []).append(record)
923
+ if self.verbose:
924
+ print(_V2JsonFormatter.line("M1", "model_call", record))
925
+
926
+ # ------------------------------------------------------------------
927
+ # M1 — probe
928
+ # ------------------------------------------------------------------
929
+
930
+ def log_probe(
931
+ self,
932
+ cycle: int,
933
+ results: "dict[str, Result]",
934
+ schema: "Any | None" = None,
935
+ *,
936
+ cases: "Any | None" = None,
937
+ judge_prompt: "str | None" = None,
938
+ judge_raw: "str | None" = None,
939
+ duration_sec: "float | None" = None,
940
+ failed_analyzers: "dict[str, str] | None" = None,
941
+ ) -> "list[Path]":
942
+ """M1: one entry in M1/log.json's "probe" list. ``artifact_paths``/
943
+ results are inlined here instead of living in separate
944
+ ``.result.json`` files."""
945
+ artifact_paths: dict[str, str] = {}
946
+ overlay_pngs: list[Path] = []
947
+ result_docs: dict[str, Any] = {}
948
+ inline_artifacts: dict[str, Any] = {}
949
+ for name, result in results.items():
950
+ for art_name, artifact in getattr(result, "artifacts", {}).items():
951
+ stem = f"c{cycle}_{name}_{art_name}"
952
+ rel = self._save_media("M1", stem, artifact)
953
+ if rel is not None:
954
+ artifact_paths[f"{name}/{art_name}"] = rel
955
+ elif isinstance(artifact, (dict, list)):
956
+ inline_artifacts[f"{name}/{art_name}"] = json.loads(
957
+ json.dumps(artifact, ensure_ascii=False, default=str)
958
+ )
959
+ image_overlays = getattr(result, "image_overlays", None)
960
+ if image_overlays is not None:
961
+ try:
962
+ overlay_pngs.extend(
963
+ image_overlays(self._stage_artifacts_dir("M1"), f"c{cycle}_{name}")
964
+ )
965
+ except Exception as exc: # noqa: BLE001 - viz must never break the probe
966
+ warnings.warn(f"RunLoggerV2: image_overlays failed for {name}: {exc}")
967
+ to_dict = getattr(result, "to_dict", None)
968
+ if callable(to_dict):
969
+ try:
970
+ doc = to_dict()
971
+ summary = getattr(result, "summary", None)
972
+ if callable(summary):
973
+ doc["summary"] = summary()
974
+ result_docs[name] = doc
975
+ except Exception as exc: # noqa: BLE001
976
+ warnings.warn(f"RunLoggerV2: could not serialise result {name!r}: {exc}")
977
+
978
+ with self._lock:
979
+ pending_calls = [c for bucket in self._pending_model_calls.values() for c in bucket]
980
+ self._pending_model_calls.clear()
981
+
982
+ entry: dict[str, Any] = {
983
+ "ts": self._ts(), "cycle": cycle,
984
+ "analyzers": list(results),
985
+ "findings": {name: r.findings for name, r in results.items()},
986
+ "results": result_docs,
987
+ "artifact_paths": artifact_paths,
988
+ "artifacts": inline_artifacts,
989
+ "n_model_calls": len(pending_calls),
990
+ }
991
+ examples = _probe_examples(results, cases)
992
+ if examples:
993
+ entry["examples"] = examples
994
+ if failed_analyzers:
995
+ entry["failed_analyzers"] = dict(failed_analyzers)
996
+ if schema is not None:
997
+ entry["selection_rationale"] = getattr(schema, "rationale", "")
998
+ selected = getattr(schema, "selected_analyzers", None)
999
+ if selected is not None:
1000
+ entry["selected_analyzers"] = list(selected)
1001
+ if judge_prompt:
1002
+ entry["judge_prompt"] = judge_prompt
1003
+ if judge_raw:
1004
+ entry["judge_response"] = judge_raw
1005
+ if duration_sec is not None:
1006
+ entry["duration_sec"] = round(duration_sec, 3)
1007
+ self._append_stage("M1", "probe", entry)
1008
+ if judge_prompt or judge_raw:
1009
+ self.log_model_exchange(
1010
+ "M1", role="analyzer_selection_judge", operation="generate",
1011
+ inputs=judge_prompt or "", output=judge_raw or "", cycle=cycle,
1012
+ duration_sec=duration_sec,
1013
+ )
1014
+
1015
+ png_figures: list[Path] = list(overlay_pngs)
1016
+ for rel_npy in artifact_paths.values():
1017
+ if not rel_npy.endswith(".npy"):
1018
+ continue
1019
+ png = self.run_dir / (rel_npy[: -len(".npy")] + ".png")
1020
+ if png.exists():
1021
+ png_figures.append(png)
1022
+
1023
+ m1_span = self.tracer.start_span(
1024
+ name=f"M1: Multi-Dimensional Checkup (Cycle {cycle})",
1025
+ stage="M1",
1026
+ input_data={"analyzers": list(results.keys()), "selected_analyzers": entry.get("selected_analyzers", [])},
1027
+ metadata={"duration_sec": duration_sec, "artifacts": {
1028
+ "artifact_paths": artifact_paths, "figures": [str(p) for p in png_figures],
1029
+ }},
1030
+ )
1031
+ if judge_prompt or judge_raw:
1032
+ self.tracer.log_generation(
1033
+ name="M1 Analyzer Selection", model="judge",
1034
+ prompt=judge_prompt or "", completion=judge_raw or "", span_id=m1_span,
1035
+ )
1036
+ for name, r in results.items():
1037
+ findings = getattr(r, "findings", {}) or {}
1038
+ per_case = findings.get("per_case") or []
1039
+ probe_span = self.tracer.start_span(
1040
+ name=f"Probe: {name}", stage=f"M1_{name}", input_data={"probe": name},
1041
+ parent_id=m1_span,
1042
+ metadata={
1043
+ "n_scored": len(per_case) if per_case else (findings.get("n_cases") or findings.get("n_scored")),
1044
+ },
1045
+ )
1046
+ calls_for_analyzer = [c for c in pending_calls if c.get("analyzer") == name]
1047
+ for call in calls_for_analyzer[:50]:
1048
+ self.tracer.log_generation(
1049
+ name=f"{name} · {call.get('method')} #{call.get('call_index')}",
1050
+ model="target_model", prompt=call.get("inputs"),
1051
+ completion=call.get("error") or call.get("output"),
1052
+ span_id=probe_span,
1053
+ metadata={"duration_sec": call.get("duration_sec"), "error": call.get("error")},
1054
+ )
1055
+ self.tracer.end_span(probe_span, output_data={
1056
+ "findings": findings, "n_model_calls": len(calls_for_analyzer),
1057
+ })
1058
+ orphaned = {c["analyzer"] for c in pending_calls} - set(results)
1059
+ for name in orphaned:
1060
+ failed_span = self.tracer.start_span(
1061
+ name=f"Probe: {name} (failed)", stage=f"M1_{name}", input_data={"probe": name},
1062
+ parent_id=m1_span, metadata={"note": "analyzer raised before producing a Result"},
1063
+ )
1064
+ calls_for_analyzer = [c for c in pending_calls if c.get("analyzer") == name]
1065
+ for call in calls_for_analyzer[:50]:
1066
+ self.tracer.log_generation(
1067
+ name=f"{name} · {call.get('method')} #{call.get('call_index')}",
1068
+ model="target_model", prompt=call.get("inputs"),
1069
+ completion=call.get("error") or call.get("output"),
1070
+ span_id=failed_span,
1071
+ metadata={"duration_sec": call.get("duration_sec"), "error": call.get("error")},
1072
+ )
1073
+ self.tracer.end_span(failed_span, status="failed", output_data={
1074
+ "n_model_calls": len(calls_for_analyzer),
1075
+ })
1076
+ self.tracer.end_span(m1_span, output_data={"n_probes": len(results)})
1077
+ return png_figures
1078
+
1079
+ # ------------------------------------------------------------------
1080
+ # M2 — analysis + explore
1081
+ # ------------------------------------------------------------------
1082
+
1083
+ def log_analysis(
1084
+ self, cycle: int, report: "AnalysisReport", *, duration_sec: "float | None" = None,
1085
+ ) -> None:
1086
+ entry: dict[str, Any] = {
1087
+ "ts": self._ts(), "cycle": cycle,
1088
+ "severity": report.severity,
1089
+ "n_findings": len(report.findings),
1090
+ "findings": [str(f) for f in report.findings],
1091
+ "narrative": report.narrative,
1092
+ "descriptive_only": bool(getattr(report, "descriptive_only", False)),
1093
+ }
1094
+ stats_tool = getattr(report, "stats_tool", None)
1095
+ if stats_tool:
1096
+ entry["stats_tool"] = stats_tool
1097
+ fallback_reason = getattr(report, "llm_fallback_reason", None)
1098
+ if fallback_reason:
1099
+ entry["llm_fallback_reason"] = fallback_reason
1100
+ conclusion = getattr(report, "conclusion", None)
1101
+ if conclusion:
1102
+ entry["conclusion"] = conclusion
1103
+ evidence_chain = getattr(report, "evidence_chain", None)
1104
+ if evidence_chain:
1105
+ entry["evidence_chain"] = list(evidence_chain)
1106
+ stats_tool_results = getattr(report, "stats_tool_results", None)
1107
+ if stats_tool_results:
1108
+ entry["stats_tool_results"] = list(stats_tool_results)
1109
+ visualizations = getattr(report, "visualizations", None)
1110
+ if visualizations:
1111
+ entry["visualizations"] = list(visualizations)
1112
+ stats_plan = getattr(report, "stats_plan", None)
1113
+ if stats_plan:
1114
+ entry["stats_plan"] = stats_plan
1115
+ stats_results = getattr(report, "stats_results", None)
1116
+ if stats_results:
1117
+ entry["stats_results"] = [r.to_dict() for r in stats_results]
1118
+ corrected = getattr(report, "corrected_rejections", None)
1119
+ if corrected:
1120
+ entry["corrected_rejections"] = corrected
1121
+ figures = getattr(report, "figures", None)
1122
+ if figures:
1123
+ entry["figures"] = [self._portable_path(f) for f in figures]
1124
+ llm_prompt = getattr(report, "llm_prompt", None)
1125
+ llm_raw = getattr(report, "llm_raw", None)
1126
+ if llm_prompt:
1127
+ entry["judge_prompt"] = llm_prompt
1128
+ if llm_raw:
1129
+ entry["judge_response"] = llm_raw
1130
+ if duration_sec is not None:
1131
+ entry["duration_sec"] = round(duration_sec, 3)
1132
+ self._append_stage("M2", "analysis", entry)
1133
+ if llm_prompt or llm_raw:
1134
+ self.log_model_exchange(
1135
+ "M2", role="statistics_judge", operation="generate",
1136
+ inputs=llm_prompt or "", output=llm_raw or "", cycle=cycle,
1137
+ duration_sec=duration_sec,
1138
+ )
1139
+
1140
+ m2_span = self.tracer.start_span(
1141
+ name=f"M2: Screening & Confirmatory Signals (Cycle {cycle})", stage="M2",
1142
+ input_data={"severity": report.severity, "n_findings": len(report.findings)},
1143
+ metadata={"duration_sec": duration_sec, "stats_tool": stats_tool,
1144
+ "artifacts": {"figures": [str(f) for f in (figures or [])]}},
1145
+ )
1146
+ if llm_prompt or llm_raw:
1147
+ self.tracer.log_generation(
1148
+ name="M2 Statistical Screening Analysis", model="judge",
1149
+ prompt=llm_prompt or "", completion=llm_raw or "", span_id=m2_span,
1150
+ )
1151
+ for s in (stats_results or []):
1152
+ s_dict = s.to_dict() if hasattr(s, "to_dict") else (s if isinstance(s, dict) else {})
1153
+ sig_name = s_dict.get("config", {}).get("signal") or s_dict.get("tool") or "signal"
1154
+ eff = s_dict.get("effect")
1155
+ if eff is not None:
1156
+ self.tracer.log_score(
1157
+ name=f"m2_effect_{sig_name}", value=float(eff),
1158
+ comment=f"p={s_dict.get('p_value')}", span_id=m2_span,
1159
+ )
1160
+ self.tracer.end_span(m2_span, output_data={"conclusion": conclusion or ""})
1161
+
1162
+ def log_explore(
1163
+ self, cycle: int, report: "Any | None", *,
1164
+ out_dir: "Path | str | None" = None, duration_sec: "float | None" = None,
1165
+ ) -> None:
1166
+ ok = bool(getattr(report, "ok", False)) if report is not None else False
1167
+ charts = list(getattr(report, "charts", None) or []) if report is not None else []
1168
+ rendered = [
1169
+ str(c.get("figure_path")) for c in charts
1170
+ if isinstance(c, dict) and c.get("figure_path")
1171
+ ]
1172
+ workspace = None
1173
+ if out_dir is not None:
1174
+ workspace = _inline_workspace(
1175
+ out_dir, self._stage_artifacts_dir("M2"), run_dir=self.run_dir,
1176
+ )
1177
+ if workspace and workspace.get("media"):
1178
+ rendered = [
1179
+ path for path in workspace["media"]
1180
+ if Path(path).suffix.lower() in {".png", ".jpg", ".jpeg", ".gif", ".webp", ".bmp"}
1181
+ ]
1182
+ tables = getattr(report, "tables", None) or {}
1183
+ adjudication = dict(getattr(report, "adjudication", None) or {}) if report is not None else {}
1184
+ entry: dict[str, Any] = {
1185
+ "ts": self._ts(), "cycle": cycle, "ok": ok,
1186
+ "n_observations": len(getattr(report, "observations", None) or []) if report is not None else 0,
1187
+ "n_charts": len(charts), "n_charts_rendered": len(rendered),
1188
+ "n_tables": len(tables) if isinstance(tables, dict) else len(list(tables or [])),
1189
+ "n_candidate_signals": len(getattr(report, "candidate_signals", None) or []) if report is not None else 0,
1190
+ "n_hypotheses": len(getattr(report, "hypotheses", None) or []) if report is not None else 0,
1191
+ "adjudication": {
1192
+ k: adjudication[k] for k in (
1193
+ "method", "alpha", "split", "n_host_adjudicated", "n_rejected",
1194
+ "n_in_family", "n_descriptive_only",
1195
+ ) if k in adjudication
1196
+ },
1197
+ "observations": [str(o) for o in (getattr(report, "observations", None) or [])[:12]] if report is not None else [],
1198
+ "caveats": [str(c) for c in (getattr(report, "caveats", None) or [])[:8]] if report is not None else [],
1199
+ "figures": rendered,
1200
+ "workspace_snapshot": workspace,
1201
+ "attempts": int(getattr(report, "attempts", 0) or 0) if report is not None else 0,
1202
+ }
1203
+ error = str(getattr(report, "error", "") or "") if report is not None else "explorer produced no report"
1204
+ if error:
1205
+ entry["error"] = error
1206
+ if out_dir is not None and workspace is None:
1207
+ entry["out_dir"] = str(out_dir)
1208
+ report_path = Path(out_dir) / "exploratory_report.json"
1209
+ if report_path.exists():
1210
+ entry["report_path"] = str(report_path)
1211
+ if report is not None and getattr(report, "code", None):
1212
+ entry["code"] = str(report.code)
1213
+ if report is not None and getattr(report, "raw_outputs", None):
1214
+ entry["raw_outputs"] = [str(r) for r in report.raw_outputs]
1215
+ if duration_sec is not None:
1216
+ entry["duration_sec"] = round(duration_sec, 3)
1217
+ self._append_stage("M2", "explore", entry)
1218
+ for call in list(getattr(report, "model_calls", None) or []):
1219
+ self.log_model_exchange(
1220
+ "M2",
1221
+ role=str(call.get("role") or "explore_coder"),
1222
+ operation=str(call.get("operation") or "generate"),
1223
+ inputs=call.get("inputs"), output=call.get("output"),
1224
+ error=call.get("error"), duration_sec=call.get("duration_sec"),
1225
+ metadata=dict(call.get("metadata") or {}), cycle=cycle,
1226
+ )
1227
+
1228
+ exp_span = self.tracer.start_span(
1229
+ name=f"Explore: Free-form EDA (Cycle {cycle})", stage="EXPLORE",
1230
+ input_data={"ok": ok, "attempts": entry.get("attempts", 0)},
1231
+ metadata={"duration_sec": duration_sec, "artifacts": {
1232
+ "out_dir": str(out_dir) if out_dir is not None else None,
1233
+ "report_path": entry.get("report_path"), "figures": rendered,
1234
+ }},
1235
+ )
1236
+ if report is not None:
1237
+ for i, raw in enumerate(getattr(report, "raw_outputs", None) or []):
1238
+ self.tracer.log_generation(
1239
+ name=f"Explore Coder Agent (attempt {i + 1})", model="coder_agent",
1240
+ prompt=None, completion=str(raw), span_id=exp_span,
1241
+ )
1242
+ if getattr(report, "code", None):
1243
+ self.tracer.log_generation(
1244
+ name="Explore Analysis Code (analysis.py)", model="coder_agent",
1245
+ prompt=None, completion=str(report.code), span_id=exp_span,
1246
+ )
1247
+ self.tracer.end_span(exp_span, output_data={
1248
+ "n_observations": entry.get("n_observations", 0),
1249
+ "n_candidate_signals": entry.get("n_candidate_signals", 0),
1250
+ })
1251
+
1252
+ # ------------------------------------------------------------------
1253
+ # M3 — diagnosis
1254
+ # ------------------------------------------------------------------
1255
+
1256
+ def log_diagnosis(
1257
+ self, cycle: int, diag: "DiagnosisResult", *,
1258
+ duration_sec: "float | None" = None, explore_figures: "list[str] | None" = None,
1259
+ ) -> None:
1260
+ entry: dict[str, Any] = {
1261
+ "ts": self._ts(), "cycle": cycle,
1262
+ "model_name": diag.model_name,
1263
+ "n_hypotheses": len(diag.hypotheses),
1264
+ "hypotheses": [
1265
+ {
1266
+ # Join key for M4/M5 entries logged against this same
1267
+ # hypothesis later (log_surgery/log_experiment/log_fix) —
1268
+ # see evalrx.eval_agent.hypothesis.hypothesis_id.
1269
+ "id": hypothesis_id(h),
1270
+ "statement": h.statement, "plain_statement": h.plain_statement,
1271
+ "failure_mode": h.predicted_failure_mode,
1272
+ "status": h.status.value if h.status else None,
1273
+ "test_design": h.test_design,
1274
+ "critic": (h.metadata or {}).get("critic"),
1275
+ "critic_reason": (h.metadata or {}).get("critic_reason"),
1276
+ }
1277
+ for h in diag.hypotheses
1278
+ ],
1279
+ "raw_judge_output": diag.raw_judge_output,
1280
+ "n_critic_kept": int(getattr(diag, "n_critic_kept", 0) or 0),
1281
+ "n_critic_rejected": int(getattr(diag, "n_critic_rejected", 0) or 0),
1282
+ }
1283
+ critic_raw = getattr(diag, "critic_raw_output", "") or ""
1284
+ if critic_raw:
1285
+ entry["critic_raw_output"] = critic_raw
1286
+ entry["critic_prompt"] = getattr(diag, "critic_prompt", "") or ""
1287
+ proposed = list(getattr(diag, "proposed_hypotheses", None) or [])
1288
+ if proposed:
1289
+ entry["proposed_hypotheses"] = [
1290
+ {"statement": h.statement, "plain_statement": h.plain_statement,
1291
+ "failure_mode": h.predicted_failure_mode, "test_design": h.test_design}
1292
+ for h in proposed
1293
+ ]
1294
+ review_decisions = list(getattr(diag, "review_decisions", None) or [])
1295
+ if review_decisions:
1296
+ entry["review"] = {
1297
+ "n_kept": sum(d.get("decision") == "keep" for d in review_decisions),
1298
+ "n_rejected": sum(d.get("decision") == "reject" for d in review_decisions),
1299
+ "decisions": review_decisions,
1300
+ }
1301
+ referenced = getattr(diag, "referenced_charts", None)
1302
+ if referenced:
1303
+ entry["referenced_charts"] = list(referenced)
1304
+ if getattr(diag, "explore_context_used", False):
1305
+ entry["explore_context_used"] = True
1306
+ if getattr(diag, "failure_modes_used", False):
1307
+ entry["failure_modes_used"] = True
1308
+ if explore_figures:
1309
+ entry["explore_figures"] = list(explore_figures)
1310
+ m3_prompt = getattr(diag, "prompt", None) or ""
1311
+ if m3_prompt:
1312
+ entry["judge_prompt"] = m3_prompt
1313
+ review_prompt = getattr(diag, "review_prompt", None) or ""
1314
+ review_raw = getattr(diag, "review_raw", None) or ""
1315
+ if review_prompt or review_raw:
1316
+ entry["review_prompt"] = review_prompt
1317
+ entry["review_response"] = review_raw
1318
+ if duration_sec is not None:
1319
+ entry["duration_sec"] = round(duration_sec, 3)
1320
+ self._append_stage("M3", "diagnosis", entry)
1321
+ detailed_calls = list(getattr(diag, "model_calls", None) or [])
1322
+ if detailed_calls:
1323
+ for call in detailed_calls:
1324
+ self.log_model_exchange(
1325
+ "M3",
1326
+ role=str(call.get("role") or "diagnosis_judge"),
1327
+ operation=str(call.get("operation") or "generate"),
1328
+ inputs=call.get("inputs"), output=call.get("output"),
1329
+ error=call.get("error"), duration_sec=call.get("duration_sec"),
1330
+ metadata=dict(call.get("metadata") or {}), cycle=cycle,
1331
+ )
1332
+ else:
1333
+ # Compatibility for DiagnosisResult values created by older callers.
1334
+ if m3_prompt or diag.raw_judge_output:
1335
+ self.log_model_exchange(
1336
+ "M3", role="diagnosis_judge", operation="generate",
1337
+ inputs=m3_prompt, output=diag.raw_judge_output or "", cycle=cycle,
1338
+ duration_sec=duration_sec,
1339
+ )
1340
+ if critic_raw or review_prompt or review_raw:
1341
+ self.log_model_exchange(
1342
+ "M3", role="hypothesis_critic", operation="generate",
1343
+ inputs=entry.get("critic_prompt") or review_prompt,
1344
+ output=critic_raw or review_raw, cycle=cycle,
1345
+ )
1346
+
1347
+ m3_span = self.tracer.start_span(
1348
+ name=f"M3: Root-Cause Diagnosis (Cycle {cycle})", stage="M3",
1349
+ input_data={"model_name": diag.model_name, "n_hypotheses": len(diag.hypotheses)},
1350
+ metadata={"duration_sec": duration_sec},
1351
+ )
1352
+ self.tracer.log_generation(
1353
+ name="AI Doctor Diagnostician",
1354
+ model=str(self.tracer.trace_metadata.get("judge") or "diagnosis_judge"),
1355
+ prompt=m3_prompt, completion=diag.raw_judge_output or "", span_id=m3_span,
1356
+ metadata={"hypotheses": [h.statement for h in diag.hypotheses]},
1357
+ )
1358
+ if review_prompt or review_raw:
1359
+ self.tracer.log_generation(
1360
+ name="M3 Adversarial Evidence Review",
1361
+ model=str(self.tracer.trace_metadata.get("judge") or "diagnosis_judge"),
1362
+ prompt=review_prompt, completion=review_raw, span_id=m3_span,
1363
+ metadata={"decisions": review_decisions},
1364
+ )
1365
+ self.tracer.end_span(m3_span, output_data={
1366
+ "n_hypotheses": len(diag.hypotheses),
1367
+ "n_proposed": len(proposed) or len(diag.hypotheses),
1368
+ "review": entry.get("review"),
1369
+ })
1370
+
1371
+ # ------------------------------------------------------------------
1372
+ # M4/M5 — hypothesis verification & intervention
1373
+ # ------------------------------------------------------------------
1374
+
1375
+ def log_surgery(
1376
+ self, cycle: int, hypothesis: "Hypothesis", iv: "InterventionResult", *,
1377
+ validation_cases: "Any | None" = None, duration_sec: "float | None" = None,
1378
+ judge_prompt: "str | None" = None, judge_raw: "str | None" = None,
1379
+ ) -> None:
1380
+ """M4 (hypothesis verification) or M5 (intervention) — split on
1381
+ whether "m4_test_name" is present in ``iv.evidence``."""
1382
+ is_m4 = "m4_test_name" in (iv.evidence or {})
1383
+ stage = "M4" if is_m4 else "M5"
1384
+ entry: dict[str, Any] = {
1385
+ "ts": self._ts(), "cycle": cycle,
1386
+ "module": stage.lower(),
1387
+
1388
+ # Joins this verdict back to its M3 hypotheses[] entry (same id).
1389
+ "hypothesis_id": hypothesis_id(hypothesis),
1390
+ "hypothesis": hypothesis.statement,
1391
+ "failure_mode": hypothesis.predicted_failure_mode,
1392
+ "status": iv.status.value, "fixed": iv.fixed,
1393
+ "confidence_score": iv.confidence_score,
1394
+ "evidence_dimensions": iv.evidence_dimensions,
1395
+ "evidence": iv.evidence,
1396
+ "n_refocused_cases": len(iv.new_data) if iv.new_data else None,
1397
+ }
1398
+ if is_m4:
1399
+ candidates = _iter_cases(validation_cases)
1400
+ if candidates:
1401
+ snapshots = [_case_snapshot(case) for case in candidates]
1402
+ snapshot = next(
1403
+ (item for item in snapshots if str(item.get("outcome", "")).lower() == "fail"),
1404
+ snapshots[0],
1405
+ )
1406
+ entry["validation_examples"] = [{
1407
+ "id": f"m4-{snapshot.get('id')}", "kind": "validation_case",
1408
+ "case_id": snapshot.get("id"), **snapshot,
1409
+ "plain_reading": "This is one case in the independent validation pool. The verdict is determined from the full pool, not this case alone.",
1410
+ "evidence_scope": "one case in the independent validation pool",
1411
+ }]
1412
+ if judge_prompt:
1413
+ entry["judge_prompt"] = judge_prompt
1414
+ if judge_raw:
1415
+ entry["judge_response"] = judge_raw
1416
+ if duration_sec is not None:
1417
+ entry["duration_sec"] = round(duration_sec, 3)
1418
+ self._append_stage(stage, "surgery", entry)
1419
+ if judge_prompt or judge_raw:
1420
+ self.log_model_exchange(
1421
+ stage, role="protocol_consistency_judge", operation="generate",
1422
+ inputs=judge_prompt or "", output=judge_raw or "", cycle=cycle,
1423
+ duration_sec=duration_sec,
1424
+ )
1425
+
1426
+ stage_title = "M4 Adjudication" if is_m4 else "M5 Intervention"
1427
+ surg_span = self.tracer.start_span(
1428
+ name=f"{stage_title}: {hypothesis.statement[:60]}",
1429
+ stage="M4" if is_m4 else "M5_SURGERY",
1430
+ input_data={"hypothesis": hypothesis.statement, "failure_mode": hypothesis.predicted_failure_mode},
1431
+ metadata={"status": iv.status.value, "fixed": iv.fixed, "confidence_score": iv.confidence_score},
1432
+ )
1433
+ if judge_prompt or judge_raw:
1434
+ self.tracer.log_generation(
1435
+ name=f"{stage_title}: Protocol Consistency Judge", model="judge",
1436
+ prompt=judge_prompt or "", completion=judge_raw or "", span_id=surg_span,
1437
+ )
1438
+ if iv.confidence_score is not None:
1439
+ self.tracer.log_score(
1440
+ name="adjudication_confidence", value=float(iv.confidence_score),
1441
+ comment=f"status={iv.status.value}, fixed={iv.fixed}", span_id=surg_span,
1442
+ )
1443
+ self.tracer.end_span(surg_span, output_data={"evidence": iv.evidence or {}})
1444
+
1445
+ def log_experiment(
1446
+ self, cycle: int, hypothesis: "Hypothesis", iv: "InterventionResult", *,
1447
+ module: str = "m5",
1448
+ ) -> None:
1449
+ """The experiment the agent wrote and ran to test *hypothesis*.
1450
+
1451
+ No separate ``experiments/``/``workspace/`` files: the generated
1452
+ source (``exp["files"]``/``exp["code"]``), stdout/stderr, the coder
1453
+ agent's raw narration, and the validation log are already plain
1454
+ strings in *iv.experiment* and go straight into the entry as JSON
1455
+ string values.
1456
+ """
1457
+ exp = getattr(iv, "experiment", None) or {}
1458
+ files = exp.get("files") or {}
1459
+ if not files and exp.get("code"):
1460
+ files = {"main.py": exp["code"]}
1461
+
1462
+ # An "experiment" is always M4/M5-shaped content by definition, so it
1463
+ # gets the same resolve-with-M5-floor treatment as its workspace
1464
+ # media, rather than trusting `_append_stage`'s generic unroutable
1465
+ # fallback (which would file a genuinely M5-ish record under
1466
+ # run.json["unrouted"] if *module* is ever something the M1-M5 regex
1467
+ # and alias table don't recognize).
1468
+ stage = _resolve_stage(module) or "M5"
1469
+
1470
+ workspace = None
1471
+ workdir = exp.get("workdir")
1472
+ if workdir:
1473
+ workspace = _inline_workspace(
1474
+ workdir, self._stage_artifacts_dir(stage), run_dir=self.run_dir,
1475
+ )
1476
+
1477
+ entry: dict[str, Any] = {
1478
+ "ts": self._ts(), "cycle": cycle, "module": module,
1479
+ # Joins this experiment back to its M3 hypotheses[] entry (same id).
1480
+ "hypothesis_id": hypothesis_id(hypothesis),
1481
+ "hypothesis": hypothesis.statement,
1482
+ "failure_mode": hypothesis.predicted_failure_mode,
1483
+ "status": iv.status.value if iv.status else None,
1484
+ "fixed": iv.fixed,
1485
+ "provider": exp.get("provider"), "verdict": exp.get("verdict"),
1486
+ "metrics": exp.get("metrics"), "returncode": exp.get("returncode"),
1487
+ "timed_out": exp.get("timed_out"), "cli_usage": exp.get("cli_usage"),
1488
+ "llm_calls": exp.get("llm_calls"), "sandbox_runs": exp.get("sandbox_runs"),
1489
+ "code": files,
1490
+ "stdout": exp.get("stdout"), "stderr": exp.get("stderr"),
1491
+ "blueprint": exp.get("blueprint"), "cli_raw_output": exp.get("cli_raw_output"),
1492
+ "validation_log": list(exp.get("validation_log") or []) or None,
1493
+ "workspace_snapshot": workspace,
1494
+ "trial_root": exp.get("trial_root"),
1495
+ }
1496
+ self._append_stage(stage, "experiment", entry)
1497
+ for call in exp.get("model_calls") or []:
1498
+ if isinstance(call, dict):
1499
+ self.log_model_exchange(
1500
+ stage,
1501
+ role=str(call.get("role") or "experiment_writer"),
1502
+ operation=str(call.get("operation") or "generate"),
1503
+ inputs=call.get("inputs"), output=call.get("output"),
1504
+ error=call.get("error"), duration_sec=call.get("duration_sec"),
1505
+ metadata=call.get("metadata"),
1506
+ )
1507
+
1508
+ exp_span = self.tracer.start_span(
1509
+ name=f"{stage}: Experiment — {hypothesis.statement[:60]}",
1510
+ stage=f"{stage}_EXPERIMENT",
1511
+ input_data={"hypothesis": hypothesis.statement, "failure_mode": hypothesis.predicted_failure_mode},
1512
+ metadata={"provider": exp.get("provider"), "returncode": exp.get("returncode")},
1513
+ )
1514
+ cli_raw = exp.get("cli_raw_output")
1515
+ if cli_raw:
1516
+ self.tracer.log_generation(
1517
+ name=f"{module.upper()} Coder Agent", model=str(exp.get("provider") or "coder_agent"),
1518
+ prompt=None, completion=str(cli_raw), span_id=exp_span,
1519
+ )
1520
+ vlog = exp.get("validation_log")
1521
+ if vlog:
1522
+ self.tracer.log_generation(
1523
+ name=f"{module.upper()} Validation Log", model=str(exp.get("provider") or "coder_agent"),
1524
+ prompt=None, completion="\n".join(str(x) for x in vlog), span_id=exp_span,
1525
+ )
1526
+ self.tracer.end_span(exp_span, output_data={
1527
+ "status": entry.get("status"), "fixed": entry.get("fixed"), "verdict": entry.get("verdict"),
1528
+ })
1529
+
1530
+ def log_fix(self, outcome: "Any") -> None:
1531
+ """Post-loop fix module: the tiered repair attempt + recommendation.
1532
+
1533
+ Per-case outputs are NOT popped out to a sibling ``outputs.jsonl`` —
1534
+ they stay inline in ``M5/log.json`` under each candidate's own
1535
+ ``"outputs"`` key. Fewer files was the whole point; a bulkier single
1536
+ JSON is the intended trade for that.
1537
+ """
1538
+ d = json.loads(json.dumps(outcome.to_dict(), ensure_ascii=False, default=str))
1539
+ for attempt in [*(d.get("attempted") or []), *(d.get("selection_attempted") or [])]:
1540
+ trial_root = attempt.get("trial_root")
1541
+ if trial_root:
1542
+ attempt["workspace_snapshot"] = _inline_workspace(
1543
+ trial_root, self._stage_artifacts_dir("M5"),
1544
+ run_dir=self.run_dir, max_bytes=None, preserve_all=True,
1545
+ )
1546
+ best_ref = d.get("best")
1547
+ if isinstance(best_ref, dict):
1548
+ best = best_ref
1549
+ elif isinstance(best_ref, str):
1550
+ best = next((a for a in d.get("attempted") or [] if a.get("name") == best_ref), {})
1551
+ else:
1552
+ best = {}
1553
+ entry: dict[str, Any] = {"ts": self._ts(), "cycle": -1, "module": "fix"}
1554
+ entry.update(d)
1555
+ entry["best"] = best
1556
+ self._append_stage("M5", "fix", entry)
1557
+
1558
+ fix_span = self.tracer.start_span(
1559
+ name="M5: Targeted Repair & Confirmation", stage="M5_FIX",
1560
+ input_data={"candidates_evaluated": len(d.get("attempted", []))},
1561
+ )
1562
+ if best.get("effect") is not None:
1563
+ self.tracer.log_score(
1564
+ name="repair_net_accuracy_gain", value=float(best.get("effect", 0.0)),
1565
+ comment=f"cured={best.get('n_fixed')}, broken={best.get('n_broken')}",
1566
+ span_id=fix_span,
1567
+ )
1568
+ if (best.get("payload") or {}).get("prompt_template"):
1569
+ self.tracer.log_generation(
1570
+ name="Winning Repair Patch", model="evalrx_repair",
1571
+ prompt="Repair Candidate Search",
1572
+ completion=best["payload"]["prompt_template"], span_id=fix_span,
1573
+ )
1574
+ self.tracer.end_span(fix_span, output_data={"selected": best.get("name")})
1575
+
1576
+ # ------------------------------------------------------------------
1577
+ # AgenticDiagnoseLoop dispatch layer — run-level, not stage content
1578
+ # ------------------------------------------------------------------
1579
+
1580
+ def log_agent_decision(
1581
+ self, step: int, *, action: str, params: "dict[str, Any] | None" = None,
1582
+ rationale: str = "", valid: bool = True, repair_attempts: int = 0,
1583
+ fallback_used: bool = False, judge_prompt: "str | None" = None,
1584
+ judge_raw: "str | None" = None, duration_sec: "float | None" = None,
1585
+ judge_calls: "list[dict[str, Any]] | None" = None,
1586
+ ) -> None:
1587
+ entry: dict[str, Any] = {
1588
+ "ts": self._ts(), "step": step, "action": action, "params": params or {},
1589
+ "rationale": rationale, "valid": valid, "repair_attempts": repair_attempts,
1590
+ "fallback_used": fallback_used,
1591
+ }
1592
+ if judge_prompt:
1593
+ entry["judge_prompt"] = judge_prompt
1594
+ if judge_raw:
1595
+ entry["judge_response"] = judge_raw
1596
+ if judge_calls:
1597
+ entry["model_calls"] = json.loads(json.dumps(judge_calls, default=str))
1598
+ if duration_sec is not None:
1599
+ entry["duration_sec"] = round(duration_sec, 3)
1600
+ self._append_run("agent_decisions", entry)
1601
+
1602
+ decision_span = self.tracer.start_span(
1603
+ name=f"Agent Decision: step {step}", stage="AGENT_DECISION",
1604
+ input_data={"action": action, "params": params or {}},
1605
+ metadata={"valid": valid, "repair_attempts": repair_attempts, "fallback_used": fallback_used},
1606
+ )
1607
+ if judge_prompt or judge_raw:
1608
+ self.tracer.log_generation(
1609
+ name="Agent Decision Judge", model="judge",
1610
+ prompt=judge_prompt or "", completion=judge_raw or "", span_id=decision_span,
1611
+ )
1612
+ self.tracer.end_span(decision_span, output_data={"action": action, "rationale": rationale})
1613
+
1614
+ def log_agent_tool(
1615
+ self, step: int, *, tool: str, ok: bool, summary: str = "",
1616
+ error: "str | None" = None, duration_sec: "float | None" = None,
1617
+ ) -> None:
1618
+ entry: dict[str, Any] = {
1619
+ "ts": self._ts(), "step": step, "tool": tool, "ok": ok, "summary": summary,
1620
+ }
1621
+ if error is not None:
1622
+ entry["error"] = error
1623
+ if duration_sec is not None:
1624
+ entry["duration_sec"] = round(duration_sec, 3)
1625
+ self._append_run("agent_tool_calls", entry)
1626
+
1627
+ def log_stage_skipped(self, stage: str, reason_code: str, *, cycle: int = -1, detail: str = "") -> None:
1628
+ entry = {"ts": self._ts(), "stage": stage, "cycle": cycle, "reason_code": reason_code, "detail": detail}
1629
+ self._append_stage(stage, "stage_skipped", entry)
1630
+ span = self.tracer.start_span(
1631
+ name=f"{stage}: skipped", stage=stage,
1632
+ input_data={"reason_code": reason_code}, metadata={"detail": detail},
1633
+ )
1634
+ self.tracer.end_span(span, output_data={"reason_code": reason_code}, status="skipped")
1635
+
1636
+ def log_loop_end(
1637
+ self, report: "AutoDiagnoseReport", *,
1638
+ tokens_used: "int | None" = None, timings: "dict[str, float] | None" = None,
1639
+ ) -> None:
1640
+ entry: dict[str, Any] = {"ts": self._ts(), "cycles": report.cycles}
1641
+ if tokens_used is not None:
1642
+ entry["tokens_used"] = tokens_used
1643
+ if timings:
1644
+ entry["timings_sec"] = {k: round(v, 3) for k, v in timings.items()}
1645
+ entry["total_duration_sec"] = round(sum(timings.values()), 3)
1646
+ if hasattr(report, "resolved"):
1647
+ entry["resolved"] = report.resolved
1648
+ hyps = getattr(report, "final_hypotheses", [])
1649
+ entry["n_hypotheses"] = len(hyps)
1650
+ entry["final_hypotheses"] = [
1651
+ {"statement": h.statement, "plain_statement": h.plain_statement,
1652
+ "failure_mode": h.predicted_failure_mode, "status": h.status.value if h.status else None}
1653
+ for h in hyps
1654
+ ]
1655
+ if hasattr(report, "stopped_by"):
1656
+ entry["stopped_by"] = report.stopped_by
1657
+ all_hyps = getattr(report, "all_hypotheses", [])
1658
+ verified = getattr(report, "verified_hypotheses", [])
1659
+ entry["n_hypotheses"] = len(all_hyps)
1660
+ entry["n_verified"] = len(verified)
1661
+ entry["verified_hypotheses"] = [
1662
+ {"statement": tr.hypothesis.statement,
1663
+ "failure_mode": tr.hypothesis.predicted_failure_mode,
1664
+ "status": tr.status.value,
1665
+ "confidence": tr.confidence,
1666
+ "protocol_consistent": tr.is_consistent_with_protocol,
1667
+ "verdict": getattr(tr, "verdict", None)}
1668
+ for tr in verified
1669
+ ]
1670
+ self._append_run("loop_end", entry)
1671
+
1672
+ # ------------------------------------------------------------------
1673
+ # Tool synthesis (M1/M2 probes+stats tools, generated on demand)
1674
+ # ------------------------------------------------------------------
1675
+
1676
+ def log_tool_codegen(
1677
+ self, *, module: str, name: str, need: str, source: str, ok: bool,
1678
+ code: str = "", prompt: str = "", raw_output: str = "", raw_stream: str = "",
1679
+ error: str = "", stdout: str = "", cycle: "int | None" = None,
1680
+ extra: "dict[str, Any] | None" = None,
1681
+ ) -> None:
1682
+ cyc = self.current_cycle if cycle is None else cycle
1683
+ entry: dict[str, Any] = {
1684
+ "ts": self._ts(), "cycle": cyc, "module": module, "tool_name": name,
1685
+ "need": need, "source": source, "ok": ok, "error": error or None,
1686
+ "code": code or None, "prompt": prompt or None,
1687
+ "raw_output": raw_output or None, "raw_stream": raw_stream or None,
1688
+ "stdout": stdout or None,
1689
+ }
1690
+ if extra:
1691
+ entry.update(extra)
1692
+ self._append_stage(module, "tool_codegen", entry)
1693
+ if prompt or raw_stream or raw_output:
1694
+ self.log_model_exchange(
1695
+ _resolve_stage(module) or "M5", role="tool_codegen",
1696
+ operation=source, inputs=prompt or "",
1697
+ output=raw_stream or raw_output or code,
1698
+ error=error or None, cycle=cyc,
1699
+ metadata={"tool_name": name, "ok": ok},
1700
+ )
1701
+
1702
+ cg_span = self.tracer.start_span(
1703
+ name=f"Tool Codegen: {module}/{name}", stage=f"CODEGEN_{module}",
1704
+ input_data={"need": need, "source": source},
1705
+ metadata={"ok": ok, "error": error or None},
1706
+ )
1707
+ if prompt or raw_output:
1708
+ self.tracer.log_generation(
1709
+ name=f"Tool Synthesis: {name}", model=source,
1710
+ prompt=prompt or "", completion=raw_stream or raw_output or code, span_id=cg_span,
1711
+ )
1712
+ self.tracer.end_span(cg_span, output_data={"ok": ok, "code_chars": len(code or "")})
1713
+
1714
+ def log_tool_registry(self, cycle: int, module: str, generated: "list[Any]") -> None:
1715
+ if not generated:
1716
+ return
1717
+ tools = [
1718
+ {
1719
+ "name": getattr(g, "name", "tool"), "need": getattr(g, "need", ""),
1720
+ "source": getattr(g, "source", ""), "code": getattr(g, "code", "") or None,
1721
+ }
1722
+ for g in generated
1723
+ ]
1724
+ self._append_stage(module, "tool_registry", {
1725
+ "ts": self._ts(), "cycle": cycle, "module": module, "n_tools": len(tools), "tools": tools,
1726
+ })
1727
+
1728
+ # ------------------------------------------------------------------
1729
+ # Lifecycle
1730
+ # ------------------------------------------------------------------
1731
+
1732
+ def close(self) -> None:
1733
+ """Flush every doc one last time and persist the Langfuse trace bundle."""
1734
+ if self._closed:
1735
+ return
1736
+ with self._lock:
1737
+ self._flush_run()
1738
+ for stage in _STAGES:
1739
+ self._flush_stage(stage)
1740
+ self.tracer.end_trace({
1741
+ "spans": len(self.tracer.spans),
1742
+ "generations": len(self.tracer.generations),
1743
+ "scores": len(self.tracer.scores),
1744
+ })
1745
+ self.tracer.flush()
1746
+ try:
1747
+ self.tracer.export_bundle(self.run_dir / "langfuse_trace.json")
1748
+ except Exception: # noqa: BLE001
1749
+ pass
1750
+ # Offline runs need no retry queue once the complete JSON trace bundle
1751
+ # has been exported. A live run keeps a non-empty queue for retry.
1752
+ if self.tracer.mode == "offline" or self.tracer.outbox.pending_count() == 0:
1753
+ try:
1754
+ self._outbox_path.unlink(missing_ok=True)
1755
+ self._outbox_path.parent.rmdir()
1756
+ except OSError:
1757
+ pass
1758
+ self._closed = True
1759
+
1760
+ def __enter__(self) -> "RunLoggerV2":
1761
+ return self
1762
+
1763
+ def __exit__(self, *exc_info: Any) -> None:
1764
+ self.close()