evalrx 0.1.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (339) hide show
  1. evalrx/__init__.py +139 -0
  2. evalrx/agent_assets/__init__.py +2 -0
  3. evalrx/agent_assets/skills/README.md +28 -0
  4. evalrx/agent_assets/skills/eval-chart-style/SKILL.md +172 -0
  5. evalrx/agent_assets/skills/evalrx-report-ui/SKILL.md +116 -0
  6. evalrx/agent_assets/skills/nature-figure/LICENSE +201 -0
  7. evalrx/agent_assets/skills/nature-figure/README.md +412 -0
  8. evalrx/agent_assets/skills/nature-figure/SKILL.md +60 -0
  9. evalrx/agent_assets/skills/nature-figure/manifest.yaml +59 -0
  10. evalrx/agent_assets/skills/nature-figure/references/api.md +436 -0
  11. evalrx/agent_assets/skills/nature-figure/references/backend-selection.md +100 -0
  12. evalrx/agent_assets/skills/nature-figure/references/chart-types.md +281 -0
  13. evalrx/agent_assets/skills/nature-figure/references/common-patterns.md +350 -0
  14. evalrx/agent_assets/skills/nature-figure/references/demos.md +65 -0
  15. evalrx/agent_assets/skills/nature-figure/references/design-theory.md +439 -0
  16. evalrx/agent_assets/skills/nature-figure/references/figure-contract.md +93 -0
  17. evalrx/agent_assets/skills/nature-figure/references/figure-legend-conventions.md +71 -0
  18. evalrx/agent_assets/skills/nature-figure/references/nature-2026-observations.md +112 -0
  19. evalrx/agent_assets/skills/nature-figure/references/qa-contract.md +119 -0
  20. evalrx/agent_assets/skills/nature-figure/references/r-template-index.md +66 -0
  21. evalrx/agent_assets/skills/nature-figure/references/r-workflow.md +161 -0
  22. evalrx/agent_assets/skills/nature-figure/references/tutorials.md +251 -0
  23. evalrx/agent_assets/skills/nature-figure/static/core/contract.md +29 -0
  24. evalrx/agent_assets/skills/nature-figure/static/core/stance.md +37 -0
  25. evalrx/agent_assets/skills/nature-figure/static/fragments/backend/python.md +37 -0
  26. evalrx/agent_assets/skills/nature-figure/static/fragments/backend/r.md +44 -0
  27. evalrx/agent_assets/skills/outcome-driver-analysis/SKILL.md +213 -0
  28. evalrx/agent_assets/skills/outcome-driver-analysis/assets/analysis_report_template.md +53 -0
  29. evalrx/agent_assets/skills/outcome-driver-analysis/references/model_selection.md +72 -0
  30. evalrx/agent_assets/skills/outcome-driver-analysis/scripts/explanatory_var_eda.R +130 -0
  31. evalrx/agent_assets/skills/outcome-driver-analysis/scripts/explanatory_var_eda.py +150 -0
  32. evalrx/agent_assets/skills/outcome-driver-analysis/scripts/fit_outcome_model.R +181 -0
  33. evalrx/agent_assets/skills/outcome-driver-analysis/scripts/fit_outcome_model.py +186 -0
  34. evalrx/agent_assets/skills/outcome-driver-analysis/scripts/univariate_eda.R +149 -0
  35. evalrx/agent_assets/skills/outcome-driver-analysis/scripts/univariate_eda.py +177 -0
  36. evalrx/agent_assets/skills.py +27 -0
  37. evalrx/agent_runtime/__init__.py +78 -0
  38. evalrx/agent_runtime/_docker_runner.py +89 -0
  39. evalrx/agent_runtime/cli_runtime.py +103 -0
  40. evalrx/agent_runtime/cli_transcript.py +138 -0
  41. evalrx/agent_runtime/cli_types.py +68 -0
  42. evalrx/agent_runtime/codegen/__init__.py +5 -0
  43. evalrx/agent_runtime/codegen/runner.py +94 -0
  44. evalrx/agent_runtime/experiment_harness.py +117 -0
  45. evalrx/agent_runtime/factory.py +102 -0
  46. evalrx/agent_runtime/json_shape.py +44 -0
  47. evalrx/agent_runtime/judges/__init__.py +28 -0
  48. evalrx/agent_runtime/judges/agy.py +179 -0
  49. evalrx/agent_runtime/judges/autodetect.py +135 -0
  50. evalrx/agent_runtime/judges/claude.py +159 -0
  51. evalrx/agent_runtime/judges/codex.py +120 -0
  52. evalrx/agent_runtime/providers/__init__.py +21 -0
  53. evalrx/agent_runtime/providers/antigravity.py +31 -0
  54. evalrx/agent_runtime/providers/base.py +145 -0
  55. evalrx/agent_runtime/providers/claude_code.py +49 -0
  56. evalrx/agent_runtime/providers/codex.py +37 -0
  57. evalrx/agent_runtime/providers/gemini_cli.py +26 -0
  58. evalrx/agent_runtime/providers/kimi_cli.py +27 -0
  59. evalrx/agent_runtime/providers/opencode.py +27 -0
  60. evalrx/agent_runtime/providers/registry.py +58 -0
  61. evalrx/agent_runtime/sandbox.py +517 -0
  62. evalrx/agent_runtime/skill_audit.py +143 -0
  63. evalrx/agent_runtime/skills/__init__.py +19 -0
  64. evalrx/agent_runtime/skills/installer.py +68 -0
  65. evalrx/agent_runtime/skills/prompt_policy.py +86 -0
  66. evalrx/agent_runtime/skills/resolver.py +19 -0
  67. evalrx/analysis/__init__.py +132 -0
  68. evalrx/analysis/adjudicate.py +154 -0
  69. evalrx/analysis/analysis_module.py +361 -0
  70. evalrx/analysis/api.py +171 -0
  71. evalrx/analysis/case_studio.py +651 -0
  72. evalrx/analysis/cli.py +114 -0
  73. evalrx/analysis/dashboard.py +350 -0
  74. evalrx/analysis/eval_case_matrix.py +118 -0
  75. evalrx/analysis/eval_viz_theme.py +833 -0
  76. evalrx/analysis/explore_run.py +333 -0
  77. evalrx/analysis/explorer.py +1276 -0
  78. evalrx/analysis/failure_modes.py +607 -0
  79. evalrx/analysis/fused_pipeline.py +489 -0
  80. evalrx/analysis/holdout.py +300 -0
  81. evalrx/analysis/hypothesis_agent.py +230 -0
  82. evalrx/analysis/narration.py +177 -0
  83. evalrx/analysis/operationalize.py +442 -0
  84. evalrx/analysis/plain_language.py +42 -0
  85. evalrx/analysis/planner.py +283 -0
  86. evalrx/analysis/probe_search.py +203 -0
  87. evalrx/analysis/profile.py +268 -0
  88. evalrx/analysis/prompts/__init__.py +0 -0
  89. evalrx/analysis/prompts/explorer.py +417 -0
  90. evalrx/analysis/prompts/failure_modes.py +33 -0
  91. evalrx/analysis/prompts/holdout.py +27 -0
  92. evalrx/analysis/prompts/hypothesis_agent.py +78 -0
  93. evalrx/analysis/prompts/run_codebase.py +47 -0
  94. evalrx/analysis/prompts/stats_agent.py +72 -0
  95. evalrx/analysis/prompts/stats_tool_generator.py +43 -0
  96. evalrx/analysis/result_marker.py +47 -0
  97. evalrx/analysis/run_codebase.py +242 -0
  98. evalrx/analysis/run_view.py +205 -0
  99. evalrx/analysis/stage_views.py +93 -0
  100. evalrx/analysis/stats_agent.py +944 -0
  101. evalrx/analysis/stats_tool_agent.py +261 -0
  102. evalrx/analysis/stats_tool_generator.py +415 -0
  103. evalrx/analysis/stats_tools.py +1153 -0
  104. evalrx/analysis/trajectory_records.py +193 -0
  105. evalrx/analysis/workbench.py +431 -0
  106. evalrx/analyzers/__init__.py +42 -0
  107. evalrx/analyzers/agent/__init__.py +25 -0
  108. evalrx/analyzers/agent/counterfactual.py +84 -0
  109. evalrx/analyzers/agent/first_error_judge.py +96 -0
  110. evalrx/analyzers/agent/ignored_obs.py +81 -0
  111. evalrx/analyzers/agent/loop_detect.py +79 -0
  112. evalrx/analyzers/agent/reliability.py +165 -0
  113. evalrx/analyzers/agent/tool_shap.py +225 -0
  114. evalrx/analyzers/agent/trajectory_rubric.py +168 -0
  115. evalrx/analyzers/attention/__init__.py +19 -0
  116. evalrx/analyzers/attention/relative_attn.py +610 -0
  117. evalrx/analyzers/attention/rollout.py +73 -0
  118. evalrx/analyzers/attention/sink.py +56 -0
  119. evalrx/analyzers/attention/summary.py +190 -0
  120. evalrx/analyzers/attribution/__init__.py +6 -0
  121. evalrx/analyzers/attribution/generic_attn.py +31 -0
  122. evalrx/analyzers/attribution/gradcam.py +30 -0
  123. evalrx/analyzers/base.py +12 -0
  124. evalrx/analyzers/geometry/__init__.py +6 -0
  125. evalrx/analyzers/geometry/cka.py +70 -0
  126. evalrx/analyzers/geometry/linear_probe.py +157 -0
  127. evalrx/analyzers/hallucination/__init__.py +9 -0
  128. evalrx/analyzers/hallucination/chair.py +78 -0
  129. evalrx/analyzers/hallucination/opera.py +29 -0
  130. evalrx/analyzers/hallucination/pope.py +119 -0
  131. evalrx/analyzers/hallucination/selfcheck.py +155 -0
  132. evalrx/analyzers/hallucination/vcd.py +29 -0
  133. evalrx/analyzers/lens/__init__.py +7 -0
  134. evalrx/analyzers/lens/layer_contrast.py +133 -0
  135. evalrx/analyzers/lens/logit_lens.py +138 -0
  136. evalrx/analyzers/lens/tuned_lens.py +30 -0
  137. evalrx/analyzers/patching/__init__.py +5 -0
  138. evalrx/analyzers/patching/causal_trace.py +30 -0
  139. evalrx/analyzers/perturbation/__init__.py +23 -0
  140. evalrx/analyzers/perturbation/_shapley.py +54 -0
  141. evalrx/analyzers/perturbation/context_shap.py +174 -0
  142. evalrx/analyzers/perturbation/cot_faithfulness.py +239 -0
  143. evalrx/analyzers/perturbation/format_sensitivity.py +237 -0
  144. evalrx/analyzers/perturbation/mm_shap.py +146 -0
  145. evalrx/analyzers/perturbation/modality_ablation.py +196 -0
  146. evalrx/analyzers/perturbation/perturbation_battery.py +274 -0
  147. evalrx/analyzers/perturbation/prompt_contrast.py +265 -0
  148. evalrx/analyzers/perturbation/rise.py +94 -0
  149. evalrx/analyzers/perturbation/vl_shap.py +102 -0
  150. evalrx/analyzers/reasoning/__init__.py +33 -0
  151. evalrx/analyzers/reasoning/_text.py +328 -0
  152. evalrx/analyzers/reasoning/answer_extraction_audit.py +327 -0
  153. evalrx/analyzers/reasoning/arith_audit.py +226 -0
  154. evalrx/analyzers/reasoning/contamination.py +214 -0
  155. evalrx/analyzers/reasoning/knowledge_split.py +253 -0
  156. evalrx/analyzers/reasoning/self_repair.py +246 -0
  157. evalrx/analyzers/reasoning/step_rollout_value.py +216 -0
  158. evalrx/analyzers/reasoning/termination_audit.py +258 -0
  159. evalrx/analyzers/uncertainty/__init__.py +18 -0
  160. evalrx/analyzers/uncertainty/calibration.py +174 -0
  161. evalrx/analyzers/uncertainty/coverage_gap.py +199 -0
  162. evalrx/analyzers/uncertainty/entropy.py +90 -0
  163. evalrx/analyzers/uncertainty/logprob_entropy.py +69 -0
  164. evalrx/analyzers/uncertainty/self_consistency.py +204 -0
  165. evalrx/analyzers/uncertainty/verbalized_conf.py +64 -0
  166. evalrx/cli.py +411 -0
  167. evalrx/config.py +77 -0
  168. evalrx/contract/__init__.py +179 -0
  169. evalrx/contract/common.py +452 -0
  170. evalrx/contract/emit.py +948 -0
  171. evalrx/contract/export.py +237 -0
  172. evalrx/contract/m1.py +325 -0
  173. evalrx/contract/m2.py +317 -0
  174. evalrx/contract/m3.py +165 -0
  175. evalrx/contract/m4.py +130 -0
  176. evalrx/contract/m5.py +292 -0
  177. evalrx/contract/methodology.py +76 -0
  178. evalrx/contract/pre_m1.py +58 -0
  179. evalrx/contract/typescript.py +140 -0
  180. evalrx/core/__init__.py +85 -0
  181. evalrx/core/analyzer.py +174 -0
  182. evalrx/core/capability.py +54 -0
  183. evalrx/core/case.py +443 -0
  184. evalrx/core/experiment.py +106 -0
  185. evalrx/core/model.py +198 -0
  186. evalrx/core/pipeline.py +42 -0
  187. evalrx/core/registry.py +142 -0
  188. evalrx/core/result.py +64 -0
  189. evalrx/core/spec.py +173 -0
  190. evalrx/core/tokentype.py +165 -0
  191. evalrx/core/tool.py +92 -0
  192. evalrx/datasets/__init__.py +41 -0
  193. evalrx/datasets/base.py +68 -0
  194. evalrx/datasets/gui_os.py +52 -0
  195. evalrx/datasets/llm_qa.py +57 -0
  196. evalrx/datasets/pure_qa.py +12 -0
  197. evalrx/datasets/vlm_qa.py +695 -0
  198. evalrx/datasets/web_search_qa.py +52 -0
  199. evalrx/eval_agent/__init__.py +341 -0
  200. evalrx/eval_agent/_tools.py +81 -0
  201. evalrx/eval_agent/ab_runner.py +50 -0
  202. evalrx/eval_agent/agentic/__init__.py +43 -0
  203. evalrx/eval_agent/agentic/actions.py +216 -0
  204. evalrx/eval_agent/agentic/board.py +107 -0
  205. evalrx/eval_agent/agentic/loop.py +190 -0
  206. evalrx/eval_agent/agentic/tools.py +538 -0
  207. evalrx/eval_agent/checkpoint.py +57 -0
  208. evalrx/eval_agent/cli_agent.py +59 -0
  209. evalrx/eval_agent/cli_skills.py +5 -0
  210. evalrx/eval_agent/evolution.py +396 -0
  211. evalrx/eval_agent/git_manager.py +215 -0
  212. evalrx/eval_agent/hypothesis.py +172 -0
  213. evalrx/eval_agent/label_quarantine.py +209 -0
  214. evalrx/eval_agent/legacy.py +530 -0
  215. evalrx/eval_agent/log_schema.py +497 -0
  216. evalrx/eval_agent/loop.py +2159 -0
  217. evalrx/eval_agent/loop_reports.py +116 -0
  218. evalrx/eval_agent/model_instrumentation.py +282 -0
  219. evalrx/eval_agent/narration.py +193 -0
  220. evalrx/eval_agent/nl_runner.py +460 -0
  221. evalrx/eval_agent/orchestrator.py +61 -0
  222. evalrx/eval_agent/preregister.py +93 -0
  223. evalrx/eval_agent/prompts/__init__.py +1 -0
  224. evalrx/eval_agent/prompts/agentic.py +46 -0
  225. evalrx/eval_agent/prompts/case_discovery.py +25 -0
  226. evalrx/eval_agent/prompts/diagnosis.py +125 -0
  227. evalrx/eval_agent/prompts/experiment_writer.py +265 -0
  228. evalrx/eval_agent/prompts/explore_step.py +37 -0
  229. evalrx/eval_agent/prompts/fix_agent.py +257 -0
  230. evalrx/eval_agent/prompts/hypothesis_tester.py +15 -0
  231. evalrx/eval_agent/prompts/nl_runner.py +38 -0
  232. evalrx/eval_agent/prompts/probe_agent.py +25 -0
  233. evalrx/eval_agent/prompts/probe_candidate_generator.py +14 -0
  234. evalrx/eval_agent/prompts/probe_generator.py +35 -0
  235. evalrx/eval_agent/prompts/whitebox_probe_generator.py +38 -0
  236. evalrx/eval_agent/report.py +58 -0
  237. evalrx/eval_agent/run_context.py +354 -0
  238. evalrx/eval_agent/run_log.schema.json +1215 -0
  239. evalrx/eval_agent/run_logger_v2.py +1764 -0
  240. evalrx/eval_agent/run_metadata.py +208 -0
  241. evalrx/eval_agent/stages/__init__.py +56 -0
  242. evalrx/eval_agent/stages/case_discovery.py +293 -0
  243. evalrx/eval_agent/stages/diagnosis.py +1017 -0
  244. evalrx/eval_agent/stages/experiment_writer.py +1634 -0
  245. evalrx/eval_agent/stages/fix_agent.py +3916 -0
  246. evalrx/eval_agent/stages/fix_internals.py +499 -0
  247. evalrx/eval_agent/stages/fix_pipeline.py +725 -0
  248. evalrx/eval_agent/stages/fix_tiers.py +187 -0
  249. evalrx/eval_agent/stages/fix_tools.py +1034 -0
  250. evalrx/eval_agent/stages/hypothesis_tester.py +1014 -0
  251. evalrx/eval_agent/stages/probe.py +439 -0
  252. evalrx/eval_agent/stages/probe_agent.py +1079 -0
  253. evalrx/eval_agent/stages/probe_candidate_generator.py +128 -0
  254. evalrx/eval_agent/stages/probe_generator.py +326 -0
  255. evalrx/eval_agent/stages/probe_search_agent.py +106 -0
  256. evalrx/eval_agent/stages/protocol.py +112 -0
  257. evalrx/eval_agent/stages/repair_catalog.py +273 -0
  258. evalrx/eval_agent/stages/surgery.py +524 -0
  259. evalrx/eval_agent/stages/whitebox_probe_generator.py +351 -0
  260. evalrx/eval_agent/store.py +231 -0
  261. evalrx/logging_utils.py +112 -0
  262. evalrx/models/__init__.py +161 -0
  263. evalrx/models/_discover.py +101 -0
  264. evalrx/models/agent.py +380 -0
  265. evalrx/models/backends/__init__.py +58 -0
  266. evalrx/models/backends/api.py +169 -0
  267. evalrx/models/backends/base.py +57 -0
  268. evalrx/models/backends/gemini_compat.py +579 -0
  269. evalrx/models/backends/hf_local.py +2074 -0
  270. evalrx/models/backends/openai_compat.py +301 -0
  271. evalrx/models/backends/vllm_offline.py +116 -0
  272. evalrx/models/base.py +24 -0
  273. evalrx/models/blackbox/__init__.py +4 -0
  274. evalrx/models/blackbox/agent.py +31 -0
  275. evalrx/models/blackbox/base.py +29 -0
  276. evalrx/models/blackbox/gemini.py +279 -0
  277. evalrx/models/blackbox/llm_api.py +17 -0
  278. evalrx/models/blackbox/vlm_api.py +17 -0
  279. evalrx/models/compose.py +66 -0
  280. evalrx/models/inference.py +88 -0
  281. evalrx/models/paper_methods/__init__.py +8 -0
  282. evalrx/models/paper_methods/aad.py +53 -0
  283. evalrx/models/paper_methods/ifcd.py +204 -0
  284. evalrx/models/paper_methods/pai.py +164 -0
  285. evalrx/models/paper_methods/tcd.py +202 -0
  286. evalrx/models/paper_methods/vcd.py +45 -0
  287. evalrx/models/paper_methods/vicrop.py +137 -0
  288. evalrx/models/toolcodec.py +143 -0
  289. evalrx/models/tools/__init__.py +20 -0
  290. evalrx/models/tools/perception.py +300 -0
  291. evalrx/models/tools/visual.py +174 -0
  292. evalrx/models/whitebox/__init__.py +26 -0
  293. evalrx/models/whitebox/agent.py +31 -0
  294. evalrx/models/whitebox/base.py +24 -0
  295. evalrx/models/whitebox/qwen.py +61 -0
  296. evalrx/models/whitebox/qwen2_5_omni.py +29 -0
  297. evalrx/models/whitebox/qwen2_audio.py +25 -0
  298. evalrx/models/whitebox/qwen_omni.py +53 -0
  299. evalrx/models/whitebox/qwen_vl.py +62 -0
  300. evalrx/observability/__init__.py +21 -0
  301. evalrx/observability/envelope.py +122 -0
  302. evalrx/observability/outbox.py +111 -0
  303. evalrx/observability/tracer.py +882 -0
  304. evalrx/reporting/__init__.py +28 -0
  305. evalrx/reporting/case_study.py +947 -0
  306. evalrx/reporting/compiler.py +587 -0
  307. evalrx/reporting/dynamic.py +1882 -0
  308. evalrx/reporting/html_report.py +2225 -0
  309. evalrx/reporting/langfuse_exporter.py +38 -0
  310. evalrx/reporting/langfuse_source.py +155 -0
  311. evalrx/reporting/model.py +151 -0
  312. evalrx/reporting/run_events.py +184 -0
  313. evalrx/reporting/server.py +557 -0
  314. evalrx/reporting/stages.py +58 -0
  315. evalrx/reporting/static_export.py +142 -0
  316. evalrx/reporting/web_dist/index.html +146 -0
  317. evalrx/specs.py +727 -0
  318. evalrx/stats/__init__.py +47 -0
  319. evalrx/stats/api.py +192 -0
  320. evalrx/stats/bootstrap.py +86 -0
  321. evalrx/stats/ebh.py +27 -0
  322. evalrx/stats/evalue.py +98 -0
  323. evalrx/stats/friedman.py +138 -0
  324. evalrx/stats/mcnemar.py +40 -0
  325. evalrx/stats/multiplicity.py +159 -0
  326. evalrx/stats/subset_sampling.py +55 -0
  327. evalrx/term_links.py +43 -0
  328. evalrx/viz/__init__.py +7 -0
  329. evalrx/viz/labels.py +77 -0
  330. evalrx/viz/prompts.py +39 -0
  331. evalrx/viz/renderer.py +590 -0
  332. evalrx/viz/schema.py +36 -0
  333. evalrx/viz/style.py +134 -0
  334. evalrx-0.1.2.dist-info/METADATA +532 -0
  335. evalrx-0.1.2.dist-info/RECORD +339 -0
  336. evalrx-0.1.2.dist-info/WHEEL +5 -0
  337. evalrx-0.1.2.dist-info/entry_points.txt +3 -0
  338. evalrx-0.1.2.dist-info/licenses/LICENSE +121 -0
  339. evalrx-0.1.2.dist-info/top_level.txt +1 -0
@@ -0,0 +1,128 @@
1
+ """VLM probe candidate generator — ProbeLLM Macro/Micro generators (Sec 3.3),
2
+ scoped to VLM QA and plugged into :class:`~evalrx.analysis.probe_search.ProbeSearch`.
3
+
4
+ Scope (v1): both Macro and Micro produce a *paraphrase* of an existing seed's
5
+ question over the SAME image, never new imagery and never an altered
6
+ semantic target — so the seed's ``expected`` answer stays valid for the new
7
+ candidate without needing a vision-capable judge or an image/answer
8
+ generation tool (the paper's tool-augmented generation, which invokes web/code
9
+ tools to obtain or verify a *new* gold answer, is out of scope here — see
10
+ ``ProbeSearch``'s docstring: a richer generator with those tools can be
11
+ substituted without touching the search algorithm itself).
12
+
13
+ - **Macro** (broad coverage): picks the seed question least similar (by token
14
+ overlap) to what the macro tree has already explored, then paraphrases it —
15
+ diversifies which part of the fixed image pool gets visited next.
16
+ - **Micro** (local refinement): paraphrases the *current search node's own*
17
+ case — same image, same gold answer, different wording — probing surface
18
+ robustness (does the model's correctness flip on a reworded but
19
+ semantically identical question) rather than the paper's full
20
+ entity/attribute substitution.
21
+ """
22
+
23
+ from __future__ import annotations
24
+
25
+ import logging
26
+ import re
27
+ from dataclasses import dataclass
28
+ from typing import TYPE_CHECKING
29
+
30
+ from evalrx.core.case import CaseBatch, FailureCase, Inputs
31
+ from evalrx.eval_agent.prompts.probe_candidate_generator import PARAPHRASE_PROMPT
32
+
33
+ if TYPE_CHECKING:
34
+ from evalrx.analysis.probe_search import ProbeNode
35
+ from evalrx.core.model import Model
36
+
37
+ logger = logging.getLogger(__name__)
38
+
39
+
40
+ def _tokenize(text: str) -> set[str]:
41
+ return set(re.findall(r"[a-z0-9]+", text.lower()))
42
+
43
+
44
+ def _jaccard_distance(a: str, b: str) -> float:
45
+ """1.0 = no shared tokens (maximally distinct), 0.0 = identical token sets."""
46
+ ta, tb = _tokenize(a), _tokenize(b)
47
+ union = ta | tb
48
+ if not union:
49
+ return 0.0
50
+ return 1.0 - len(ta & tb) / len(union)
51
+
52
+
53
+ def _extract_question(raw: str) -> str:
54
+ text = raw.strip()
55
+ if text.startswith("```"):
56
+ text = re.sub(r"^```\w*\n?", "", text)
57
+ text = re.sub(r"\n?```\s*$", "", text)
58
+ text = re.sub(r"^(question|paraphrase)\s*:\s*", "", text, flags=re.IGNORECASE)
59
+ return text.strip().strip('"')
60
+
61
+
62
+ @dataclass
63
+ class VLMProbeCandidateGenerator:
64
+ """ProbeLLM-style Macro/Micro candidate generation over a fixed VLM seed pool.
65
+
66
+ Args:
67
+ seed_pool: Existing (image, question, expected) cases — e.g. loaded via
68
+ ``VLMQADataset`` — the only source of images/gold answers
69
+ (Scope v1: no new imagery, no new gold synthesis).
70
+ judge: Text-only judge used to paraphrase questions (only the
71
+ question text is sent — never the image, so no
72
+ vision-capable judge is required). Required for either
73
+ regime to produce candidates; ``available`` is False and
74
+ both ``macro``/``micro`` return ``None`` without one
75
+ (mirrors ``ProbeGenerator.available``).
76
+ max_repairs: Retries if the judge echoes the question unchanged.
77
+ """
78
+
79
+ seed_pool: CaseBatch
80
+ judge: "Model | None" = None
81
+ max_repairs: int = 1
82
+
83
+ @property
84
+ def available(self) -> bool:
85
+ return self.judge is not None and len(self.seed_pool) > 0
86
+
87
+ def macro(self, node: "ProbeNode", explored: "list[ProbeNode]") -> "FailureCase | None":
88
+ """Diversify: paraphrase the pool seed least similar to what the macro
89
+ tree has already visited (paper Eq.11's "under-represented" frontier,
90
+ approximated here by question-token overlap rather than embeddings)."""
91
+ if not self.available:
92
+ return None
93
+ explored_prompts = [n.case.inputs.prompt for n in explored]
94
+ seed = max(
95
+ self.seed_pool,
96
+ key=lambda c: min(
97
+ (_jaccard_distance(c.inputs.prompt, p) for p in explored_prompts),
98
+ default=1.0,
99
+ ),
100
+ )
101
+ return self._paraphrase(seed, style="a very differently worded question")
102
+
103
+ def micro(self, node: "ProbeNode") -> "FailureCase | None":
104
+ """Refine: paraphrase the search node's own case for local
105
+ surface-robustness probing (same image, same gold answer)."""
106
+ if not self.available:
107
+ return None
108
+ return self._paraphrase(
109
+ node.case, style="a lightly reworded variant (synonyms/reordering)"
110
+ )
111
+
112
+ def _paraphrase(self, base: FailureCase, *, style: str) -> "FailureCase | None":
113
+ prompt = PARAPHRASE_PROMPT.format(question=base.inputs.prompt, style=style)
114
+ for _attempt in range(self.max_repairs + 1):
115
+ try:
116
+ raw = self.judge.generate(prompt) # type: ignore[union-attr]
117
+ except Exception as exc: # noqa: BLE001 — generation is best-effort
118
+ logger.warning("VLMProbeCandidateGenerator: judge.generate failed: %s", exc)
119
+ return None
120
+ question = _extract_question(str(raw))
121
+ if question and question.strip().lower() != base.inputs.prompt.strip().lower():
122
+ return FailureCase(
123
+ inputs=Inputs(prompt=question, image=base.inputs.image),
124
+ expected=base.expected,
125
+ tags={"probe_search_candidate"},
126
+ metadata={"seed_case_id": base.id},
127
+ )
128
+ return None
@@ -0,0 +1,326 @@
1
+ """M1 tier (b) — ProbeGenerator: synthesise a new black-box probe on demand.
2
+
3
+ When no registered analyzer in the catalog targets the observed failure, this
4
+ generator creates a bespoke *probe* — but adapted to M1's reality: a probe needs
5
+ the model, which (unlike M2's data-only stats tools) cannot be shipped into a
6
+ subprocess. So the split is:
7
+
8
+ 1. **Host collects** the model's outputs on the cases (the host owns the loaded
9
+ model) into ``m1_probe_input.json``.
10
+ 2. A **sandboxed generated script** reads that JSON and computes a per-case probe
11
+ metric over the *outputs* (refusal detection, language drift, format/printf
12
+ adherence, length, keyword presence, answer-extraction failure, …), printing a
13
+ strict ``PROBE_RESULT_JSON=`` line.
14
+ 3. The host wraps the parsed findings into a
15
+ :class:`~evalrx.core.result.Result` whose ``per_case`` entries flow into
16
+ M2 (stats tools) → M4 exactly like any catalog analyzer's output.
17
+
18
+ This keeps the M2-tier(b) safety model: generated code never touches the repo
19
+ source, never sees the weights, and runs in an
20
+ :class:`~evalrx.agent_runtime.sandbox.ExperimentSandbox` subprocess.
21
+
22
+ Scope (v1): probes that are **functions over a single forward-pass output**.
23
+ Multi-sample (self-consistency), perturbation, and white-box probes need the
24
+ model handle and are out of scope here — they belong to an in-process path.
25
+ """
26
+
27
+ from __future__ import annotations
28
+
29
+ import json
30
+ import logging
31
+ import re
32
+ from dataclasses import dataclass
33
+ from pathlib import Path
34
+ from typing import TYPE_CHECKING, Any
35
+
36
+ from evalrx.agent_runtime.sandbox import ExperimentSandbox
37
+ from evalrx.core.result import Result
38
+ from evalrx.eval_agent.prompts.probe_generator import (
39
+ _GENERATE_PROMPT,
40
+ _INPUT_FILENAME,
41
+ _MAX_OUTPUT_CHARS,
42
+ _RESULT_MARKER,
43
+ )
44
+
45
+ if TYPE_CHECKING:
46
+ from evalrx.agent_runtime.cli_types import CliAgentConfig
47
+ from evalrx.core.case import CaseBatch
48
+ from evalrx.core.model import Model
49
+
50
+ logger = logging.getLogger(__name__)
51
+
52
+
53
+
54
+ @dataclass
55
+ class GeneratedProbe:
56
+ """A validated, sandbox-executed probe that can be re-run on new cases.
57
+
58
+ Attributes:
59
+ name: Short identifier (becomes ``generated:<name>``).
60
+ code: The Python source that was written and executed.
61
+ need: The natural-language failure pattern it probes for.
62
+ source: Which backend wrote it (``"cli:<provider>"`` or ``"llm"``).
63
+ """
64
+
65
+ name: str
66
+ code: str
67
+ need: str = ""
68
+ source: str = ""
69
+
70
+
71
+ class ProbeGenerator:
72
+ """Generate, run, and cache bespoke black-box probes in a sandbox.
73
+
74
+ Args:
75
+ judge: LLM used for the single-pass code-writing path.
76
+ cli_config: CLI coding-agent config (``provider != "llm"``) used before
77
+ the judge when present.
78
+ sandbox: Execution sandbox (fresh temp-dir when ``None``).
79
+ timeout_sec: Hard wall-clock limit per sandbox run.
80
+ max_cases: Cap on cases whose outputs are collected (cost guard); 0 (the default) = every case.
81
+ """
82
+
83
+ def __init__(
84
+ self,
85
+ judge: "Model | None" = None,
86
+ cli_config: "CliAgentConfig | None" = None,
87
+ sandbox: "ExperimentSandbox | None" = None,
88
+ timeout_sec: int = 60,
89
+ max_cases: int = 0,
90
+ run_logger: "Any | None" = None,
91
+ ) -> None:
92
+ self._judge = judge
93
+ self._cli_config = cli_config
94
+ self._timeout_sec = timeout_sec
95
+ self._max_cases = max_cases
96
+ self._sandbox = sandbox or ExperimentSandbox()
97
+ # Optional RunLogger — when set, every code-writing attempt (the prompt,
98
+ # the code produced, the backend used, and the pass/fail outcome) is
99
+ # recorded as a "tool_codegen" event so tool synthesis is fully traceable.
100
+ self.run_logger = run_logger
101
+ self._last_prompt: str = ""
102
+ self._last_raw: str = ""
103
+ self._last_raw_stream: str = ""
104
+ self._last_usage: dict | None = None
105
+
106
+ @property
107
+ def available(self) -> bool:
108
+ return self._judge is not None or (
109
+ self._cli_config is not None and self._cli_config.provider != "llm"
110
+ )
111
+
112
+ # ------------------------------------------------------------------
113
+ # Public interface
114
+ # ------------------------------------------------------------------
115
+
116
+ def generate(
117
+ self,
118
+ need: str,
119
+ model: "Model",
120
+ cases: "CaseBatch",
121
+ name: str = "custom",
122
+ ) -> "tuple[Result | None, GeneratedProbe | None]":
123
+ """Collect outputs, write+run a probe over them, return (Result, probe)."""
124
+ if not self.available:
125
+ logger.debug("ProbeGenerator: no code-writing backend configured")
126
+ return None, None
127
+
128
+ self._collect_outputs(model, cases)
129
+ self._last_prompt = ""
130
+ self._last_raw = ""
131
+ self._last_raw_stream = ""
132
+ self._last_usage = None
133
+ try:
134
+ code, source = self._write_code(need)
135
+ except Exception as exc:
136
+ logger.warning("ProbeGenerator: code writing failed: %s", exc)
137
+ self._emit_codegen(name, need, "", "", ok=False, error=f"code writing failed: {exc}")
138
+ return None, None
139
+ if not code.strip():
140
+ self._emit_codegen(name, need, source, "", ok=False, error="empty code produced")
141
+ return None, None
142
+
143
+ result = self._run_code(code, name, model, cases)
144
+ self._emit_codegen(
145
+ name, need, source, code, ok=result is not None,
146
+ error="" if result is not None else "sandbox produced no parseable result",
147
+ )
148
+ if result is None:
149
+ return None, None
150
+ probe = GeneratedProbe(name=name, code=code, need=need, source=source)
151
+ return result, probe
152
+
153
+ def _emit_codegen(
154
+ self, name: str, need: str, source: str, code: str, *, ok: bool, error: str = ""
155
+ ) -> None:
156
+ """Record one code-writing attempt to the RunLogger, if attached."""
157
+ if self.run_logger is None:
158
+ return
159
+ extra = ({"cli_usage": self._last_usage}
160
+ if source.startswith("cli:") and self._last_usage else None)
161
+ try:
162
+ self.run_logger.log_tool_codegen(
163
+ module="m1_probe", name=name, need=need, source=source, ok=ok,
164
+ code=code, prompt=self._last_prompt, raw_output=self._last_raw,
165
+ raw_stream=self._last_raw_stream, error=error,
166
+ extra=extra,
167
+ )
168
+ except Exception as exc: # logging must never break generation
169
+ logger.debug("ProbeGenerator: log_tool_codegen failed: %s", exc)
170
+
171
+ def run_cached(
172
+ self,
173
+ probe: GeneratedProbe,
174
+ model: "Model",
175
+ cases: "CaseBatch",
176
+ ) -> "Result | None":
177
+ """Re-run an already-generated probe on fresh cases (no LLM call)."""
178
+ self._collect_outputs(model, cases)
179
+ return self._run_code(probe.code, probe.name, model, cases)
180
+
181
+ # ------------------------------------------------------------------
182
+ # Internals
183
+ # ------------------------------------------------------------------
184
+
185
+ def _collect_outputs(self, model: "Model", cases: "CaseBatch") -> None:
186
+ """Run the model on each case and serialise outputs to the sandbox dir."""
187
+ records: list[dict[str, Any]] = []
188
+ selected = list(cases)
189
+ if self._max_cases > 0: # 0 = every case
190
+ selected = selected[: self._max_cases]
191
+ if getattr(self.run_logger, "preserve_full_model_io", False):
192
+ from evalrx.eval_agent.model_instrumentation import InstrumentedModel
193
+
194
+ model = InstrumentedModel(
195
+ model, self.run_logger,
196
+ cycle=int(getattr(self.run_logger, "current_cycle", -1)),
197
+ analyzer="generated_probe_input_collection",
198
+ case_prompts={c.inputs.prompt: c.id for c in selected},
199
+ batch_case_ids=[c.id for c in selected],
200
+ )
201
+ for case in selected:
202
+ inp = getattr(case, "inputs", None)
203
+ try:
204
+ output = str(model.generate(inp)) if inp is not None else ""
205
+ except Exception as exc: # a probe over partial outputs is still useful
206
+ logger.debug("ProbeGenerator: generate failed for %s: %s", case.id, exc)
207
+ output = ""
208
+ label = getattr(case, "label", None)
209
+ records.append({
210
+ "id": case.id,
211
+ "prompt": str(getattr(inp, "prompt", "")) if inp is not None else "",
212
+ "expected": getattr(case, "expected", None),
213
+ "label": getattr(label, "value", None),
214
+ "output": output[:_MAX_OUTPUT_CHARS],
215
+ })
216
+ path = Path(self._sandbox.workdir) / _INPUT_FILENAME
217
+ path.write_text(json.dumps({"cases": records}, default=str), encoding="utf-8")
218
+
219
+ def _write_code(self, need: str) -> tuple[str, str]:
220
+ if self._cli_config is not None and self._cli_config.provider != "llm":
221
+ code = self._write_code_cli(need)
222
+ if code:
223
+ return code, f"cli:{self._cli_config.provider}"
224
+ prompt = self._build_prompt(need, fenced=True)
225
+ self._last_prompt = prompt
226
+ raw = self._judge.generate(prompt) # type: ignore[union-attr]
227
+ self._last_raw = str(raw)
228
+ return _extract_code(str(raw)), "llm"
229
+
230
+ def _write_code_cli(self, need: str) -> str:
231
+ from evalrx.agent_runtime.codegen import CodegenRunner
232
+
233
+ prompt = self._build_prompt(need, fenced=False)
234
+ self._last_prompt = prompt
235
+ result = CodegenRunner(self._cli_config).write_code( # type: ignore[arg-type]
236
+ prompt,
237
+ workdir=Path(self._sandbox.workdir),
238
+ timeout_sec=self._timeout_sec,
239
+ preferred_filenames=("probe.py",),
240
+ )
241
+ self._last_raw = result.raw_output
242
+ self._last_raw_stream = ""
243
+ if result.raw_stream_path:
244
+ try:
245
+ self._last_raw_stream = (
246
+ Path(self._sandbox.workdir) / result.raw_stream_path
247
+ ).read_text(encoding="utf-8")
248
+ except OSError:
249
+ pass
250
+ self._last_usage = result.usage
251
+ return result.code
252
+
253
+ def _build_prompt(self, need: str, *, fenced: bool) -> str:
254
+ return _GENERATE_PROMPT.format(
255
+ need=need.strip() or "Detect outputs that fail the task.",
256
+ input_filename=_INPUT_FILENAME,
257
+ marker=_RESULT_MARKER,
258
+ fences_hint=" inside a ```python code block" if fenced else
259
+ ", written to a file named probe.py",
260
+ )
261
+
262
+ def _run_code(
263
+ self,
264
+ code: str,
265
+ name: str,
266
+ model: "Model",
267
+ cases: "CaseBatch",
268
+ ) -> "Result | None":
269
+ sandbox_result = self._sandbox.run(code, timeout_sec=self._timeout_sec)
270
+ if not sandbox_result.ok:
271
+ logger.warning(
272
+ "ProbeGenerator: sandbox run failed (rc=%s): %s",
273
+ sandbox_result.returncode, (sandbox_result.stderr or "").strip()[:200],
274
+ )
275
+ return None
276
+ return _parse_result(sandbox_result.stdout, name, model, cases)
277
+
278
+
279
+ # ---------------------------------------------------------------------------
280
+ # Module helpers
281
+ # ---------------------------------------------------------------------------
282
+
283
+ def _extract_code(raw: str) -> str:
284
+ cleaned = re.sub(r"<think>.*?</think>", "", raw, flags=re.DOTALL)
285
+ fence = re.search(r"```(?:python)?\s*\n(.*?)```", cleaned, flags=re.DOTALL)
286
+ if fence:
287
+ return fence.group(1).strip()
288
+ return cleaned.strip()
289
+
290
+
291
+ def _parse_result(
292
+ stdout: str,
293
+ name: str,
294
+ model: "Model",
295
+ cases: "CaseBatch",
296
+ ) -> "Result | None":
297
+ """Parse the last ``PROBE_RESULT_JSON=`` line into a Result, or None."""
298
+ marker_line = None
299
+ for line in stdout.splitlines():
300
+ s = line.strip()
301
+ if s.startswith(_RESULT_MARKER):
302
+ marker_line = s[len(_RESULT_MARKER):]
303
+ if marker_line is None:
304
+ logger.warning("ProbeGenerator: no PROBE_RESULT_JSON line in probe output")
305
+ return None
306
+ try:
307
+ data = json.loads(marker_line)
308
+ except json.JSONDecodeError as exc:
309
+ logger.warning("ProbeGenerator: unparseable PROBE_RESULT_JSON: %s", exc)
310
+ return None
311
+
312
+ findings: dict[str, Any] = {}
313
+ raw_findings = data.get("findings")
314
+ if isinstance(raw_findings, dict):
315
+ findings.update(raw_findings)
316
+ per_case = data.get("per_case")
317
+ if isinstance(per_case, list):
318
+ findings["per_case"] = [e for e in per_case if isinstance(e, dict)]
319
+
320
+ return Result(
321
+ analyzer=f"generated:{name}",
322
+ model=repr(model),
323
+ cases=cases,
324
+ findings=findings,
325
+ metadata={"generated": True, "probe": name},
326
+ )
@@ -0,0 +1,106 @@
1
+ """ProbeSearchAgent — wires ProbeLLM's hierarchical MCTS
2
+ (:mod:`evalrx.analysis.probe_search`) to a real target model + judge.
3
+
4
+ ``analysis.probe_search`` stays standalone (a generic tree search over
5
+ injected callables); this eval_agent-layer module supplies those callables
6
+ from real components:
7
+
8
+ - verifier V -> :class:`~evalrx.eval_agent.stages.case_discovery.CaseDiscoveryAgent`
9
+ - Macro/Micro generators -> :class:`~evalrx.eval_agent.stages.probe_candidate_generator.VLMProbeCandidateGenerator`
10
+
11
+ The discovered failure cases (``ProbeSearchResult.failure_cases``) are plain
12
+ ``FailureCase`` objects and feed directly into
13
+ :func:`evalrx.analysis.failure_modes.cluster_failures` for failure-mode
14
+ synthesis, or into M1-M4 like any other labeled batch.
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ import logging
20
+ from dataclasses import dataclass
21
+ from typing import TYPE_CHECKING, Any
22
+
23
+ from evalrx.analysis.probe_search import ProbeSearch, ProbeSearchResult
24
+ from evalrx.eval_agent.stages.case_discovery import CaseDiscoveryAgent
25
+ from evalrx.eval_agent.stages.probe_candidate_generator import VLMProbeCandidateGenerator
26
+
27
+ if TYPE_CHECKING:
28
+ from evalrx.core.case import CaseBatch, FailureCase
29
+ from evalrx.core.model import Model
30
+ from evalrx.eval_agent.stages.protocol import ExperimentProtocol
31
+
32
+ logger = logging.getLogger(__name__)
33
+
34
+
35
+ @dataclass
36
+ class ProbeSearchAgent:
37
+ """Run a hierarchical Macro/Micro MCTS probe search against a target model.
38
+
39
+ Args:
40
+ judge: Text-only judge used both to score PASS/FAIL (via
41
+ ``CaseDiscoveryAgent``) and to paraphrase Macro/Micro
42
+ candidates (via ``VLMProbeCandidateGenerator``). Required —
43
+ without it, generation is unavailable and the search finds
44
+ nothing (``ProbeSearchResult.n_simulations == 0``).
45
+ protocol: Optional experiment protocol passed to the discovery judge
46
+ for scoring context.
47
+ budget: Total simulations (T_max in the paper's Eq.4).
48
+ beta: UCB exploration constant (Eq.7).
49
+ w_max: Max children per search-tree node before progressive
50
+ widening forces a deeper descent instead of a new sibling.
51
+ """
52
+
53
+ judge: "Model"
54
+ protocol: "ExperimentProtocol | None" = None
55
+ budget: int = 20
56
+ beta: float = 1.0
57
+ w_max: int = 3
58
+ run_logger: Any | None = None
59
+
60
+ def __post_init__(self) -> None:
61
+ if self.judge is None:
62
+ raise ValueError(
63
+ "ProbeSearchAgent requires a judge (e.g. ClaudeModel() or AgyModel()) "
64
+ "— without one, candidate generation is unavailable and the search "
65
+ "would silently discover nothing."
66
+ )
67
+
68
+ def run(self, model: "Model", seed_pool: "CaseBatch") -> ProbeSearchResult:
69
+ judge = self.judge
70
+ if getattr(self.run_logger, "preserve_full_model_io", False):
71
+ from evalrx.eval_agent.model_instrumentation import InstrumentedModel
72
+
73
+ cycle = int(getattr(self.run_logger, "current_cycle", -1))
74
+ ids = [case.id for case in seed_pool]
75
+ prompts = {case.inputs.prompt: case.id for case in seed_pool}
76
+ judge = InstrumentedModel(
77
+ judge, self.run_logger, cycle=cycle, analyzer="probe_search_judge",
78
+ batch_case_ids=ids,
79
+ )
80
+ model = InstrumentedModel(
81
+ model, self.run_logger, cycle=cycle, analyzer="probe_search_target",
82
+ case_prompts=prompts, batch_case_ids=ids,
83
+ )
84
+ discovery = CaseDiscoveryAgent(judge=judge)
85
+ generator = VLMProbeCandidateGenerator(seed_pool=seed_pool, judge=judge)
86
+
87
+ def verify(case: "FailureCase") -> "FailureCase":
88
+ report = discovery.discover(model, [case], protocol=self.protocol)
89
+ cases = list(report.cases)
90
+ return cases[0] if cases else case
91
+
92
+ search = ProbeSearch(
93
+ generate_macro=generator.macro,
94
+ generate_micro=generator.micro,
95
+ verify=verify,
96
+ seeds=seed_pool,
97
+ budget=self.budget,
98
+ beta=self.beta,
99
+ w_max=self.w_max,
100
+ )
101
+ result = search.run()
102
+ logger.info(
103
+ "ProbeSearchAgent: %d simulation(s) (macro=%d micro=%d), error_rate=%.2f",
104
+ result.n_simulations, result.n_macro, result.n_micro, result.error_rate,
105
+ )
106
+ return result
@@ -0,0 +1,112 @@
1
+ """ExperimentProtocol — user-supplied description of what an experiment tests.
2
+
3
+ The protocol is the human prior that anchors the self-evolving loop:
4
+
5
+ - **M1** passes the protocol to :class:`~evalrx.eval_agent.probe_agent.ProbeAgent`,
6
+ which uses an LLM judge to select analyzers from the description.
7
+ - **M2** uses the protocol to frame its statistical narrative.
8
+ - **M4** uses it to verify that a hypothesis is consistent with what the user
9
+ actually set out to investigate.
10
+
11
+ The description should be written in plain researcher language describing the
12
+ *task* and *observed behaviour*. No failure-mode tags or internal jargon are
13
+ needed — the judge LLM interprets the text and selects relevant analyzers.
14
+
15
+ Usage::
16
+
17
+ protocol = ExperimentProtocol(
18
+ description=(
19
+ "We test QwenVL on spatial reasoning. Given an image with two "
20
+ "objects, the model frequently gives wrong left/right and "
21
+ "above/below positions, and sometimes names objects not visible "
22
+ "in the image at all."
23
+ ),
24
+ task_domain="spatial reasoning",
25
+ success_criteria="Positions and object names must match what is visible.",
26
+ target_modalities=frozenset({"text", "image"}),
27
+ )
28
+ """
29
+
30
+ from __future__ import annotations
31
+
32
+ import json
33
+ from dataclasses import dataclass, field
34
+ from typing import Any
35
+
36
+
37
+ @dataclass
38
+ class ExperimentProtocol:
39
+ """Natural-language description of an evaluation experiment.
40
+
41
+ This is the *human prior* passed into the loop so it can make
42
+ informed, targeted decisions rather than running every analyzer
43
+ blindly.
44
+
45
+ Attributes:
46
+ description: What the experiment tests — free text (required).
47
+ task_domain: Short label, e.g. ``"spatial reasoning"``,
48
+ ``"GUI navigation"``.
49
+ success_criteria: What counts as a pass (used by M4 verifier).
50
+ failure_patterns: Optional free-text observations about what the
51
+ researcher has already noticed — passed verbatim
52
+ to the LLM judge as additional context.
53
+ target_modalities: ``{"text", "image"}`` for VLMs;
54
+ ``{"text"}`` for text-only LLMs.
55
+ output_contract: Optional machine-readable response contract. It
56
+ prevents a valid short answer from being confused
57
+ with a truncated reasoning trace.
58
+ metadata: Free-form extras (dataset names, hyperparams …).
59
+ """
60
+
61
+ description: str
62
+ task_domain: str = ""
63
+ success_criteria: str = ""
64
+ failure_patterns: str = ""
65
+ target_modalities: frozenset[str] = field(
66
+ default_factory=lambda: frozenset({"text"})
67
+ )
68
+ output_contract: dict[str, Any] = field(default_factory=dict)
69
+ metadata: dict[str, Any] = field(default_factory=dict)
70
+
71
+ def to_dict(self) -> dict[str, Any]:
72
+ return {
73
+ "description": self.description,
74
+ "task_domain": self.task_domain,
75
+ "success_criteria": self.success_criteria,
76
+ "failure_patterns": self.failure_patterns,
77
+ "target_modalities": sorted(self.target_modalities),
78
+ "output_contract": self.output_contract,
79
+ "metadata": self.metadata,
80
+ }
81
+
82
+ def prompt_text(self) -> str:
83
+ """Serialize the complete evaluation contract for judge prompts.
84
+
85
+ ``description`` alone is not enough for short-answer benchmarks: the
86
+ success criteria and output contract determine whether a terse answer
87
+ is valid. Keeping one formatter prevents M2/M3 prompt handoffs from
88
+ silently dropping those fields.
89
+ """
90
+ return json.dumps(self.to_dict(), ensure_ascii=False, indent=2, default=str)
91
+
92
+
93
+ @dataclass
94
+ class ProbingSchema:
95
+ """What M1 decided to probe and why.
96
+
97
+ Returned by :meth:`~evalrx.eval_agent.probe_agent.ProbeAgent.probe_with_schema`
98
+ alongside the raw ``{analyzer: Result}`` dict so callers can understand
99
+ *why* those analyzers were chosen.
100
+
101
+ Attributes:
102
+ selected_analyzers: Analyzers to run, in priority order.
103
+ rationale: NL explanation of the selection.
104
+ custom_params: Per-analyzer parameter overrides (e.g.
105
+ ``{"attention": {"layer": -1}}``).
106
+ protocol: The protocol that shaped this schema, if any.
107
+ """
108
+
109
+ selected_analyzers: list[str]
110
+ rationale: str = ""
111
+ custom_params: dict[str, dict[str, Any]] = field(default_factory=dict)
112
+ protocol: ExperimentProtocol | None = None