evalrx 0.1.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- evalrx/__init__.py +139 -0
- evalrx/agent_assets/__init__.py +2 -0
- evalrx/agent_assets/skills/README.md +28 -0
- evalrx/agent_assets/skills/eval-chart-style/SKILL.md +172 -0
- evalrx/agent_assets/skills/evalrx-report-ui/SKILL.md +116 -0
- evalrx/agent_assets/skills/nature-figure/LICENSE +201 -0
- evalrx/agent_assets/skills/nature-figure/README.md +412 -0
- evalrx/agent_assets/skills/nature-figure/SKILL.md +60 -0
- evalrx/agent_assets/skills/nature-figure/manifest.yaml +59 -0
- evalrx/agent_assets/skills/nature-figure/references/api.md +436 -0
- evalrx/agent_assets/skills/nature-figure/references/backend-selection.md +100 -0
- evalrx/agent_assets/skills/nature-figure/references/chart-types.md +281 -0
- evalrx/agent_assets/skills/nature-figure/references/common-patterns.md +350 -0
- evalrx/agent_assets/skills/nature-figure/references/demos.md +65 -0
- evalrx/agent_assets/skills/nature-figure/references/design-theory.md +439 -0
- evalrx/agent_assets/skills/nature-figure/references/figure-contract.md +93 -0
- evalrx/agent_assets/skills/nature-figure/references/figure-legend-conventions.md +71 -0
- evalrx/agent_assets/skills/nature-figure/references/nature-2026-observations.md +112 -0
- evalrx/agent_assets/skills/nature-figure/references/qa-contract.md +119 -0
- evalrx/agent_assets/skills/nature-figure/references/r-template-index.md +66 -0
- evalrx/agent_assets/skills/nature-figure/references/r-workflow.md +161 -0
- evalrx/agent_assets/skills/nature-figure/references/tutorials.md +251 -0
- evalrx/agent_assets/skills/nature-figure/static/core/contract.md +29 -0
- evalrx/agent_assets/skills/nature-figure/static/core/stance.md +37 -0
- evalrx/agent_assets/skills/nature-figure/static/fragments/backend/python.md +37 -0
- evalrx/agent_assets/skills/nature-figure/static/fragments/backend/r.md +44 -0
- evalrx/agent_assets/skills/outcome-driver-analysis/SKILL.md +213 -0
- evalrx/agent_assets/skills/outcome-driver-analysis/assets/analysis_report_template.md +53 -0
- evalrx/agent_assets/skills/outcome-driver-analysis/references/model_selection.md +72 -0
- evalrx/agent_assets/skills/outcome-driver-analysis/scripts/explanatory_var_eda.R +130 -0
- evalrx/agent_assets/skills/outcome-driver-analysis/scripts/explanatory_var_eda.py +150 -0
- evalrx/agent_assets/skills/outcome-driver-analysis/scripts/fit_outcome_model.R +181 -0
- evalrx/agent_assets/skills/outcome-driver-analysis/scripts/fit_outcome_model.py +186 -0
- evalrx/agent_assets/skills/outcome-driver-analysis/scripts/univariate_eda.R +149 -0
- evalrx/agent_assets/skills/outcome-driver-analysis/scripts/univariate_eda.py +177 -0
- evalrx/agent_assets/skills.py +27 -0
- evalrx/agent_runtime/__init__.py +78 -0
- evalrx/agent_runtime/_docker_runner.py +89 -0
- evalrx/agent_runtime/cli_runtime.py +103 -0
- evalrx/agent_runtime/cli_transcript.py +138 -0
- evalrx/agent_runtime/cli_types.py +68 -0
- evalrx/agent_runtime/codegen/__init__.py +5 -0
- evalrx/agent_runtime/codegen/runner.py +94 -0
- evalrx/agent_runtime/experiment_harness.py +117 -0
- evalrx/agent_runtime/factory.py +102 -0
- evalrx/agent_runtime/json_shape.py +44 -0
- evalrx/agent_runtime/judges/__init__.py +28 -0
- evalrx/agent_runtime/judges/agy.py +179 -0
- evalrx/agent_runtime/judges/autodetect.py +135 -0
- evalrx/agent_runtime/judges/claude.py +159 -0
- evalrx/agent_runtime/judges/codex.py +120 -0
- evalrx/agent_runtime/providers/__init__.py +21 -0
- evalrx/agent_runtime/providers/antigravity.py +31 -0
- evalrx/agent_runtime/providers/base.py +145 -0
- evalrx/agent_runtime/providers/claude_code.py +49 -0
- evalrx/agent_runtime/providers/codex.py +37 -0
- evalrx/agent_runtime/providers/gemini_cli.py +26 -0
- evalrx/agent_runtime/providers/kimi_cli.py +27 -0
- evalrx/agent_runtime/providers/opencode.py +27 -0
- evalrx/agent_runtime/providers/registry.py +58 -0
- evalrx/agent_runtime/sandbox.py +517 -0
- evalrx/agent_runtime/skill_audit.py +143 -0
- evalrx/agent_runtime/skills/__init__.py +19 -0
- evalrx/agent_runtime/skills/installer.py +68 -0
- evalrx/agent_runtime/skills/prompt_policy.py +86 -0
- evalrx/agent_runtime/skills/resolver.py +19 -0
- evalrx/analysis/__init__.py +132 -0
- evalrx/analysis/adjudicate.py +154 -0
- evalrx/analysis/analysis_module.py +361 -0
- evalrx/analysis/api.py +171 -0
- evalrx/analysis/case_studio.py +651 -0
- evalrx/analysis/cli.py +114 -0
- evalrx/analysis/dashboard.py +350 -0
- evalrx/analysis/eval_case_matrix.py +118 -0
- evalrx/analysis/eval_viz_theme.py +833 -0
- evalrx/analysis/explore_run.py +333 -0
- evalrx/analysis/explorer.py +1276 -0
- evalrx/analysis/failure_modes.py +607 -0
- evalrx/analysis/fused_pipeline.py +489 -0
- evalrx/analysis/holdout.py +300 -0
- evalrx/analysis/hypothesis_agent.py +230 -0
- evalrx/analysis/narration.py +177 -0
- evalrx/analysis/operationalize.py +442 -0
- evalrx/analysis/plain_language.py +42 -0
- evalrx/analysis/planner.py +283 -0
- evalrx/analysis/probe_search.py +203 -0
- evalrx/analysis/profile.py +268 -0
- evalrx/analysis/prompts/__init__.py +0 -0
- evalrx/analysis/prompts/explorer.py +417 -0
- evalrx/analysis/prompts/failure_modes.py +33 -0
- evalrx/analysis/prompts/holdout.py +27 -0
- evalrx/analysis/prompts/hypothesis_agent.py +78 -0
- evalrx/analysis/prompts/run_codebase.py +47 -0
- evalrx/analysis/prompts/stats_agent.py +72 -0
- evalrx/analysis/prompts/stats_tool_generator.py +43 -0
- evalrx/analysis/result_marker.py +47 -0
- evalrx/analysis/run_codebase.py +242 -0
- evalrx/analysis/run_view.py +205 -0
- evalrx/analysis/stage_views.py +93 -0
- evalrx/analysis/stats_agent.py +944 -0
- evalrx/analysis/stats_tool_agent.py +261 -0
- evalrx/analysis/stats_tool_generator.py +415 -0
- evalrx/analysis/stats_tools.py +1153 -0
- evalrx/analysis/trajectory_records.py +193 -0
- evalrx/analysis/workbench.py +431 -0
- evalrx/analyzers/__init__.py +42 -0
- evalrx/analyzers/agent/__init__.py +25 -0
- evalrx/analyzers/agent/counterfactual.py +84 -0
- evalrx/analyzers/agent/first_error_judge.py +96 -0
- evalrx/analyzers/agent/ignored_obs.py +81 -0
- evalrx/analyzers/agent/loop_detect.py +79 -0
- evalrx/analyzers/agent/reliability.py +165 -0
- evalrx/analyzers/agent/tool_shap.py +225 -0
- evalrx/analyzers/agent/trajectory_rubric.py +168 -0
- evalrx/analyzers/attention/__init__.py +19 -0
- evalrx/analyzers/attention/relative_attn.py +610 -0
- evalrx/analyzers/attention/rollout.py +73 -0
- evalrx/analyzers/attention/sink.py +56 -0
- evalrx/analyzers/attention/summary.py +190 -0
- evalrx/analyzers/attribution/__init__.py +6 -0
- evalrx/analyzers/attribution/generic_attn.py +31 -0
- evalrx/analyzers/attribution/gradcam.py +30 -0
- evalrx/analyzers/base.py +12 -0
- evalrx/analyzers/geometry/__init__.py +6 -0
- evalrx/analyzers/geometry/cka.py +70 -0
- evalrx/analyzers/geometry/linear_probe.py +157 -0
- evalrx/analyzers/hallucination/__init__.py +9 -0
- evalrx/analyzers/hallucination/chair.py +78 -0
- evalrx/analyzers/hallucination/opera.py +29 -0
- evalrx/analyzers/hallucination/pope.py +119 -0
- evalrx/analyzers/hallucination/selfcheck.py +155 -0
- evalrx/analyzers/hallucination/vcd.py +29 -0
- evalrx/analyzers/lens/__init__.py +7 -0
- evalrx/analyzers/lens/layer_contrast.py +133 -0
- evalrx/analyzers/lens/logit_lens.py +138 -0
- evalrx/analyzers/lens/tuned_lens.py +30 -0
- evalrx/analyzers/patching/__init__.py +5 -0
- evalrx/analyzers/patching/causal_trace.py +30 -0
- evalrx/analyzers/perturbation/__init__.py +23 -0
- evalrx/analyzers/perturbation/_shapley.py +54 -0
- evalrx/analyzers/perturbation/context_shap.py +174 -0
- evalrx/analyzers/perturbation/cot_faithfulness.py +239 -0
- evalrx/analyzers/perturbation/format_sensitivity.py +237 -0
- evalrx/analyzers/perturbation/mm_shap.py +146 -0
- evalrx/analyzers/perturbation/modality_ablation.py +196 -0
- evalrx/analyzers/perturbation/perturbation_battery.py +274 -0
- evalrx/analyzers/perturbation/prompt_contrast.py +265 -0
- evalrx/analyzers/perturbation/rise.py +94 -0
- evalrx/analyzers/perturbation/vl_shap.py +102 -0
- evalrx/analyzers/reasoning/__init__.py +33 -0
- evalrx/analyzers/reasoning/_text.py +328 -0
- evalrx/analyzers/reasoning/answer_extraction_audit.py +327 -0
- evalrx/analyzers/reasoning/arith_audit.py +226 -0
- evalrx/analyzers/reasoning/contamination.py +214 -0
- evalrx/analyzers/reasoning/knowledge_split.py +253 -0
- evalrx/analyzers/reasoning/self_repair.py +246 -0
- evalrx/analyzers/reasoning/step_rollout_value.py +216 -0
- evalrx/analyzers/reasoning/termination_audit.py +258 -0
- evalrx/analyzers/uncertainty/__init__.py +18 -0
- evalrx/analyzers/uncertainty/calibration.py +174 -0
- evalrx/analyzers/uncertainty/coverage_gap.py +199 -0
- evalrx/analyzers/uncertainty/entropy.py +90 -0
- evalrx/analyzers/uncertainty/logprob_entropy.py +69 -0
- evalrx/analyzers/uncertainty/self_consistency.py +204 -0
- evalrx/analyzers/uncertainty/verbalized_conf.py +64 -0
- evalrx/cli.py +411 -0
- evalrx/config.py +77 -0
- evalrx/contract/__init__.py +179 -0
- evalrx/contract/common.py +452 -0
- evalrx/contract/emit.py +948 -0
- evalrx/contract/export.py +237 -0
- evalrx/contract/m1.py +325 -0
- evalrx/contract/m2.py +317 -0
- evalrx/contract/m3.py +165 -0
- evalrx/contract/m4.py +130 -0
- evalrx/contract/m5.py +292 -0
- evalrx/contract/methodology.py +76 -0
- evalrx/contract/pre_m1.py +58 -0
- evalrx/contract/typescript.py +140 -0
- evalrx/core/__init__.py +85 -0
- evalrx/core/analyzer.py +174 -0
- evalrx/core/capability.py +54 -0
- evalrx/core/case.py +443 -0
- evalrx/core/experiment.py +106 -0
- evalrx/core/model.py +198 -0
- evalrx/core/pipeline.py +42 -0
- evalrx/core/registry.py +142 -0
- evalrx/core/result.py +64 -0
- evalrx/core/spec.py +173 -0
- evalrx/core/tokentype.py +165 -0
- evalrx/core/tool.py +92 -0
- evalrx/datasets/__init__.py +41 -0
- evalrx/datasets/base.py +68 -0
- evalrx/datasets/gui_os.py +52 -0
- evalrx/datasets/llm_qa.py +57 -0
- evalrx/datasets/pure_qa.py +12 -0
- evalrx/datasets/vlm_qa.py +695 -0
- evalrx/datasets/web_search_qa.py +52 -0
- evalrx/eval_agent/__init__.py +341 -0
- evalrx/eval_agent/_tools.py +81 -0
- evalrx/eval_agent/ab_runner.py +50 -0
- evalrx/eval_agent/agentic/__init__.py +43 -0
- evalrx/eval_agent/agentic/actions.py +216 -0
- evalrx/eval_agent/agentic/board.py +107 -0
- evalrx/eval_agent/agentic/loop.py +190 -0
- evalrx/eval_agent/agentic/tools.py +538 -0
- evalrx/eval_agent/checkpoint.py +57 -0
- evalrx/eval_agent/cli_agent.py +59 -0
- evalrx/eval_agent/cli_skills.py +5 -0
- evalrx/eval_agent/evolution.py +396 -0
- evalrx/eval_agent/git_manager.py +215 -0
- evalrx/eval_agent/hypothesis.py +172 -0
- evalrx/eval_agent/label_quarantine.py +209 -0
- evalrx/eval_agent/legacy.py +530 -0
- evalrx/eval_agent/log_schema.py +497 -0
- evalrx/eval_agent/loop.py +2159 -0
- evalrx/eval_agent/loop_reports.py +116 -0
- evalrx/eval_agent/model_instrumentation.py +282 -0
- evalrx/eval_agent/narration.py +193 -0
- evalrx/eval_agent/nl_runner.py +460 -0
- evalrx/eval_agent/orchestrator.py +61 -0
- evalrx/eval_agent/preregister.py +93 -0
- evalrx/eval_agent/prompts/__init__.py +1 -0
- evalrx/eval_agent/prompts/agentic.py +46 -0
- evalrx/eval_agent/prompts/case_discovery.py +25 -0
- evalrx/eval_agent/prompts/diagnosis.py +125 -0
- evalrx/eval_agent/prompts/experiment_writer.py +265 -0
- evalrx/eval_agent/prompts/explore_step.py +37 -0
- evalrx/eval_agent/prompts/fix_agent.py +257 -0
- evalrx/eval_agent/prompts/hypothesis_tester.py +15 -0
- evalrx/eval_agent/prompts/nl_runner.py +38 -0
- evalrx/eval_agent/prompts/probe_agent.py +25 -0
- evalrx/eval_agent/prompts/probe_candidate_generator.py +14 -0
- evalrx/eval_agent/prompts/probe_generator.py +35 -0
- evalrx/eval_agent/prompts/whitebox_probe_generator.py +38 -0
- evalrx/eval_agent/report.py +58 -0
- evalrx/eval_agent/run_context.py +354 -0
- evalrx/eval_agent/run_log.schema.json +1215 -0
- evalrx/eval_agent/run_logger_v2.py +1764 -0
- evalrx/eval_agent/run_metadata.py +208 -0
- evalrx/eval_agent/stages/__init__.py +56 -0
- evalrx/eval_agent/stages/case_discovery.py +293 -0
- evalrx/eval_agent/stages/diagnosis.py +1017 -0
- evalrx/eval_agent/stages/experiment_writer.py +1634 -0
- evalrx/eval_agent/stages/fix_agent.py +3916 -0
- evalrx/eval_agent/stages/fix_internals.py +499 -0
- evalrx/eval_agent/stages/fix_pipeline.py +725 -0
- evalrx/eval_agent/stages/fix_tiers.py +187 -0
- evalrx/eval_agent/stages/fix_tools.py +1034 -0
- evalrx/eval_agent/stages/hypothesis_tester.py +1014 -0
- evalrx/eval_agent/stages/probe.py +439 -0
- evalrx/eval_agent/stages/probe_agent.py +1079 -0
- evalrx/eval_agent/stages/probe_candidate_generator.py +128 -0
- evalrx/eval_agent/stages/probe_generator.py +326 -0
- evalrx/eval_agent/stages/probe_search_agent.py +106 -0
- evalrx/eval_agent/stages/protocol.py +112 -0
- evalrx/eval_agent/stages/repair_catalog.py +273 -0
- evalrx/eval_agent/stages/surgery.py +524 -0
- evalrx/eval_agent/stages/whitebox_probe_generator.py +351 -0
- evalrx/eval_agent/store.py +231 -0
- evalrx/logging_utils.py +112 -0
- evalrx/models/__init__.py +161 -0
- evalrx/models/_discover.py +101 -0
- evalrx/models/agent.py +380 -0
- evalrx/models/backends/__init__.py +58 -0
- evalrx/models/backends/api.py +169 -0
- evalrx/models/backends/base.py +57 -0
- evalrx/models/backends/gemini_compat.py +579 -0
- evalrx/models/backends/hf_local.py +2074 -0
- evalrx/models/backends/openai_compat.py +301 -0
- evalrx/models/backends/vllm_offline.py +116 -0
- evalrx/models/base.py +24 -0
- evalrx/models/blackbox/__init__.py +4 -0
- evalrx/models/blackbox/agent.py +31 -0
- evalrx/models/blackbox/base.py +29 -0
- evalrx/models/blackbox/gemini.py +279 -0
- evalrx/models/blackbox/llm_api.py +17 -0
- evalrx/models/blackbox/vlm_api.py +17 -0
- evalrx/models/compose.py +66 -0
- evalrx/models/inference.py +88 -0
- evalrx/models/paper_methods/__init__.py +8 -0
- evalrx/models/paper_methods/aad.py +53 -0
- evalrx/models/paper_methods/ifcd.py +204 -0
- evalrx/models/paper_methods/pai.py +164 -0
- evalrx/models/paper_methods/tcd.py +202 -0
- evalrx/models/paper_methods/vcd.py +45 -0
- evalrx/models/paper_methods/vicrop.py +137 -0
- evalrx/models/toolcodec.py +143 -0
- evalrx/models/tools/__init__.py +20 -0
- evalrx/models/tools/perception.py +300 -0
- evalrx/models/tools/visual.py +174 -0
- evalrx/models/whitebox/__init__.py +26 -0
- evalrx/models/whitebox/agent.py +31 -0
- evalrx/models/whitebox/base.py +24 -0
- evalrx/models/whitebox/qwen.py +61 -0
- evalrx/models/whitebox/qwen2_5_omni.py +29 -0
- evalrx/models/whitebox/qwen2_audio.py +25 -0
- evalrx/models/whitebox/qwen_omni.py +53 -0
- evalrx/models/whitebox/qwen_vl.py +62 -0
- evalrx/observability/__init__.py +21 -0
- evalrx/observability/envelope.py +122 -0
- evalrx/observability/outbox.py +111 -0
- evalrx/observability/tracer.py +882 -0
- evalrx/reporting/__init__.py +28 -0
- evalrx/reporting/case_study.py +947 -0
- evalrx/reporting/compiler.py +587 -0
- evalrx/reporting/dynamic.py +1882 -0
- evalrx/reporting/html_report.py +2225 -0
- evalrx/reporting/langfuse_exporter.py +38 -0
- evalrx/reporting/langfuse_source.py +155 -0
- evalrx/reporting/model.py +151 -0
- evalrx/reporting/run_events.py +184 -0
- evalrx/reporting/server.py +557 -0
- evalrx/reporting/stages.py +58 -0
- evalrx/reporting/static_export.py +142 -0
- evalrx/reporting/web_dist/index.html +146 -0
- evalrx/specs.py +727 -0
- evalrx/stats/__init__.py +47 -0
- evalrx/stats/api.py +192 -0
- evalrx/stats/bootstrap.py +86 -0
- evalrx/stats/ebh.py +27 -0
- evalrx/stats/evalue.py +98 -0
- evalrx/stats/friedman.py +138 -0
- evalrx/stats/mcnemar.py +40 -0
- evalrx/stats/multiplicity.py +159 -0
- evalrx/stats/subset_sampling.py +55 -0
- evalrx/term_links.py +43 -0
- evalrx/viz/__init__.py +7 -0
- evalrx/viz/labels.py +77 -0
- evalrx/viz/prompts.py +39 -0
- evalrx/viz/renderer.py +590 -0
- evalrx/viz/schema.py +36 -0
- evalrx/viz/style.py +134 -0
- evalrx-0.1.2.dist-info/METADATA +532 -0
- evalrx-0.1.2.dist-info/RECORD +339 -0
- evalrx-0.1.2.dist-info/WHEEL +5 -0
- evalrx-0.1.2.dist-info/entry_points.txt +3 -0
- evalrx-0.1.2.dist-info/licenses/LICENSE +121 -0
- evalrx-0.1.2.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,128 @@
|
|
|
1
|
+
"""VLM probe candidate generator — ProbeLLM Macro/Micro generators (Sec 3.3),
|
|
2
|
+
scoped to VLM QA and plugged into :class:`~evalrx.analysis.probe_search.ProbeSearch`.
|
|
3
|
+
|
|
4
|
+
Scope (v1): both Macro and Micro produce a *paraphrase* of an existing seed's
|
|
5
|
+
question over the SAME image, never new imagery and never an altered
|
|
6
|
+
semantic target — so the seed's ``expected`` answer stays valid for the new
|
|
7
|
+
candidate without needing a vision-capable judge or an image/answer
|
|
8
|
+
generation tool (the paper's tool-augmented generation, which invokes web/code
|
|
9
|
+
tools to obtain or verify a *new* gold answer, is out of scope here — see
|
|
10
|
+
``ProbeSearch``'s docstring: a richer generator with those tools can be
|
|
11
|
+
substituted without touching the search algorithm itself).
|
|
12
|
+
|
|
13
|
+
- **Macro** (broad coverage): picks the seed question least similar (by token
|
|
14
|
+
overlap) to what the macro tree has already explored, then paraphrases it —
|
|
15
|
+
diversifies which part of the fixed image pool gets visited next.
|
|
16
|
+
- **Micro** (local refinement): paraphrases the *current search node's own*
|
|
17
|
+
case — same image, same gold answer, different wording — probing surface
|
|
18
|
+
robustness (does the model's correctness flip on a reworded but
|
|
19
|
+
semantically identical question) rather than the paper's full
|
|
20
|
+
entity/attribute substitution.
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
from __future__ import annotations
|
|
24
|
+
|
|
25
|
+
import logging
|
|
26
|
+
import re
|
|
27
|
+
from dataclasses import dataclass
|
|
28
|
+
from typing import TYPE_CHECKING
|
|
29
|
+
|
|
30
|
+
from evalrx.core.case import CaseBatch, FailureCase, Inputs
|
|
31
|
+
from evalrx.eval_agent.prompts.probe_candidate_generator import PARAPHRASE_PROMPT
|
|
32
|
+
|
|
33
|
+
if TYPE_CHECKING:
|
|
34
|
+
from evalrx.analysis.probe_search import ProbeNode
|
|
35
|
+
from evalrx.core.model import Model
|
|
36
|
+
|
|
37
|
+
logger = logging.getLogger(__name__)
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def _tokenize(text: str) -> set[str]:
|
|
41
|
+
return set(re.findall(r"[a-z0-9]+", text.lower()))
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def _jaccard_distance(a: str, b: str) -> float:
|
|
45
|
+
"""1.0 = no shared tokens (maximally distinct), 0.0 = identical token sets."""
|
|
46
|
+
ta, tb = _tokenize(a), _tokenize(b)
|
|
47
|
+
union = ta | tb
|
|
48
|
+
if not union:
|
|
49
|
+
return 0.0
|
|
50
|
+
return 1.0 - len(ta & tb) / len(union)
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def _extract_question(raw: str) -> str:
|
|
54
|
+
text = raw.strip()
|
|
55
|
+
if text.startswith("```"):
|
|
56
|
+
text = re.sub(r"^```\w*\n?", "", text)
|
|
57
|
+
text = re.sub(r"\n?```\s*$", "", text)
|
|
58
|
+
text = re.sub(r"^(question|paraphrase)\s*:\s*", "", text, flags=re.IGNORECASE)
|
|
59
|
+
return text.strip().strip('"')
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
@dataclass
|
|
63
|
+
class VLMProbeCandidateGenerator:
|
|
64
|
+
"""ProbeLLM-style Macro/Micro candidate generation over a fixed VLM seed pool.
|
|
65
|
+
|
|
66
|
+
Args:
|
|
67
|
+
seed_pool: Existing (image, question, expected) cases — e.g. loaded via
|
|
68
|
+
``VLMQADataset`` — the only source of images/gold answers
|
|
69
|
+
(Scope v1: no new imagery, no new gold synthesis).
|
|
70
|
+
judge: Text-only judge used to paraphrase questions (only the
|
|
71
|
+
question text is sent — never the image, so no
|
|
72
|
+
vision-capable judge is required). Required for either
|
|
73
|
+
regime to produce candidates; ``available`` is False and
|
|
74
|
+
both ``macro``/``micro`` return ``None`` without one
|
|
75
|
+
(mirrors ``ProbeGenerator.available``).
|
|
76
|
+
max_repairs: Retries if the judge echoes the question unchanged.
|
|
77
|
+
"""
|
|
78
|
+
|
|
79
|
+
seed_pool: CaseBatch
|
|
80
|
+
judge: "Model | None" = None
|
|
81
|
+
max_repairs: int = 1
|
|
82
|
+
|
|
83
|
+
@property
|
|
84
|
+
def available(self) -> bool:
|
|
85
|
+
return self.judge is not None and len(self.seed_pool) > 0
|
|
86
|
+
|
|
87
|
+
def macro(self, node: "ProbeNode", explored: "list[ProbeNode]") -> "FailureCase | None":
|
|
88
|
+
"""Diversify: paraphrase the pool seed least similar to what the macro
|
|
89
|
+
tree has already visited (paper Eq.11's "under-represented" frontier,
|
|
90
|
+
approximated here by question-token overlap rather than embeddings)."""
|
|
91
|
+
if not self.available:
|
|
92
|
+
return None
|
|
93
|
+
explored_prompts = [n.case.inputs.prompt for n in explored]
|
|
94
|
+
seed = max(
|
|
95
|
+
self.seed_pool,
|
|
96
|
+
key=lambda c: min(
|
|
97
|
+
(_jaccard_distance(c.inputs.prompt, p) for p in explored_prompts),
|
|
98
|
+
default=1.0,
|
|
99
|
+
),
|
|
100
|
+
)
|
|
101
|
+
return self._paraphrase(seed, style="a very differently worded question")
|
|
102
|
+
|
|
103
|
+
def micro(self, node: "ProbeNode") -> "FailureCase | None":
|
|
104
|
+
"""Refine: paraphrase the search node's own case for local
|
|
105
|
+
surface-robustness probing (same image, same gold answer)."""
|
|
106
|
+
if not self.available:
|
|
107
|
+
return None
|
|
108
|
+
return self._paraphrase(
|
|
109
|
+
node.case, style="a lightly reworded variant (synonyms/reordering)"
|
|
110
|
+
)
|
|
111
|
+
|
|
112
|
+
def _paraphrase(self, base: FailureCase, *, style: str) -> "FailureCase | None":
|
|
113
|
+
prompt = PARAPHRASE_PROMPT.format(question=base.inputs.prompt, style=style)
|
|
114
|
+
for _attempt in range(self.max_repairs + 1):
|
|
115
|
+
try:
|
|
116
|
+
raw = self.judge.generate(prompt) # type: ignore[union-attr]
|
|
117
|
+
except Exception as exc: # noqa: BLE001 — generation is best-effort
|
|
118
|
+
logger.warning("VLMProbeCandidateGenerator: judge.generate failed: %s", exc)
|
|
119
|
+
return None
|
|
120
|
+
question = _extract_question(str(raw))
|
|
121
|
+
if question and question.strip().lower() != base.inputs.prompt.strip().lower():
|
|
122
|
+
return FailureCase(
|
|
123
|
+
inputs=Inputs(prompt=question, image=base.inputs.image),
|
|
124
|
+
expected=base.expected,
|
|
125
|
+
tags={"probe_search_candidate"},
|
|
126
|
+
metadata={"seed_case_id": base.id},
|
|
127
|
+
)
|
|
128
|
+
return None
|
|
@@ -0,0 +1,326 @@
|
|
|
1
|
+
"""M1 tier (b) — ProbeGenerator: synthesise a new black-box probe on demand.
|
|
2
|
+
|
|
3
|
+
When no registered analyzer in the catalog targets the observed failure, this
|
|
4
|
+
generator creates a bespoke *probe* — but adapted to M1's reality: a probe needs
|
|
5
|
+
the model, which (unlike M2's data-only stats tools) cannot be shipped into a
|
|
6
|
+
subprocess. So the split is:
|
|
7
|
+
|
|
8
|
+
1. **Host collects** the model's outputs on the cases (the host owns the loaded
|
|
9
|
+
model) into ``m1_probe_input.json``.
|
|
10
|
+
2. A **sandboxed generated script** reads that JSON and computes a per-case probe
|
|
11
|
+
metric over the *outputs* (refusal detection, language drift, format/printf
|
|
12
|
+
adherence, length, keyword presence, answer-extraction failure, …), printing a
|
|
13
|
+
strict ``PROBE_RESULT_JSON=`` line.
|
|
14
|
+
3. The host wraps the parsed findings into a
|
|
15
|
+
:class:`~evalrx.core.result.Result` whose ``per_case`` entries flow into
|
|
16
|
+
M2 (stats tools) → M4 exactly like any catalog analyzer's output.
|
|
17
|
+
|
|
18
|
+
This keeps the M2-tier(b) safety model: generated code never touches the repo
|
|
19
|
+
source, never sees the weights, and runs in an
|
|
20
|
+
:class:`~evalrx.agent_runtime.sandbox.ExperimentSandbox` subprocess.
|
|
21
|
+
|
|
22
|
+
Scope (v1): probes that are **functions over a single forward-pass output**.
|
|
23
|
+
Multi-sample (self-consistency), perturbation, and white-box probes need the
|
|
24
|
+
model handle and are out of scope here — they belong to an in-process path.
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
from __future__ import annotations
|
|
28
|
+
|
|
29
|
+
import json
|
|
30
|
+
import logging
|
|
31
|
+
import re
|
|
32
|
+
from dataclasses import dataclass
|
|
33
|
+
from pathlib import Path
|
|
34
|
+
from typing import TYPE_CHECKING, Any
|
|
35
|
+
|
|
36
|
+
from evalrx.agent_runtime.sandbox import ExperimentSandbox
|
|
37
|
+
from evalrx.core.result import Result
|
|
38
|
+
from evalrx.eval_agent.prompts.probe_generator import (
|
|
39
|
+
_GENERATE_PROMPT,
|
|
40
|
+
_INPUT_FILENAME,
|
|
41
|
+
_MAX_OUTPUT_CHARS,
|
|
42
|
+
_RESULT_MARKER,
|
|
43
|
+
)
|
|
44
|
+
|
|
45
|
+
if TYPE_CHECKING:
|
|
46
|
+
from evalrx.agent_runtime.cli_types import CliAgentConfig
|
|
47
|
+
from evalrx.core.case import CaseBatch
|
|
48
|
+
from evalrx.core.model import Model
|
|
49
|
+
|
|
50
|
+
logger = logging.getLogger(__name__)
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
@dataclass
|
|
55
|
+
class GeneratedProbe:
|
|
56
|
+
"""A validated, sandbox-executed probe that can be re-run on new cases.
|
|
57
|
+
|
|
58
|
+
Attributes:
|
|
59
|
+
name: Short identifier (becomes ``generated:<name>``).
|
|
60
|
+
code: The Python source that was written and executed.
|
|
61
|
+
need: The natural-language failure pattern it probes for.
|
|
62
|
+
source: Which backend wrote it (``"cli:<provider>"`` or ``"llm"``).
|
|
63
|
+
"""
|
|
64
|
+
|
|
65
|
+
name: str
|
|
66
|
+
code: str
|
|
67
|
+
need: str = ""
|
|
68
|
+
source: str = ""
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
class ProbeGenerator:
|
|
72
|
+
"""Generate, run, and cache bespoke black-box probes in a sandbox.
|
|
73
|
+
|
|
74
|
+
Args:
|
|
75
|
+
judge: LLM used for the single-pass code-writing path.
|
|
76
|
+
cli_config: CLI coding-agent config (``provider != "llm"``) used before
|
|
77
|
+
the judge when present.
|
|
78
|
+
sandbox: Execution sandbox (fresh temp-dir when ``None``).
|
|
79
|
+
timeout_sec: Hard wall-clock limit per sandbox run.
|
|
80
|
+
max_cases: Cap on cases whose outputs are collected (cost guard); 0 (the default) = every case.
|
|
81
|
+
"""
|
|
82
|
+
|
|
83
|
+
def __init__(
|
|
84
|
+
self,
|
|
85
|
+
judge: "Model | None" = None,
|
|
86
|
+
cli_config: "CliAgentConfig | None" = None,
|
|
87
|
+
sandbox: "ExperimentSandbox | None" = None,
|
|
88
|
+
timeout_sec: int = 60,
|
|
89
|
+
max_cases: int = 0,
|
|
90
|
+
run_logger: "Any | None" = None,
|
|
91
|
+
) -> None:
|
|
92
|
+
self._judge = judge
|
|
93
|
+
self._cli_config = cli_config
|
|
94
|
+
self._timeout_sec = timeout_sec
|
|
95
|
+
self._max_cases = max_cases
|
|
96
|
+
self._sandbox = sandbox or ExperimentSandbox()
|
|
97
|
+
# Optional RunLogger — when set, every code-writing attempt (the prompt,
|
|
98
|
+
# the code produced, the backend used, and the pass/fail outcome) is
|
|
99
|
+
# recorded as a "tool_codegen" event so tool synthesis is fully traceable.
|
|
100
|
+
self.run_logger = run_logger
|
|
101
|
+
self._last_prompt: str = ""
|
|
102
|
+
self._last_raw: str = ""
|
|
103
|
+
self._last_raw_stream: str = ""
|
|
104
|
+
self._last_usage: dict | None = None
|
|
105
|
+
|
|
106
|
+
@property
|
|
107
|
+
def available(self) -> bool:
|
|
108
|
+
return self._judge is not None or (
|
|
109
|
+
self._cli_config is not None and self._cli_config.provider != "llm"
|
|
110
|
+
)
|
|
111
|
+
|
|
112
|
+
# ------------------------------------------------------------------
|
|
113
|
+
# Public interface
|
|
114
|
+
# ------------------------------------------------------------------
|
|
115
|
+
|
|
116
|
+
def generate(
|
|
117
|
+
self,
|
|
118
|
+
need: str,
|
|
119
|
+
model: "Model",
|
|
120
|
+
cases: "CaseBatch",
|
|
121
|
+
name: str = "custom",
|
|
122
|
+
) -> "tuple[Result | None, GeneratedProbe | None]":
|
|
123
|
+
"""Collect outputs, write+run a probe over them, return (Result, probe)."""
|
|
124
|
+
if not self.available:
|
|
125
|
+
logger.debug("ProbeGenerator: no code-writing backend configured")
|
|
126
|
+
return None, None
|
|
127
|
+
|
|
128
|
+
self._collect_outputs(model, cases)
|
|
129
|
+
self._last_prompt = ""
|
|
130
|
+
self._last_raw = ""
|
|
131
|
+
self._last_raw_stream = ""
|
|
132
|
+
self._last_usage = None
|
|
133
|
+
try:
|
|
134
|
+
code, source = self._write_code(need)
|
|
135
|
+
except Exception as exc:
|
|
136
|
+
logger.warning("ProbeGenerator: code writing failed: %s", exc)
|
|
137
|
+
self._emit_codegen(name, need, "", "", ok=False, error=f"code writing failed: {exc}")
|
|
138
|
+
return None, None
|
|
139
|
+
if not code.strip():
|
|
140
|
+
self._emit_codegen(name, need, source, "", ok=False, error="empty code produced")
|
|
141
|
+
return None, None
|
|
142
|
+
|
|
143
|
+
result = self._run_code(code, name, model, cases)
|
|
144
|
+
self._emit_codegen(
|
|
145
|
+
name, need, source, code, ok=result is not None,
|
|
146
|
+
error="" if result is not None else "sandbox produced no parseable result",
|
|
147
|
+
)
|
|
148
|
+
if result is None:
|
|
149
|
+
return None, None
|
|
150
|
+
probe = GeneratedProbe(name=name, code=code, need=need, source=source)
|
|
151
|
+
return result, probe
|
|
152
|
+
|
|
153
|
+
def _emit_codegen(
|
|
154
|
+
self, name: str, need: str, source: str, code: str, *, ok: bool, error: str = ""
|
|
155
|
+
) -> None:
|
|
156
|
+
"""Record one code-writing attempt to the RunLogger, if attached."""
|
|
157
|
+
if self.run_logger is None:
|
|
158
|
+
return
|
|
159
|
+
extra = ({"cli_usage": self._last_usage}
|
|
160
|
+
if source.startswith("cli:") and self._last_usage else None)
|
|
161
|
+
try:
|
|
162
|
+
self.run_logger.log_tool_codegen(
|
|
163
|
+
module="m1_probe", name=name, need=need, source=source, ok=ok,
|
|
164
|
+
code=code, prompt=self._last_prompt, raw_output=self._last_raw,
|
|
165
|
+
raw_stream=self._last_raw_stream, error=error,
|
|
166
|
+
extra=extra,
|
|
167
|
+
)
|
|
168
|
+
except Exception as exc: # logging must never break generation
|
|
169
|
+
logger.debug("ProbeGenerator: log_tool_codegen failed: %s", exc)
|
|
170
|
+
|
|
171
|
+
def run_cached(
|
|
172
|
+
self,
|
|
173
|
+
probe: GeneratedProbe,
|
|
174
|
+
model: "Model",
|
|
175
|
+
cases: "CaseBatch",
|
|
176
|
+
) -> "Result | None":
|
|
177
|
+
"""Re-run an already-generated probe on fresh cases (no LLM call)."""
|
|
178
|
+
self._collect_outputs(model, cases)
|
|
179
|
+
return self._run_code(probe.code, probe.name, model, cases)
|
|
180
|
+
|
|
181
|
+
# ------------------------------------------------------------------
|
|
182
|
+
# Internals
|
|
183
|
+
# ------------------------------------------------------------------
|
|
184
|
+
|
|
185
|
+
def _collect_outputs(self, model: "Model", cases: "CaseBatch") -> None:
|
|
186
|
+
"""Run the model on each case and serialise outputs to the sandbox dir."""
|
|
187
|
+
records: list[dict[str, Any]] = []
|
|
188
|
+
selected = list(cases)
|
|
189
|
+
if self._max_cases > 0: # 0 = every case
|
|
190
|
+
selected = selected[: self._max_cases]
|
|
191
|
+
if getattr(self.run_logger, "preserve_full_model_io", False):
|
|
192
|
+
from evalrx.eval_agent.model_instrumentation import InstrumentedModel
|
|
193
|
+
|
|
194
|
+
model = InstrumentedModel(
|
|
195
|
+
model, self.run_logger,
|
|
196
|
+
cycle=int(getattr(self.run_logger, "current_cycle", -1)),
|
|
197
|
+
analyzer="generated_probe_input_collection",
|
|
198
|
+
case_prompts={c.inputs.prompt: c.id for c in selected},
|
|
199
|
+
batch_case_ids=[c.id for c in selected],
|
|
200
|
+
)
|
|
201
|
+
for case in selected:
|
|
202
|
+
inp = getattr(case, "inputs", None)
|
|
203
|
+
try:
|
|
204
|
+
output = str(model.generate(inp)) if inp is not None else ""
|
|
205
|
+
except Exception as exc: # a probe over partial outputs is still useful
|
|
206
|
+
logger.debug("ProbeGenerator: generate failed for %s: %s", case.id, exc)
|
|
207
|
+
output = ""
|
|
208
|
+
label = getattr(case, "label", None)
|
|
209
|
+
records.append({
|
|
210
|
+
"id": case.id,
|
|
211
|
+
"prompt": str(getattr(inp, "prompt", "")) if inp is not None else "",
|
|
212
|
+
"expected": getattr(case, "expected", None),
|
|
213
|
+
"label": getattr(label, "value", None),
|
|
214
|
+
"output": output[:_MAX_OUTPUT_CHARS],
|
|
215
|
+
})
|
|
216
|
+
path = Path(self._sandbox.workdir) / _INPUT_FILENAME
|
|
217
|
+
path.write_text(json.dumps({"cases": records}, default=str), encoding="utf-8")
|
|
218
|
+
|
|
219
|
+
def _write_code(self, need: str) -> tuple[str, str]:
|
|
220
|
+
if self._cli_config is not None and self._cli_config.provider != "llm":
|
|
221
|
+
code = self._write_code_cli(need)
|
|
222
|
+
if code:
|
|
223
|
+
return code, f"cli:{self._cli_config.provider}"
|
|
224
|
+
prompt = self._build_prompt(need, fenced=True)
|
|
225
|
+
self._last_prompt = prompt
|
|
226
|
+
raw = self._judge.generate(prompt) # type: ignore[union-attr]
|
|
227
|
+
self._last_raw = str(raw)
|
|
228
|
+
return _extract_code(str(raw)), "llm"
|
|
229
|
+
|
|
230
|
+
def _write_code_cli(self, need: str) -> str:
|
|
231
|
+
from evalrx.agent_runtime.codegen import CodegenRunner
|
|
232
|
+
|
|
233
|
+
prompt = self._build_prompt(need, fenced=False)
|
|
234
|
+
self._last_prompt = prompt
|
|
235
|
+
result = CodegenRunner(self._cli_config).write_code( # type: ignore[arg-type]
|
|
236
|
+
prompt,
|
|
237
|
+
workdir=Path(self._sandbox.workdir),
|
|
238
|
+
timeout_sec=self._timeout_sec,
|
|
239
|
+
preferred_filenames=("probe.py",),
|
|
240
|
+
)
|
|
241
|
+
self._last_raw = result.raw_output
|
|
242
|
+
self._last_raw_stream = ""
|
|
243
|
+
if result.raw_stream_path:
|
|
244
|
+
try:
|
|
245
|
+
self._last_raw_stream = (
|
|
246
|
+
Path(self._sandbox.workdir) / result.raw_stream_path
|
|
247
|
+
).read_text(encoding="utf-8")
|
|
248
|
+
except OSError:
|
|
249
|
+
pass
|
|
250
|
+
self._last_usage = result.usage
|
|
251
|
+
return result.code
|
|
252
|
+
|
|
253
|
+
def _build_prompt(self, need: str, *, fenced: bool) -> str:
|
|
254
|
+
return _GENERATE_PROMPT.format(
|
|
255
|
+
need=need.strip() or "Detect outputs that fail the task.",
|
|
256
|
+
input_filename=_INPUT_FILENAME,
|
|
257
|
+
marker=_RESULT_MARKER,
|
|
258
|
+
fences_hint=" inside a ```python code block" if fenced else
|
|
259
|
+
", written to a file named probe.py",
|
|
260
|
+
)
|
|
261
|
+
|
|
262
|
+
def _run_code(
|
|
263
|
+
self,
|
|
264
|
+
code: str,
|
|
265
|
+
name: str,
|
|
266
|
+
model: "Model",
|
|
267
|
+
cases: "CaseBatch",
|
|
268
|
+
) -> "Result | None":
|
|
269
|
+
sandbox_result = self._sandbox.run(code, timeout_sec=self._timeout_sec)
|
|
270
|
+
if not sandbox_result.ok:
|
|
271
|
+
logger.warning(
|
|
272
|
+
"ProbeGenerator: sandbox run failed (rc=%s): %s",
|
|
273
|
+
sandbox_result.returncode, (sandbox_result.stderr or "").strip()[:200],
|
|
274
|
+
)
|
|
275
|
+
return None
|
|
276
|
+
return _parse_result(sandbox_result.stdout, name, model, cases)
|
|
277
|
+
|
|
278
|
+
|
|
279
|
+
# ---------------------------------------------------------------------------
|
|
280
|
+
# Module helpers
|
|
281
|
+
# ---------------------------------------------------------------------------
|
|
282
|
+
|
|
283
|
+
def _extract_code(raw: str) -> str:
|
|
284
|
+
cleaned = re.sub(r"<think>.*?</think>", "", raw, flags=re.DOTALL)
|
|
285
|
+
fence = re.search(r"```(?:python)?\s*\n(.*?)```", cleaned, flags=re.DOTALL)
|
|
286
|
+
if fence:
|
|
287
|
+
return fence.group(1).strip()
|
|
288
|
+
return cleaned.strip()
|
|
289
|
+
|
|
290
|
+
|
|
291
|
+
def _parse_result(
|
|
292
|
+
stdout: str,
|
|
293
|
+
name: str,
|
|
294
|
+
model: "Model",
|
|
295
|
+
cases: "CaseBatch",
|
|
296
|
+
) -> "Result | None":
|
|
297
|
+
"""Parse the last ``PROBE_RESULT_JSON=`` line into a Result, or None."""
|
|
298
|
+
marker_line = None
|
|
299
|
+
for line in stdout.splitlines():
|
|
300
|
+
s = line.strip()
|
|
301
|
+
if s.startswith(_RESULT_MARKER):
|
|
302
|
+
marker_line = s[len(_RESULT_MARKER):]
|
|
303
|
+
if marker_line is None:
|
|
304
|
+
logger.warning("ProbeGenerator: no PROBE_RESULT_JSON line in probe output")
|
|
305
|
+
return None
|
|
306
|
+
try:
|
|
307
|
+
data = json.loads(marker_line)
|
|
308
|
+
except json.JSONDecodeError as exc:
|
|
309
|
+
logger.warning("ProbeGenerator: unparseable PROBE_RESULT_JSON: %s", exc)
|
|
310
|
+
return None
|
|
311
|
+
|
|
312
|
+
findings: dict[str, Any] = {}
|
|
313
|
+
raw_findings = data.get("findings")
|
|
314
|
+
if isinstance(raw_findings, dict):
|
|
315
|
+
findings.update(raw_findings)
|
|
316
|
+
per_case = data.get("per_case")
|
|
317
|
+
if isinstance(per_case, list):
|
|
318
|
+
findings["per_case"] = [e for e in per_case if isinstance(e, dict)]
|
|
319
|
+
|
|
320
|
+
return Result(
|
|
321
|
+
analyzer=f"generated:{name}",
|
|
322
|
+
model=repr(model),
|
|
323
|
+
cases=cases,
|
|
324
|
+
findings=findings,
|
|
325
|
+
metadata={"generated": True, "probe": name},
|
|
326
|
+
)
|
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
"""ProbeSearchAgent — wires ProbeLLM's hierarchical MCTS
|
|
2
|
+
(:mod:`evalrx.analysis.probe_search`) to a real target model + judge.
|
|
3
|
+
|
|
4
|
+
``analysis.probe_search`` stays standalone (a generic tree search over
|
|
5
|
+
injected callables); this eval_agent-layer module supplies those callables
|
|
6
|
+
from real components:
|
|
7
|
+
|
|
8
|
+
- verifier V -> :class:`~evalrx.eval_agent.stages.case_discovery.CaseDiscoveryAgent`
|
|
9
|
+
- Macro/Micro generators -> :class:`~evalrx.eval_agent.stages.probe_candidate_generator.VLMProbeCandidateGenerator`
|
|
10
|
+
|
|
11
|
+
The discovered failure cases (``ProbeSearchResult.failure_cases``) are plain
|
|
12
|
+
``FailureCase`` objects and feed directly into
|
|
13
|
+
:func:`evalrx.analysis.failure_modes.cluster_failures` for failure-mode
|
|
14
|
+
synthesis, or into M1-M4 like any other labeled batch.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import logging
|
|
20
|
+
from dataclasses import dataclass
|
|
21
|
+
from typing import TYPE_CHECKING, Any
|
|
22
|
+
|
|
23
|
+
from evalrx.analysis.probe_search import ProbeSearch, ProbeSearchResult
|
|
24
|
+
from evalrx.eval_agent.stages.case_discovery import CaseDiscoveryAgent
|
|
25
|
+
from evalrx.eval_agent.stages.probe_candidate_generator import VLMProbeCandidateGenerator
|
|
26
|
+
|
|
27
|
+
if TYPE_CHECKING:
|
|
28
|
+
from evalrx.core.case import CaseBatch, FailureCase
|
|
29
|
+
from evalrx.core.model import Model
|
|
30
|
+
from evalrx.eval_agent.stages.protocol import ExperimentProtocol
|
|
31
|
+
|
|
32
|
+
logger = logging.getLogger(__name__)
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
@dataclass
|
|
36
|
+
class ProbeSearchAgent:
|
|
37
|
+
"""Run a hierarchical Macro/Micro MCTS probe search against a target model.
|
|
38
|
+
|
|
39
|
+
Args:
|
|
40
|
+
judge: Text-only judge used both to score PASS/FAIL (via
|
|
41
|
+
``CaseDiscoveryAgent``) and to paraphrase Macro/Micro
|
|
42
|
+
candidates (via ``VLMProbeCandidateGenerator``). Required —
|
|
43
|
+
without it, generation is unavailable and the search finds
|
|
44
|
+
nothing (``ProbeSearchResult.n_simulations == 0``).
|
|
45
|
+
protocol: Optional experiment protocol passed to the discovery judge
|
|
46
|
+
for scoring context.
|
|
47
|
+
budget: Total simulations (T_max in the paper's Eq.4).
|
|
48
|
+
beta: UCB exploration constant (Eq.7).
|
|
49
|
+
w_max: Max children per search-tree node before progressive
|
|
50
|
+
widening forces a deeper descent instead of a new sibling.
|
|
51
|
+
"""
|
|
52
|
+
|
|
53
|
+
judge: "Model"
|
|
54
|
+
protocol: "ExperimentProtocol | None" = None
|
|
55
|
+
budget: int = 20
|
|
56
|
+
beta: float = 1.0
|
|
57
|
+
w_max: int = 3
|
|
58
|
+
run_logger: Any | None = None
|
|
59
|
+
|
|
60
|
+
def __post_init__(self) -> None:
|
|
61
|
+
if self.judge is None:
|
|
62
|
+
raise ValueError(
|
|
63
|
+
"ProbeSearchAgent requires a judge (e.g. ClaudeModel() or AgyModel()) "
|
|
64
|
+
"— without one, candidate generation is unavailable and the search "
|
|
65
|
+
"would silently discover nothing."
|
|
66
|
+
)
|
|
67
|
+
|
|
68
|
+
def run(self, model: "Model", seed_pool: "CaseBatch") -> ProbeSearchResult:
|
|
69
|
+
judge = self.judge
|
|
70
|
+
if getattr(self.run_logger, "preserve_full_model_io", False):
|
|
71
|
+
from evalrx.eval_agent.model_instrumentation import InstrumentedModel
|
|
72
|
+
|
|
73
|
+
cycle = int(getattr(self.run_logger, "current_cycle", -1))
|
|
74
|
+
ids = [case.id for case in seed_pool]
|
|
75
|
+
prompts = {case.inputs.prompt: case.id for case in seed_pool}
|
|
76
|
+
judge = InstrumentedModel(
|
|
77
|
+
judge, self.run_logger, cycle=cycle, analyzer="probe_search_judge",
|
|
78
|
+
batch_case_ids=ids,
|
|
79
|
+
)
|
|
80
|
+
model = InstrumentedModel(
|
|
81
|
+
model, self.run_logger, cycle=cycle, analyzer="probe_search_target",
|
|
82
|
+
case_prompts=prompts, batch_case_ids=ids,
|
|
83
|
+
)
|
|
84
|
+
discovery = CaseDiscoveryAgent(judge=judge)
|
|
85
|
+
generator = VLMProbeCandidateGenerator(seed_pool=seed_pool, judge=judge)
|
|
86
|
+
|
|
87
|
+
def verify(case: "FailureCase") -> "FailureCase":
|
|
88
|
+
report = discovery.discover(model, [case], protocol=self.protocol)
|
|
89
|
+
cases = list(report.cases)
|
|
90
|
+
return cases[0] if cases else case
|
|
91
|
+
|
|
92
|
+
search = ProbeSearch(
|
|
93
|
+
generate_macro=generator.macro,
|
|
94
|
+
generate_micro=generator.micro,
|
|
95
|
+
verify=verify,
|
|
96
|
+
seeds=seed_pool,
|
|
97
|
+
budget=self.budget,
|
|
98
|
+
beta=self.beta,
|
|
99
|
+
w_max=self.w_max,
|
|
100
|
+
)
|
|
101
|
+
result = search.run()
|
|
102
|
+
logger.info(
|
|
103
|
+
"ProbeSearchAgent: %d simulation(s) (macro=%d micro=%d), error_rate=%.2f",
|
|
104
|
+
result.n_simulations, result.n_macro, result.n_micro, result.error_rate,
|
|
105
|
+
)
|
|
106
|
+
return result
|
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
"""ExperimentProtocol — user-supplied description of what an experiment tests.
|
|
2
|
+
|
|
3
|
+
The protocol is the human prior that anchors the self-evolving loop:
|
|
4
|
+
|
|
5
|
+
- **M1** passes the protocol to :class:`~evalrx.eval_agent.probe_agent.ProbeAgent`,
|
|
6
|
+
which uses an LLM judge to select analyzers from the description.
|
|
7
|
+
- **M2** uses the protocol to frame its statistical narrative.
|
|
8
|
+
- **M4** uses it to verify that a hypothesis is consistent with what the user
|
|
9
|
+
actually set out to investigate.
|
|
10
|
+
|
|
11
|
+
The description should be written in plain researcher language describing the
|
|
12
|
+
*task* and *observed behaviour*. No failure-mode tags or internal jargon are
|
|
13
|
+
needed — the judge LLM interprets the text and selects relevant analyzers.
|
|
14
|
+
|
|
15
|
+
Usage::
|
|
16
|
+
|
|
17
|
+
protocol = ExperimentProtocol(
|
|
18
|
+
description=(
|
|
19
|
+
"We test QwenVL on spatial reasoning. Given an image with two "
|
|
20
|
+
"objects, the model frequently gives wrong left/right and "
|
|
21
|
+
"above/below positions, and sometimes names objects not visible "
|
|
22
|
+
"in the image at all."
|
|
23
|
+
),
|
|
24
|
+
task_domain="spatial reasoning",
|
|
25
|
+
success_criteria="Positions and object names must match what is visible.",
|
|
26
|
+
target_modalities=frozenset({"text", "image"}),
|
|
27
|
+
)
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
from __future__ import annotations
|
|
31
|
+
|
|
32
|
+
import json
|
|
33
|
+
from dataclasses import dataclass, field
|
|
34
|
+
from typing import Any
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
@dataclass
|
|
38
|
+
class ExperimentProtocol:
|
|
39
|
+
"""Natural-language description of an evaluation experiment.
|
|
40
|
+
|
|
41
|
+
This is the *human prior* passed into the loop so it can make
|
|
42
|
+
informed, targeted decisions rather than running every analyzer
|
|
43
|
+
blindly.
|
|
44
|
+
|
|
45
|
+
Attributes:
|
|
46
|
+
description: What the experiment tests — free text (required).
|
|
47
|
+
task_domain: Short label, e.g. ``"spatial reasoning"``,
|
|
48
|
+
``"GUI navigation"``.
|
|
49
|
+
success_criteria: What counts as a pass (used by M4 verifier).
|
|
50
|
+
failure_patterns: Optional free-text observations about what the
|
|
51
|
+
researcher has already noticed — passed verbatim
|
|
52
|
+
to the LLM judge as additional context.
|
|
53
|
+
target_modalities: ``{"text", "image"}`` for VLMs;
|
|
54
|
+
``{"text"}`` for text-only LLMs.
|
|
55
|
+
output_contract: Optional machine-readable response contract. It
|
|
56
|
+
prevents a valid short answer from being confused
|
|
57
|
+
with a truncated reasoning trace.
|
|
58
|
+
metadata: Free-form extras (dataset names, hyperparams …).
|
|
59
|
+
"""
|
|
60
|
+
|
|
61
|
+
description: str
|
|
62
|
+
task_domain: str = ""
|
|
63
|
+
success_criteria: str = ""
|
|
64
|
+
failure_patterns: str = ""
|
|
65
|
+
target_modalities: frozenset[str] = field(
|
|
66
|
+
default_factory=lambda: frozenset({"text"})
|
|
67
|
+
)
|
|
68
|
+
output_contract: dict[str, Any] = field(default_factory=dict)
|
|
69
|
+
metadata: dict[str, Any] = field(default_factory=dict)
|
|
70
|
+
|
|
71
|
+
def to_dict(self) -> dict[str, Any]:
|
|
72
|
+
return {
|
|
73
|
+
"description": self.description,
|
|
74
|
+
"task_domain": self.task_domain,
|
|
75
|
+
"success_criteria": self.success_criteria,
|
|
76
|
+
"failure_patterns": self.failure_patterns,
|
|
77
|
+
"target_modalities": sorted(self.target_modalities),
|
|
78
|
+
"output_contract": self.output_contract,
|
|
79
|
+
"metadata": self.metadata,
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
def prompt_text(self) -> str:
|
|
83
|
+
"""Serialize the complete evaluation contract for judge prompts.
|
|
84
|
+
|
|
85
|
+
``description`` alone is not enough for short-answer benchmarks: the
|
|
86
|
+
success criteria and output contract determine whether a terse answer
|
|
87
|
+
is valid. Keeping one formatter prevents M2/M3 prompt handoffs from
|
|
88
|
+
silently dropping those fields.
|
|
89
|
+
"""
|
|
90
|
+
return json.dumps(self.to_dict(), ensure_ascii=False, indent=2, default=str)
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
@dataclass
|
|
94
|
+
class ProbingSchema:
|
|
95
|
+
"""What M1 decided to probe and why.
|
|
96
|
+
|
|
97
|
+
Returned by :meth:`~evalrx.eval_agent.probe_agent.ProbeAgent.probe_with_schema`
|
|
98
|
+
alongside the raw ``{analyzer: Result}`` dict so callers can understand
|
|
99
|
+
*why* those analyzers were chosen.
|
|
100
|
+
|
|
101
|
+
Attributes:
|
|
102
|
+
selected_analyzers: Analyzers to run, in priority order.
|
|
103
|
+
rationale: NL explanation of the selection.
|
|
104
|
+
custom_params: Per-analyzer parameter overrides (e.g.
|
|
105
|
+
``{"attention": {"layer": -1}}``).
|
|
106
|
+
protocol: The protocol that shaped this schema, if any.
|
|
107
|
+
"""
|
|
108
|
+
|
|
109
|
+
selected_analyzers: list[str]
|
|
110
|
+
rationale: str = ""
|
|
111
|
+
custom_params: dict[str, dict[str, Any]] = field(default_factory=dict)
|
|
112
|
+
protocol: ExperimentProtocol | None = None
|