evalrx 0.1.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- evalrx/__init__.py +139 -0
- evalrx/agent_assets/__init__.py +2 -0
- evalrx/agent_assets/skills/README.md +28 -0
- evalrx/agent_assets/skills/eval-chart-style/SKILL.md +172 -0
- evalrx/agent_assets/skills/evalrx-report-ui/SKILL.md +116 -0
- evalrx/agent_assets/skills/nature-figure/LICENSE +201 -0
- evalrx/agent_assets/skills/nature-figure/README.md +412 -0
- evalrx/agent_assets/skills/nature-figure/SKILL.md +60 -0
- evalrx/agent_assets/skills/nature-figure/manifest.yaml +59 -0
- evalrx/agent_assets/skills/nature-figure/references/api.md +436 -0
- evalrx/agent_assets/skills/nature-figure/references/backend-selection.md +100 -0
- evalrx/agent_assets/skills/nature-figure/references/chart-types.md +281 -0
- evalrx/agent_assets/skills/nature-figure/references/common-patterns.md +350 -0
- evalrx/agent_assets/skills/nature-figure/references/demos.md +65 -0
- evalrx/agent_assets/skills/nature-figure/references/design-theory.md +439 -0
- evalrx/agent_assets/skills/nature-figure/references/figure-contract.md +93 -0
- evalrx/agent_assets/skills/nature-figure/references/figure-legend-conventions.md +71 -0
- evalrx/agent_assets/skills/nature-figure/references/nature-2026-observations.md +112 -0
- evalrx/agent_assets/skills/nature-figure/references/qa-contract.md +119 -0
- evalrx/agent_assets/skills/nature-figure/references/r-template-index.md +66 -0
- evalrx/agent_assets/skills/nature-figure/references/r-workflow.md +161 -0
- evalrx/agent_assets/skills/nature-figure/references/tutorials.md +251 -0
- evalrx/agent_assets/skills/nature-figure/static/core/contract.md +29 -0
- evalrx/agent_assets/skills/nature-figure/static/core/stance.md +37 -0
- evalrx/agent_assets/skills/nature-figure/static/fragments/backend/python.md +37 -0
- evalrx/agent_assets/skills/nature-figure/static/fragments/backend/r.md +44 -0
- evalrx/agent_assets/skills/outcome-driver-analysis/SKILL.md +213 -0
- evalrx/agent_assets/skills/outcome-driver-analysis/assets/analysis_report_template.md +53 -0
- evalrx/agent_assets/skills/outcome-driver-analysis/references/model_selection.md +72 -0
- evalrx/agent_assets/skills/outcome-driver-analysis/scripts/explanatory_var_eda.R +130 -0
- evalrx/agent_assets/skills/outcome-driver-analysis/scripts/explanatory_var_eda.py +150 -0
- evalrx/agent_assets/skills/outcome-driver-analysis/scripts/fit_outcome_model.R +181 -0
- evalrx/agent_assets/skills/outcome-driver-analysis/scripts/fit_outcome_model.py +186 -0
- evalrx/agent_assets/skills/outcome-driver-analysis/scripts/univariate_eda.R +149 -0
- evalrx/agent_assets/skills/outcome-driver-analysis/scripts/univariate_eda.py +177 -0
- evalrx/agent_assets/skills.py +27 -0
- evalrx/agent_runtime/__init__.py +78 -0
- evalrx/agent_runtime/_docker_runner.py +89 -0
- evalrx/agent_runtime/cli_runtime.py +103 -0
- evalrx/agent_runtime/cli_transcript.py +138 -0
- evalrx/agent_runtime/cli_types.py +68 -0
- evalrx/agent_runtime/codegen/__init__.py +5 -0
- evalrx/agent_runtime/codegen/runner.py +94 -0
- evalrx/agent_runtime/experiment_harness.py +117 -0
- evalrx/agent_runtime/factory.py +102 -0
- evalrx/agent_runtime/json_shape.py +44 -0
- evalrx/agent_runtime/judges/__init__.py +28 -0
- evalrx/agent_runtime/judges/agy.py +179 -0
- evalrx/agent_runtime/judges/autodetect.py +135 -0
- evalrx/agent_runtime/judges/claude.py +159 -0
- evalrx/agent_runtime/judges/codex.py +120 -0
- evalrx/agent_runtime/providers/__init__.py +21 -0
- evalrx/agent_runtime/providers/antigravity.py +31 -0
- evalrx/agent_runtime/providers/base.py +145 -0
- evalrx/agent_runtime/providers/claude_code.py +49 -0
- evalrx/agent_runtime/providers/codex.py +37 -0
- evalrx/agent_runtime/providers/gemini_cli.py +26 -0
- evalrx/agent_runtime/providers/kimi_cli.py +27 -0
- evalrx/agent_runtime/providers/opencode.py +27 -0
- evalrx/agent_runtime/providers/registry.py +58 -0
- evalrx/agent_runtime/sandbox.py +517 -0
- evalrx/agent_runtime/skill_audit.py +143 -0
- evalrx/agent_runtime/skills/__init__.py +19 -0
- evalrx/agent_runtime/skills/installer.py +68 -0
- evalrx/agent_runtime/skills/prompt_policy.py +86 -0
- evalrx/agent_runtime/skills/resolver.py +19 -0
- evalrx/analysis/__init__.py +132 -0
- evalrx/analysis/adjudicate.py +154 -0
- evalrx/analysis/analysis_module.py +361 -0
- evalrx/analysis/api.py +171 -0
- evalrx/analysis/case_studio.py +651 -0
- evalrx/analysis/cli.py +114 -0
- evalrx/analysis/dashboard.py +350 -0
- evalrx/analysis/eval_case_matrix.py +118 -0
- evalrx/analysis/eval_viz_theme.py +833 -0
- evalrx/analysis/explore_run.py +333 -0
- evalrx/analysis/explorer.py +1276 -0
- evalrx/analysis/failure_modes.py +607 -0
- evalrx/analysis/fused_pipeline.py +489 -0
- evalrx/analysis/holdout.py +300 -0
- evalrx/analysis/hypothesis_agent.py +230 -0
- evalrx/analysis/narration.py +177 -0
- evalrx/analysis/operationalize.py +442 -0
- evalrx/analysis/plain_language.py +42 -0
- evalrx/analysis/planner.py +283 -0
- evalrx/analysis/probe_search.py +203 -0
- evalrx/analysis/profile.py +268 -0
- evalrx/analysis/prompts/__init__.py +0 -0
- evalrx/analysis/prompts/explorer.py +417 -0
- evalrx/analysis/prompts/failure_modes.py +33 -0
- evalrx/analysis/prompts/holdout.py +27 -0
- evalrx/analysis/prompts/hypothesis_agent.py +78 -0
- evalrx/analysis/prompts/run_codebase.py +47 -0
- evalrx/analysis/prompts/stats_agent.py +72 -0
- evalrx/analysis/prompts/stats_tool_generator.py +43 -0
- evalrx/analysis/result_marker.py +47 -0
- evalrx/analysis/run_codebase.py +242 -0
- evalrx/analysis/run_view.py +205 -0
- evalrx/analysis/stage_views.py +93 -0
- evalrx/analysis/stats_agent.py +944 -0
- evalrx/analysis/stats_tool_agent.py +261 -0
- evalrx/analysis/stats_tool_generator.py +415 -0
- evalrx/analysis/stats_tools.py +1153 -0
- evalrx/analysis/trajectory_records.py +193 -0
- evalrx/analysis/workbench.py +431 -0
- evalrx/analyzers/__init__.py +42 -0
- evalrx/analyzers/agent/__init__.py +25 -0
- evalrx/analyzers/agent/counterfactual.py +84 -0
- evalrx/analyzers/agent/first_error_judge.py +96 -0
- evalrx/analyzers/agent/ignored_obs.py +81 -0
- evalrx/analyzers/agent/loop_detect.py +79 -0
- evalrx/analyzers/agent/reliability.py +165 -0
- evalrx/analyzers/agent/tool_shap.py +225 -0
- evalrx/analyzers/agent/trajectory_rubric.py +168 -0
- evalrx/analyzers/attention/__init__.py +19 -0
- evalrx/analyzers/attention/relative_attn.py +610 -0
- evalrx/analyzers/attention/rollout.py +73 -0
- evalrx/analyzers/attention/sink.py +56 -0
- evalrx/analyzers/attention/summary.py +190 -0
- evalrx/analyzers/attribution/__init__.py +6 -0
- evalrx/analyzers/attribution/generic_attn.py +31 -0
- evalrx/analyzers/attribution/gradcam.py +30 -0
- evalrx/analyzers/base.py +12 -0
- evalrx/analyzers/geometry/__init__.py +6 -0
- evalrx/analyzers/geometry/cka.py +70 -0
- evalrx/analyzers/geometry/linear_probe.py +157 -0
- evalrx/analyzers/hallucination/__init__.py +9 -0
- evalrx/analyzers/hallucination/chair.py +78 -0
- evalrx/analyzers/hallucination/opera.py +29 -0
- evalrx/analyzers/hallucination/pope.py +119 -0
- evalrx/analyzers/hallucination/selfcheck.py +155 -0
- evalrx/analyzers/hallucination/vcd.py +29 -0
- evalrx/analyzers/lens/__init__.py +7 -0
- evalrx/analyzers/lens/layer_contrast.py +133 -0
- evalrx/analyzers/lens/logit_lens.py +138 -0
- evalrx/analyzers/lens/tuned_lens.py +30 -0
- evalrx/analyzers/patching/__init__.py +5 -0
- evalrx/analyzers/patching/causal_trace.py +30 -0
- evalrx/analyzers/perturbation/__init__.py +23 -0
- evalrx/analyzers/perturbation/_shapley.py +54 -0
- evalrx/analyzers/perturbation/context_shap.py +174 -0
- evalrx/analyzers/perturbation/cot_faithfulness.py +239 -0
- evalrx/analyzers/perturbation/format_sensitivity.py +237 -0
- evalrx/analyzers/perturbation/mm_shap.py +146 -0
- evalrx/analyzers/perturbation/modality_ablation.py +196 -0
- evalrx/analyzers/perturbation/perturbation_battery.py +274 -0
- evalrx/analyzers/perturbation/prompt_contrast.py +265 -0
- evalrx/analyzers/perturbation/rise.py +94 -0
- evalrx/analyzers/perturbation/vl_shap.py +102 -0
- evalrx/analyzers/reasoning/__init__.py +33 -0
- evalrx/analyzers/reasoning/_text.py +328 -0
- evalrx/analyzers/reasoning/answer_extraction_audit.py +327 -0
- evalrx/analyzers/reasoning/arith_audit.py +226 -0
- evalrx/analyzers/reasoning/contamination.py +214 -0
- evalrx/analyzers/reasoning/knowledge_split.py +253 -0
- evalrx/analyzers/reasoning/self_repair.py +246 -0
- evalrx/analyzers/reasoning/step_rollout_value.py +216 -0
- evalrx/analyzers/reasoning/termination_audit.py +258 -0
- evalrx/analyzers/uncertainty/__init__.py +18 -0
- evalrx/analyzers/uncertainty/calibration.py +174 -0
- evalrx/analyzers/uncertainty/coverage_gap.py +199 -0
- evalrx/analyzers/uncertainty/entropy.py +90 -0
- evalrx/analyzers/uncertainty/logprob_entropy.py +69 -0
- evalrx/analyzers/uncertainty/self_consistency.py +204 -0
- evalrx/analyzers/uncertainty/verbalized_conf.py +64 -0
- evalrx/cli.py +411 -0
- evalrx/config.py +77 -0
- evalrx/contract/__init__.py +179 -0
- evalrx/contract/common.py +452 -0
- evalrx/contract/emit.py +948 -0
- evalrx/contract/export.py +237 -0
- evalrx/contract/m1.py +325 -0
- evalrx/contract/m2.py +317 -0
- evalrx/contract/m3.py +165 -0
- evalrx/contract/m4.py +130 -0
- evalrx/contract/m5.py +292 -0
- evalrx/contract/methodology.py +76 -0
- evalrx/contract/pre_m1.py +58 -0
- evalrx/contract/typescript.py +140 -0
- evalrx/core/__init__.py +85 -0
- evalrx/core/analyzer.py +174 -0
- evalrx/core/capability.py +54 -0
- evalrx/core/case.py +443 -0
- evalrx/core/experiment.py +106 -0
- evalrx/core/model.py +198 -0
- evalrx/core/pipeline.py +42 -0
- evalrx/core/registry.py +142 -0
- evalrx/core/result.py +64 -0
- evalrx/core/spec.py +173 -0
- evalrx/core/tokentype.py +165 -0
- evalrx/core/tool.py +92 -0
- evalrx/datasets/__init__.py +41 -0
- evalrx/datasets/base.py +68 -0
- evalrx/datasets/gui_os.py +52 -0
- evalrx/datasets/llm_qa.py +57 -0
- evalrx/datasets/pure_qa.py +12 -0
- evalrx/datasets/vlm_qa.py +695 -0
- evalrx/datasets/web_search_qa.py +52 -0
- evalrx/eval_agent/__init__.py +341 -0
- evalrx/eval_agent/_tools.py +81 -0
- evalrx/eval_agent/ab_runner.py +50 -0
- evalrx/eval_agent/agentic/__init__.py +43 -0
- evalrx/eval_agent/agentic/actions.py +216 -0
- evalrx/eval_agent/agentic/board.py +107 -0
- evalrx/eval_agent/agentic/loop.py +190 -0
- evalrx/eval_agent/agentic/tools.py +538 -0
- evalrx/eval_agent/checkpoint.py +57 -0
- evalrx/eval_agent/cli_agent.py +59 -0
- evalrx/eval_agent/cli_skills.py +5 -0
- evalrx/eval_agent/evolution.py +396 -0
- evalrx/eval_agent/git_manager.py +215 -0
- evalrx/eval_agent/hypothesis.py +172 -0
- evalrx/eval_agent/label_quarantine.py +209 -0
- evalrx/eval_agent/legacy.py +530 -0
- evalrx/eval_agent/log_schema.py +497 -0
- evalrx/eval_agent/loop.py +2159 -0
- evalrx/eval_agent/loop_reports.py +116 -0
- evalrx/eval_agent/model_instrumentation.py +282 -0
- evalrx/eval_agent/narration.py +193 -0
- evalrx/eval_agent/nl_runner.py +460 -0
- evalrx/eval_agent/orchestrator.py +61 -0
- evalrx/eval_agent/preregister.py +93 -0
- evalrx/eval_agent/prompts/__init__.py +1 -0
- evalrx/eval_agent/prompts/agentic.py +46 -0
- evalrx/eval_agent/prompts/case_discovery.py +25 -0
- evalrx/eval_agent/prompts/diagnosis.py +125 -0
- evalrx/eval_agent/prompts/experiment_writer.py +265 -0
- evalrx/eval_agent/prompts/explore_step.py +37 -0
- evalrx/eval_agent/prompts/fix_agent.py +257 -0
- evalrx/eval_agent/prompts/hypothesis_tester.py +15 -0
- evalrx/eval_agent/prompts/nl_runner.py +38 -0
- evalrx/eval_agent/prompts/probe_agent.py +25 -0
- evalrx/eval_agent/prompts/probe_candidate_generator.py +14 -0
- evalrx/eval_agent/prompts/probe_generator.py +35 -0
- evalrx/eval_agent/prompts/whitebox_probe_generator.py +38 -0
- evalrx/eval_agent/report.py +58 -0
- evalrx/eval_agent/run_context.py +354 -0
- evalrx/eval_agent/run_log.schema.json +1215 -0
- evalrx/eval_agent/run_logger_v2.py +1764 -0
- evalrx/eval_agent/run_metadata.py +208 -0
- evalrx/eval_agent/stages/__init__.py +56 -0
- evalrx/eval_agent/stages/case_discovery.py +293 -0
- evalrx/eval_agent/stages/diagnosis.py +1017 -0
- evalrx/eval_agent/stages/experiment_writer.py +1634 -0
- evalrx/eval_agent/stages/fix_agent.py +3916 -0
- evalrx/eval_agent/stages/fix_internals.py +499 -0
- evalrx/eval_agent/stages/fix_pipeline.py +725 -0
- evalrx/eval_agent/stages/fix_tiers.py +187 -0
- evalrx/eval_agent/stages/fix_tools.py +1034 -0
- evalrx/eval_agent/stages/hypothesis_tester.py +1014 -0
- evalrx/eval_agent/stages/probe.py +439 -0
- evalrx/eval_agent/stages/probe_agent.py +1079 -0
- evalrx/eval_agent/stages/probe_candidate_generator.py +128 -0
- evalrx/eval_agent/stages/probe_generator.py +326 -0
- evalrx/eval_agent/stages/probe_search_agent.py +106 -0
- evalrx/eval_agent/stages/protocol.py +112 -0
- evalrx/eval_agent/stages/repair_catalog.py +273 -0
- evalrx/eval_agent/stages/surgery.py +524 -0
- evalrx/eval_agent/stages/whitebox_probe_generator.py +351 -0
- evalrx/eval_agent/store.py +231 -0
- evalrx/logging_utils.py +112 -0
- evalrx/models/__init__.py +161 -0
- evalrx/models/_discover.py +101 -0
- evalrx/models/agent.py +380 -0
- evalrx/models/backends/__init__.py +58 -0
- evalrx/models/backends/api.py +169 -0
- evalrx/models/backends/base.py +57 -0
- evalrx/models/backends/gemini_compat.py +579 -0
- evalrx/models/backends/hf_local.py +2074 -0
- evalrx/models/backends/openai_compat.py +301 -0
- evalrx/models/backends/vllm_offline.py +116 -0
- evalrx/models/base.py +24 -0
- evalrx/models/blackbox/__init__.py +4 -0
- evalrx/models/blackbox/agent.py +31 -0
- evalrx/models/blackbox/base.py +29 -0
- evalrx/models/blackbox/gemini.py +279 -0
- evalrx/models/blackbox/llm_api.py +17 -0
- evalrx/models/blackbox/vlm_api.py +17 -0
- evalrx/models/compose.py +66 -0
- evalrx/models/inference.py +88 -0
- evalrx/models/paper_methods/__init__.py +8 -0
- evalrx/models/paper_methods/aad.py +53 -0
- evalrx/models/paper_methods/ifcd.py +204 -0
- evalrx/models/paper_methods/pai.py +164 -0
- evalrx/models/paper_methods/tcd.py +202 -0
- evalrx/models/paper_methods/vcd.py +45 -0
- evalrx/models/paper_methods/vicrop.py +137 -0
- evalrx/models/toolcodec.py +143 -0
- evalrx/models/tools/__init__.py +20 -0
- evalrx/models/tools/perception.py +300 -0
- evalrx/models/tools/visual.py +174 -0
- evalrx/models/whitebox/__init__.py +26 -0
- evalrx/models/whitebox/agent.py +31 -0
- evalrx/models/whitebox/base.py +24 -0
- evalrx/models/whitebox/qwen.py +61 -0
- evalrx/models/whitebox/qwen2_5_omni.py +29 -0
- evalrx/models/whitebox/qwen2_audio.py +25 -0
- evalrx/models/whitebox/qwen_omni.py +53 -0
- evalrx/models/whitebox/qwen_vl.py +62 -0
- evalrx/observability/__init__.py +21 -0
- evalrx/observability/envelope.py +122 -0
- evalrx/observability/outbox.py +111 -0
- evalrx/observability/tracer.py +882 -0
- evalrx/reporting/__init__.py +28 -0
- evalrx/reporting/case_study.py +947 -0
- evalrx/reporting/compiler.py +587 -0
- evalrx/reporting/dynamic.py +1882 -0
- evalrx/reporting/html_report.py +2225 -0
- evalrx/reporting/langfuse_exporter.py +38 -0
- evalrx/reporting/langfuse_source.py +155 -0
- evalrx/reporting/model.py +151 -0
- evalrx/reporting/run_events.py +184 -0
- evalrx/reporting/server.py +557 -0
- evalrx/reporting/stages.py +58 -0
- evalrx/reporting/static_export.py +142 -0
- evalrx/reporting/web_dist/index.html +146 -0
- evalrx/specs.py +727 -0
- evalrx/stats/__init__.py +47 -0
- evalrx/stats/api.py +192 -0
- evalrx/stats/bootstrap.py +86 -0
- evalrx/stats/ebh.py +27 -0
- evalrx/stats/evalue.py +98 -0
- evalrx/stats/friedman.py +138 -0
- evalrx/stats/mcnemar.py +40 -0
- evalrx/stats/multiplicity.py +159 -0
- evalrx/stats/subset_sampling.py +55 -0
- evalrx/term_links.py +43 -0
- evalrx/viz/__init__.py +7 -0
- evalrx/viz/labels.py +77 -0
- evalrx/viz/prompts.py +39 -0
- evalrx/viz/renderer.py +590 -0
- evalrx/viz/schema.py +36 -0
- evalrx/viz/style.py +134 -0
- evalrx-0.1.2.dist-info/METADATA +532 -0
- evalrx-0.1.2.dist-info/RECORD +339 -0
- evalrx-0.1.2.dist-info/WHEEL +5 -0
- evalrx-0.1.2.dist-info/entry_points.txt +3 -0
- evalrx-0.1.2.dist-info/licenses/LICENSE +121 -0
- evalrx-0.1.2.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,1764 @@
|
|
|
1
|
+
"""RunLoggerV2 — the diagnose-loop logger.
|
|
2
|
+
|
|
3
|
+
The only logger: an earlier flat ``run_log.jsonl`` design this one replaced
|
|
4
|
+
coexisted with it for one migration window and has since been removed — see
|
|
5
|
+
``RUN_LOGGER_V2.md`` (next to this file) for the full design rationale,
|
|
6
|
+
layout, and trade-offs; the four rules that shaped it:
|
|
7
|
+
|
|
8
|
+
1. Few files. One JSON document per pipeline stage, not a scattered pile of
|
|
9
|
+
``prompts/*.txt`` + ``artifacts/*.json`` + ``experiments/*.py`` + ...
|
|
10
|
+
2. Same-type logging in one JSON. Every event of a given kind (all ``probe``
|
|
11
|
+
events, all ``model_call`` events, ...) lives in ONE array, in ONE file —
|
|
12
|
+
not one small file per call.
|
|
13
|
+
3. M1..M5 each get their own folder. A reader who only cares about M3 opens
|
|
14
|
+
exactly one folder.
|
|
15
|
+
4. Nothing but JSON, except real binary artifacts (images, audio, tensors). Code,
|
|
16
|
+
stdout, prompts, markdown summaries — all of that is now a STRING VALUE
|
|
17
|
+
inside the JSON, not a sibling ``.py``/``.txt``/``.md`` file.
|
|
18
|
+
|
|
19
|
+
Every ``log_*`` method keeps the same name, signature, and call-site
|
|
20
|
+
behavior for return values (e.g. ``log_probe``'s ``list[Path]``) the flat
|
|
21
|
+
JSONL logger it replaced had, so ``VLDiagnoseLoop(run_logger=...)`` /
|
|
22
|
+
``ProbeAgent(run_logger=...)`` / ``AutoDiagnoseLoop(run_logger=...)`` never
|
|
23
|
+
needed to change in ``loop.py``, ``probe_agent.py``, or any ``stages/*.py``
|
|
24
|
+
across the switch.
|
|
25
|
+
|
|
26
|
+
Scope, by design:
|
|
27
|
+
- RunContext integration uses an external ephemeral runtime tree. Generated
|
|
28
|
+
text/code is captured into stage JSON and the runtime tree is removed at
|
|
29
|
+
finalization instead of becoming a forest of trial files.
|
|
30
|
+
- No human-readable Markdown summaries (``record.md``, ``outcome.md``) —
|
|
31
|
+
the same information is in the JSON for a renderer to build one from.
|
|
32
|
+
- Verbose console narration is a plain one-line-per-event summary, not a
|
|
33
|
+
multi-line stage narration.
|
|
34
|
+
Native Langfuse/OpenTelemetry mirroring (:class:`DiagnosticTracer`) IS kept,
|
|
35
|
+
reused unchanged — it is orthogonal to file layout.
|
|
36
|
+
"""
|
|
37
|
+
|
|
38
|
+
from __future__ import annotations
|
|
39
|
+
|
|
40
|
+
import json
|
|
41
|
+
import os
|
|
42
|
+
import re
|
|
43
|
+
import tempfile
|
|
44
|
+
import threading
|
|
45
|
+
import uuid
|
|
46
|
+
import warnings
|
|
47
|
+
from datetime import datetime, timezone
|
|
48
|
+
from pathlib import Path
|
|
49
|
+
from typing import TYPE_CHECKING, Any
|
|
50
|
+
|
|
51
|
+
from evalrx.eval_agent.hypothesis import hypothesis_id
|
|
52
|
+
from evalrx.eval_agent.log_schema import RUN_LOG_SCHEMA_VERSION
|
|
53
|
+
|
|
54
|
+
if TYPE_CHECKING:
|
|
55
|
+
from evalrx.analysis.analysis_module import AnalysisReport
|
|
56
|
+
from evalrx.core.result import Result
|
|
57
|
+
from evalrx.eval_agent.hypothesis import Hypothesis
|
|
58
|
+
from evalrx.eval_agent.loop_reports import AutoDiagnoseReport
|
|
59
|
+
from evalrx.eval_agent.stages.diagnosis import DiagnosisResult
|
|
60
|
+
from evalrx.eval_agent.stages.surgery import InterventionResult
|
|
61
|
+
|
|
62
|
+
RUN_LOGGER_V2_VERSION = 1
|
|
63
|
+
|
|
64
|
+
_STAGES = ("M1", "M2", "M3", "M4", "M5")
|
|
65
|
+
|
|
66
|
+
#: A tag string not matching this falls back to _STAGE_ALIASES, then to the
|
|
67
|
+
#: run-level "unrouted" bucket (never silently dropped — see _resolve_stage).
|
|
68
|
+
_STAGE_RE = re.compile(r"m([1-5])", re.IGNORECASE)
|
|
69
|
+
|
|
70
|
+
#: Tags used somewhere in the codebase that carry no "m<N>" substring at all
|
|
71
|
+
#: (checked against every literal `module=`/`stage=` value passed to a
|
|
72
|
+
#: log_* method as of this writing — see RUN_LOGGER_V2.md's "routing" table).
|
|
73
|
+
_STAGE_ALIASES: dict[str, str] = {
|
|
74
|
+
"fix_pipeline": "M5",
|
|
75
|
+
"fix": "M5",
|
|
76
|
+
"explore": "M2",
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
#: Extensions treated as genuine binary media — the one thing rule 4 still
|
|
80
|
+
#: allows as a separate file. Everything else becomes a JSON string value.
|
|
81
|
+
_MEDIA_EXTS = frozenset({
|
|
82
|
+
".npy", ".png", ".jpg", ".jpeg", ".gif", ".webp", ".bmp",
|
|
83
|
+
".wav", ".mp3", ".flac", ".ogg", ".mp4", ".avi", ".mov",
|
|
84
|
+
})
|
|
85
|
+
|
|
86
|
+
#: Text/code file suffixes worth inlining from a sandbox workspace snapshot
|
|
87
|
+
#: (skip weights/binaries — those go through the media/artifact path instead).
|
|
88
|
+
_INLINE_SUFFIXES = frozenset(
|
|
89
|
+
{".py", ".json", ".jsonl", ".md", ".txt", ".yaml", ".yml", ".csv", ".log", ".toml"}
|
|
90
|
+
)
|
|
91
|
+
_INLINE_MAX_BYTES = 2_000_000 # skip (note-only) any single file larger than this
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _resolve_stage(tag: "str | None") -> "str | None":
|
|
95
|
+
""""m1_probe" / "M4_SURGERY" / "codegen_m2_stats" / "fix_pipeline" -> "M1".."M5".
|
|
96
|
+
|
|
97
|
+
Returns ``None`` when *tag* matches nothing — the caller must not drop
|
|
98
|
+
the event in that case; route it to the run-level "unrouted" bucket
|
|
99
|
+
instead (see ``RunLoggerV2._route``). A tag is never assumed unroutable
|
|
100
|
+
without trying both the regex AND the alias table.
|
|
101
|
+
"""
|
|
102
|
+
if not tag:
|
|
103
|
+
return None
|
|
104
|
+
m = _STAGE_RE.search(tag)
|
|
105
|
+
if m:
|
|
106
|
+
return f"M{m.group(1)}"
|
|
107
|
+
return _STAGE_ALIASES.get(tag.strip().lower())
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def _atomic_write_json(path: Path, obj: Any) -> None:
|
|
111
|
+
"""Write *obj* as JSON to *path* such that a reader never sees a partial file.
|
|
112
|
+
|
|
113
|
+
Writes to a sibling temp file first, then ``os.replace`` (atomic on the
|
|
114
|
+
same filesystem) — a crash mid-write leaves the OLD complete file in
|
|
115
|
+
place, never a truncated one. This runs on every single logged event
|
|
116
|
+
(see the design doc's "durability" section for the cost trade-off that
|
|
117
|
+
was chosen deliberately here, not overlooked).
|
|
118
|
+
"""
|
|
119
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
120
|
+
tmp = path.with_name(f".{path.name}.tmp{os.getpid()}")
|
|
121
|
+
tmp.write_text(json.dumps(obj, indent=2, default=str, ensure_ascii=False), encoding="utf-8")
|
|
122
|
+
os.replace(tmp, path)
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def _inline_workspace(
|
|
126
|
+
workdir: "str | Path", media_dir: Path, *, run_dir: "Path | None" = None,
|
|
127
|
+
max_bytes: "int | None" = _INLINE_MAX_BYTES,
|
|
128
|
+
preserve_all: bool = False,
|
|
129
|
+
) -> "dict[str, Any] | None":
|
|
130
|
+
"""Read a sandbox working directory into a JSON-safe dict, inlining text.
|
|
131
|
+
|
|
132
|
+
Returns ``{"files": {relative_path: content_or_note}, "media": [rel_paths],
|
|
133
|
+
"skipped": n}`` or ``None`` when *workdir* does not exist. Text-like files
|
|
134
|
+
(see ``_INLINE_SUFFIXES``) are read and inlined verbatim; recognised media
|
|
135
|
+
extensions are COPIED into *media_dir* (a real binary artifact, rule 4's
|
|
136
|
+
one exception) with a path reference left in ``"media"``; anything else is
|
|
137
|
+
skipped with a one-line note so its existence is still visible.
|
|
138
|
+
"""
|
|
139
|
+
import hashlib
|
|
140
|
+
import shutil
|
|
141
|
+
|
|
142
|
+
src = Path(workdir)
|
|
143
|
+
if not src.exists() or not src.is_dir():
|
|
144
|
+
return None
|
|
145
|
+
files: dict[str, Any] = {}
|
|
146
|
+
media: list[str] = []
|
|
147
|
+
media_files: dict[str, str] = {}
|
|
148
|
+
skipped = 0
|
|
149
|
+
for f in sorted(src.rglob("*")):
|
|
150
|
+
if not f.is_file():
|
|
151
|
+
continue
|
|
152
|
+
rel = str(f.relative_to(src))
|
|
153
|
+
suffix = f.suffix.lower()
|
|
154
|
+
inline_content = None
|
|
155
|
+
binary = suffix in _MEDIA_EXTS
|
|
156
|
+
if preserve_all and suffix not in _INLINE_SUFFIXES and not binary:
|
|
157
|
+
try:
|
|
158
|
+
inline_content = f.read_text(encoding="utf-8")
|
|
159
|
+
if "\x00" in inline_content:
|
|
160
|
+
binary = True
|
|
161
|
+
except UnicodeError:
|
|
162
|
+
binary = True
|
|
163
|
+
if binary:
|
|
164
|
+
media_dir.mkdir(parents=True, exist_ok=True)
|
|
165
|
+
# Path + content identify a snapshot: trials cannot collide, and
|
|
166
|
+
# later writes to the same source cannot replace earlier evidence.
|
|
167
|
+
hasher = hashlib.sha256(str(f.resolve()).encode("utf-8"))
|
|
168
|
+
with f.open("rb") as stream:
|
|
169
|
+
for chunk in iter(lambda: stream.read(1024 * 1024), b""):
|
|
170
|
+
hasher.update(chunk)
|
|
171
|
+
digest = hasher.hexdigest()[:12]
|
|
172
|
+
dest = media_dir / f"{f.stem}_{digest}{suffix}"
|
|
173
|
+
try:
|
|
174
|
+
shutil.copy2(f, dest)
|
|
175
|
+
try:
|
|
176
|
+
media.append(str(dest.relative_to(run_dir)) if run_dir else str(dest))
|
|
177
|
+
except ValueError:
|
|
178
|
+
media.append(str(dest))
|
|
179
|
+
media_files[rel] = media[-1]
|
|
180
|
+
except Exception: # noqa: BLE001
|
|
181
|
+
if preserve_all:
|
|
182
|
+
raise # do not let finalization delete unarchived evidence
|
|
183
|
+
skipped += 1
|
|
184
|
+
continue
|
|
185
|
+
if suffix not in _INLINE_SUFFIXES and inline_content is None:
|
|
186
|
+
files[rel] = f"<skipped: {suffix or 'no extension'}, not a recognised text type>"
|
|
187
|
+
skipped += 1
|
|
188
|
+
continue
|
|
189
|
+
try:
|
|
190
|
+
if max_bytes is not None and f.stat().st_size > max_bytes:
|
|
191
|
+
files[rel] = f"<skipped: {f.stat().st_size} bytes, over the inline cap>"
|
|
192
|
+
skipped += 1
|
|
193
|
+
continue
|
|
194
|
+
files[rel] = inline_content if inline_content is not None else f.read_text(encoding="utf-8", errors="replace")
|
|
195
|
+
except Exception as exc: # noqa: BLE001
|
|
196
|
+
if preserve_all:
|
|
197
|
+
raise
|
|
198
|
+
files[rel] = f"<could not read: {exc}>"
|
|
199
|
+
skipped += 1
|
|
200
|
+
return {"files": files, "media": media, "media_files": media_files, "skipped": skipped}
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
# Pure, stateless content-shaping helpers with no file-writing side effects.
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
def _case_snapshot(case: Any) -> "dict[str, Any]":
|
|
207
|
+
"""Make a small, renderer-safe baseline record for an evidence example."""
|
|
208
|
+
if hasattr(case, "to_dict"):
|
|
209
|
+
value = case.to_dict()
|
|
210
|
+
elif isinstance(case, dict):
|
|
211
|
+
value = dict(case)
|
|
212
|
+
else:
|
|
213
|
+
value = {"id": str(getattr(case, "id", ""))}
|
|
214
|
+
inputs = value.get("inputs") if isinstance(value.get("inputs"), dict) else {}
|
|
215
|
+
return {
|
|
216
|
+
"id": str(value.get("id") or value.get("case_id") or ""),
|
|
217
|
+
"input": inputs.get("prompt") or value.get("prompt") or value.get("instruction") or "",
|
|
218
|
+
"baseline_output": value.get("observed", value.get("output")),
|
|
219
|
+
"expected": value.get("expected"),
|
|
220
|
+
"outcome": value.get("label") or value.get("status") or "unknown",
|
|
221
|
+
}
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
def _iter_cases(cases: Any) -> "list[Any]":
|
|
225
|
+
"""Accept CaseBatch, a plain sequence, or a generator without assumptions."""
|
|
226
|
+
if cases is None:
|
|
227
|
+
return []
|
|
228
|
+
value = getattr(cases, "cases", cases)
|
|
229
|
+
try:
|
|
230
|
+
return list(value)
|
|
231
|
+
except TypeError:
|
|
232
|
+
return []
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
def _probe_examples(results: "dict[str, Any]", cases: Any) -> "list[dict[str, Any]]":
|
|
236
|
+
"""Persist two real, bounded M1 walkthroughs beside aggregate findings.
|
|
237
|
+
|
|
238
|
+
A probe only becomes a before/after comparison when its analyzer explicitly
|
|
239
|
+
records both outputs. Otherwise this records an honest *baseline case +
|
|
240
|
+
check result* example; downstream UI must not call it an intervention.
|
|
241
|
+
"""
|
|
242
|
+
snapshots: dict[str, dict[str, Any]] = {}
|
|
243
|
+
for case in _iter_cases(cases):
|
|
244
|
+
snapshot = _case_snapshot(case)
|
|
245
|
+
if snapshot["id"]:
|
|
246
|
+
snapshots[snapshot["id"]] = snapshot
|
|
247
|
+
output: list[dict[str, Any]] = []
|
|
248
|
+
used: set[str] = set()
|
|
249
|
+
for name, result in results.items():
|
|
250
|
+
findings = getattr(result, "findings", {}) or {}
|
|
251
|
+
rows = findings.get("per_case") or []
|
|
252
|
+
if not isinstance(rows, list):
|
|
253
|
+
continue
|
|
254
|
+
rows = sorted(
|
|
255
|
+
(row for row in rows if isinstance(row, dict)),
|
|
256
|
+
key=lambda row: 0 if str(snapshots.get(str(row.get("sample_id") or row.get("case_id") or ""), {}).get("outcome", "")).lower() == "fail" else 1,
|
|
257
|
+
)
|
|
258
|
+
for row in rows:
|
|
259
|
+
case_id = str(row.get("sample_id") or row.get("case_id") or "")
|
|
260
|
+
snapshot = snapshots.get(case_id)
|
|
261
|
+
if not snapshot or case_id in used:
|
|
262
|
+
continue
|
|
263
|
+
checked = {
|
|
264
|
+
str(key).replace("_", " "): value for key, value in row.items()
|
|
265
|
+
if key not in {"sample_id", "case_id"} and isinstance(value, (str, int, float, bool))
|
|
266
|
+
}
|
|
267
|
+
if not checked:
|
|
268
|
+
continue
|
|
269
|
+
used.add(case_id)
|
|
270
|
+
output.append({
|
|
271
|
+
"id": f"m1-{name}-{case_id}", "kind": "case_measurement", "case_id": case_id,
|
|
272
|
+
"probe_title": str(name).replace("_", " ").title(),
|
|
273
|
+
**snapshot, "check_result": checked,
|
|
274
|
+
"plain_reading": "This one case illustrates the recorded check. The aggregate M1 result uses all measured cases.",
|
|
275
|
+
"evidence_scope": "one recorded case within M1",
|
|
276
|
+
})
|
|
277
|
+
break
|
|
278
|
+
if len(output) >= 2:
|
|
279
|
+
break
|
|
280
|
+
return output
|
|
281
|
+
|
|
282
|
+
|
|
283
|
+
def _artifact_to_numpy(artifact: Any) -> "Any | None":
|
|
284
|
+
"""Convert *artifact* to a numpy array, or return None if not possible.
|
|
285
|
+
|
|
286
|
+
Handles: torch.Tensor, list[torch.Tensor] (e.g. per-layer attentions),
|
|
287
|
+
and numpy arrays. A list of tensors is stacked along a new first axis so
|
|
288
|
+
that ``attentions`` (list of ``(heads, seq, seq)``) becomes
|
|
289
|
+
``(layers, heads, seq, seq)`` — a single array that retains all the data.
|
|
290
|
+
"""
|
|
291
|
+
try:
|
|
292
|
+
import numpy as np
|
|
293
|
+
except ImportError:
|
|
294
|
+
return None
|
|
295
|
+
|
|
296
|
+
if hasattr(artifact, "detach"): # torch.Tensor
|
|
297
|
+
return artifact.detach().cpu().float().numpy()
|
|
298
|
+
if isinstance(artifact, np.ndarray):
|
|
299
|
+
return artifact
|
|
300
|
+
if isinstance(artifact, list) and artifact and hasattr(artifact[0], "detach"):
|
|
301
|
+
try:
|
|
302
|
+
import torch
|
|
303
|
+
return torch.stack(artifact).detach().cpu().float().numpy()
|
|
304
|
+
except Exception: # noqa: BLE001
|
|
305
|
+
return None
|
|
306
|
+
return None
|
|
307
|
+
|
|
308
|
+
|
|
309
|
+
def _save_artifact_figure(artifact_dir: Path, stem: str, arr: Any) -> None:
|
|
310
|
+
"""Save a matplotlib figure of *arr* when the shape and stem are recognised.
|
|
311
|
+
|
|
312
|
+
Dispatch table (first match wins):
|
|
313
|
+
- 4-D + ``attn`` in stem → mean over (layers, heads) → 2-D heatmap
|
|
314
|
+
- 3-D + ``attn`` in stem → mean over heads → 2-D heatmap
|
|
315
|
+
- 2-D + heatmap keyword → direct heatmap (viridis)
|
|
316
|
+
- 1-D + curve keyword → line plot
|
|
317
|
+
Skips silently when matplotlib is unavailable or the shape is unrecognised.
|
|
318
|
+
"""
|
|
319
|
+
try:
|
|
320
|
+
import matplotlib.pyplot as plt
|
|
321
|
+
plt.ioff()
|
|
322
|
+
except ImportError:
|
|
323
|
+
return
|
|
324
|
+
|
|
325
|
+
key = stem.lower()
|
|
326
|
+
# Skip logit arrays — (seq, vocab) shape is too large for a useful figure
|
|
327
|
+
if "logit" in key:
|
|
328
|
+
return
|
|
329
|
+
|
|
330
|
+
_is_attn = any(k in key for k in ("attn", "attention"))
|
|
331
|
+
|
|
332
|
+
fig = None
|
|
333
|
+
try:
|
|
334
|
+
ndim = arr.ndim
|
|
335
|
+
if ndim == 4 and _is_attn:
|
|
336
|
+
mat = arr.mean(axis=(0, 1)) # (layers, heads, seq, seq) → (seq, seq)
|
|
337
|
+
n_layers, n_heads = arr.shape[0], arr.shape[1]
|
|
338
|
+
fig, ax = plt.subplots(figsize=(8, 7))
|
|
339
|
+
im = ax.imshow(mat, cmap="viridis", aspect="auto", vmin=0)
|
|
340
|
+
ax.set_title(f"{stem} (mean over {n_layers}L × {n_heads}H)")
|
|
341
|
+
plt.colorbar(im, ax=ax, fraction=0.046, pad=0.04)
|
|
342
|
+
plt.tight_layout()
|
|
343
|
+
elif ndim == 3 and _is_attn:
|
|
344
|
+
mat = arr.mean(axis=0) # (heads, seq, seq) → (seq, seq)
|
|
345
|
+
n_heads = arr.shape[0]
|
|
346
|
+
fig, ax = plt.subplots(figsize=(8, 7))
|
|
347
|
+
im = ax.imshow(mat, cmap="viridis", aspect="auto", vmin=0)
|
|
348
|
+
ax.set_title(f"{stem} (mean over {n_heads} heads)")
|
|
349
|
+
plt.colorbar(im, ax=ax, fraction=0.046, pad=0.04)
|
|
350
|
+
plt.tight_layout()
|
|
351
|
+
elif ndim == 2 and "diff" in key:
|
|
352
|
+
# Signed difference map (e.g. FAIL-mean minus PASS-mean attention):
|
|
353
|
+
# diverging colormap with symmetric limits so the sign is readable.
|
|
354
|
+
bound = float(max(abs(arr.min()), abs(arr.max()))) or 1.0
|
|
355
|
+
fig, ax = plt.subplots(figsize=(8, 7))
|
|
356
|
+
im = ax.imshow(arr, cmap="coolwarm", aspect="auto", vmin=-bound, vmax=bound)
|
|
357
|
+
ax.set_title(stem)
|
|
358
|
+
plt.colorbar(im, ax=ax, fraction=0.046, pad=0.04)
|
|
359
|
+
plt.tight_layout()
|
|
360
|
+
elif ndim == 2 and (_is_attn or any(k in key for k in ("rollout", "spatial", "map"))):
|
|
361
|
+
fig, ax = plt.subplots(figsize=(8, 7))
|
|
362
|
+
im = ax.imshow(arr, cmap="viridis", aspect="auto")
|
|
363
|
+
ax.set_title(stem)
|
|
364
|
+
plt.colorbar(im, ax=ax, fraction=0.046, pad=0.04)
|
|
365
|
+
plt.tight_layout()
|
|
366
|
+
elif ndim == 1 and any(k in key for k in ("entropy", "score", "prob", "weight", "rollout")):
|
|
367
|
+
fig, ax = plt.subplots(figsize=(8, 3))
|
|
368
|
+
ax.plot(arr)
|
|
369
|
+
ax.set_xlabel("position")
|
|
370
|
+
ax.set_ylabel(stem)
|
|
371
|
+
ax.set_title(stem)
|
|
372
|
+
plt.tight_layout()
|
|
373
|
+
|
|
374
|
+
if fig is not None:
|
|
375
|
+
fig.savefig(artifact_dir / f"{stem}.png", dpi=100, bbox_inches="tight")
|
|
376
|
+
except Exception: # noqa: BLE001
|
|
377
|
+
pass
|
|
378
|
+
finally:
|
|
379
|
+
if fig is not None:
|
|
380
|
+
plt.close(fig)
|
|
381
|
+
|
|
382
|
+
|
|
383
|
+
class _V2JsonFormatter:
|
|
384
|
+
"""Renders one plain one-line console summary per event, for ``verbose=True``.
|
|
385
|
+
|
|
386
|
+
Intentionally simple — file layout was this module's ask, not console UX;
|
|
387
|
+
see the design doc.
|
|
388
|
+
"""
|
|
389
|
+
|
|
390
|
+
@staticmethod
|
|
391
|
+
def line(stage: "str | None", event: str, payload: "dict[str, Any]") -> str:
|
|
392
|
+
where = f"[{stage}]" if stage else "[run]"
|
|
393
|
+
cycle = payload.get("cycle")
|
|
394
|
+
tail = f" cycle={cycle}" if cycle is not None else ""
|
|
395
|
+
return f"{where} {event}{tail}"
|
|
396
|
+
|
|
397
|
+
|
|
398
|
+
class RunLoggerV2:
|
|
399
|
+
"""A tidy, from-scratch M1..M5 logger. See the module docstring + design doc.
|
|
400
|
+
|
|
401
|
+
Args:
|
|
402
|
+
run_dir: Directory to write into. Created if missing. Defaults to
|
|
403
|
+
``runs_v2/<YYYYMMDD_HHMMSS>/`` relative to cwd.
|
|
404
|
+
verbose: Print a one-line raw summary of every event to stdout
|
|
405
|
+
(``[M1] probe cycle=0``). Ignored when *narrate* is set.
|
|
406
|
+
narrate: Print live, aligned M1-M5 narration instead — the same
|
|
407
|
+
visual style as ``evalrx explore``'s terminal output (see
|
|
408
|
+
:mod:`evalrx.eval_agent.narration.LoopNarrator`), built
|
|
409
|
+
from real per-stage counts instead of a raw event dump.
|
|
410
|
+
trace_id: Ties every event to one Langfuse trace; auto-generated if
|
|
411
|
+
omitted.
|
|
412
|
+
observability_mode: Forwarded to :class:`DiagnosticTracer` unchanged.
|
|
413
|
+
|
|
414
|
+
Layout written under *run_dir*::
|
|
415
|
+
|
|
416
|
+
run.json run-wide: run_start, cases, report_published,
|
|
417
|
+
loop_end, agent_decisions, agent_tool_calls,
|
|
418
|
+
unrouted (see _resolve_stage)
|
|
419
|
+
M1/log.json probe, model_calls, tool_codegen, tool_registry,
|
|
420
|
+
stage_skipped — all M1-tagged events
|
|
421
|
+
M2/log.json analysis, explore, ...
|
|
422
|
+
M3/log.json diagnosis, ...
|
|
423
|
+
M4/log.json surgery (hypothesis-verification kind), ...
|
|
424
|
+
M5/log.json surgery (intervention kind), experiment, fix, ...
|
|
425
|
+
M*/artifacts/ binary media for that stage only (rule 4's exception)
|
|
426
|
+
media/ case-level baseline media (images/audio referenced
|
|
427
|
+
by FailureCase.inputs)
|
|
428
|
+
artifacts/ run-global named JSON (save_artifact_json)
|
|
429
|
+
"""
|
|
430
|
+
|
|
431
|
+
def __init__(
|
|
432
|
+
self,
|
|
433
|
+
run_dir: "str | Path | None" = None,
|
|
434
|
+
*,
|
|
435
|
+
verbose: bool = False,
|
|
436
|
+
narrate: bool = False,
|
|
437
|
+
trace_id: "str | None" = None,
|
|
438
|
+
observability_mode: "str | None" = None,
|
|
439
|
+
context: "Any | None" = None,
|
|
440
|
+
) -> None:
|
|
441
|
+
if run_dir is None:
|
|
442
|
+
run_dir = Path("runs_v2") / datetime.now().strftime("%Y%m%d_%H%M%S")
|
|
443
|
+
self.run_dir = Path(run_dir)
|
|
444
|
+
self.run_dir.mkdir(parents=True, exist_ok=True)
|
|
445
|
+
self.run_json_path = self.run_dir / "run.json"
|
|
446
|
+
|
|
447
|
+
self.trace_id: str = trace_id or str(uuid.uuid4())
|
|
448
|
+
self.current_cycle: int = -1
|
|
449
|
+
# Live M1-M5 terminal narration (see evalrx.eval_agent.narration) --
|
|
450
|
+
# opt-in, takes over from the plain `verbose` one-liner below rather
|
|
451
|
+
# than stacking with it, so a run never prints each event twice.
|
|
452
|
+
self._narrator = None
|
|
453
|
+
if narrate:
|
|
454
|
+
from evalrx.eval_agent.narration import LoopNarrator
|
|
455
|
+
|
|
456
|
+
self._narrator = LoopNarrator()
|
|
457
|
+
self.verbose = verbose
|
|
458
|
+
self._context = context
|
|
459
|
+
self._closed = False
|
|
460
|
+
# Producers use this capability flag to keep text/code in log events
|
|
461
|
+
# instead of writing sibling files into trial directories.
|
|
462
|
+
self.inline_text_artifacts = True
|
|
463
|
+
self.preserve_full_model_io = True
|
|
464
|
+
|
|
465
|
+
# One in-memory doc per stage + one run-level doc. Every log_* method
|
|
466
|
+
# appends to the relevant bucket(s), then atomically rewrites exactly
|
|
467
|
+
# the doc(s) it touched — see _atomic_write_json.
|
|
468
|
+
self._lock = threading.RLock()
|
|
469
|
+
self._event_seq = 0
|
|
470
|
+
self._validate_events = bool(os.environ.get("EVALRX_VALIDATE_LOG"))
|
|
471
|
+
self._event_validator = None
|
|
472
|
+
self._run_doc: dict[str, Any] = {
|
|
473
|
+
"trace_id": self.trace_id,
|
|
474
|
+
"run_start": None,
|
|
475
|
+
"cases": [],
|
|
476
|
+
"report_published": [],
|
|
477
|
+
"diagnose_reports": [],
|
|
478
|
+
"manifest": None,
|
|
479
|
+
"loop_end": [],
|
|
480
|
+
"agent_decisions": [],
|
|
481
|
+
"agent_tool_calls": [],
|
|
482
|
+
"unrouted": [],
|
|
483
|
+
}
|
|
484
|
+
self._stage_docs: dict[str, dict[str, Any]] = {s: {} for s in _STAGES}
|
|
485
|
+
self._logged_case_ids: set[str] = set()
|
|
486
|
+
self._model_call_seq = 0
|
|
487
|
+
self._codegen_seq = 0
|
|
488
|
+
# See log_model_call / log_probe: calls are recorded immediately into
|
|
489
|
+
# M1's doc AND buffered here so log_probe can replay them into
|
|
490
|
+
# Langfuse nested under the right probe span once it exists.
|
|
491
|
+
self._pending_model_calls: "dict[int, list[dict[str, Any]]]" = {}
|
|
492
|
+
|
|
493
|
+
from evalrx.observability.tracer import DiagnosticTracer
|
|
494
|
+
# The SQLite delivery queue is runtime state, not part of the tidy run
|
|
495
|
+
# artifact. Keep it outside the run tree; langfuse_trace.json remains
|
|
496
|
+
# the durable, portable JSON trace bundled with the run.
|
|
497
|
+
outbox_dir = Path(tempfile.gettempdir()) / "evalrx-v2-outbox"
|
|
498
|
+
self._outbox_path = outbox_dir / f"{self.trace_id}.sqlite3"
|
|
499
|
+
self.tracer = DiagnosticTracer(
|
|
500
|
+
run_dir=self.run_dir, mode=observability_mode, auto_sync=True,
|
|
501
|
+
outbox_path=self._outbox_path,
|
|
502
|
+
)
|
|
503
|
+
self.tracer.trace_id = self.trace_id
|
|
504
|
+
|
|
505
|
+
self._flush_run()
|
|
506
|
+
|
|
507
|
+
# ------------------------------------------------------------------
|
|
508
|
+
# Internal: doc access, routing, durability
|
|
509
|
+
# ------------------------------------------------------------------
|
|
510
|
+
|
|
511
|
+
def _stage_dir(self, stage: str) -> Path:
|
|
512
|
+
d = self.run_dir / stage
|
|
513
|
+
d.mkdir(parents=True, exist_ok=True)
|
|
514
|
+
return d
|
|
515
|
+
|
|
516
|
+
def _stage_artifacts_dir(self, stage: str) -> Path:
|
|
517
|
+
d = self._stage_dir(stage) / "artifacts"
|
|
518
|
+
d.mkdir(parents=True, exist_ok=True)
|
|
519
|
+
return d
|
|
520
|
+
|
|
521
|
+
def _flush_run(self) -> None:
|
|
522
|
+
_atomic_write_json(self.run_json_path, self._run_doc)
|
|
523
|
+
|
|
524
|
+
def _flush_stage(self, stage: str) -> None:
|
|
525
|
+
_atomic_write_json(self._stage_dir(stage) / "log.json", self._stage_docs[stage])
|
|
526
|
+
|
|
527
|
+
def _bucket(self, stage: str, key: str) -> list:
|
|
528
|
+
return self._stage_docs[stage].setdefault(key, [])
|
|
529
|
+
|
|
530
|
+
def _stamp_event(self, key: str, record: dict[str, Any], stage: str) -> None:
|
|
531
|
+
"""Assign durable event identity under ``_lock``; no telemetry side effects."""
|
|
532
|
+
event = {
|
|
533
|
+
"cases": "case_record", "model_calls": "model_call",
|
|
534
|
+
"diagnose_reports": "diagnose_report", "agent_decisions": "agent_decision",
|
|
535
|
+
"agent_tool_calls": "agent_tool",
|
|
536
|
+
}.get(key, key)
|
|
537
|
+
self._event_seq += 1
|
|
538
|
+
cycle = record.get("cycle", -1)
|
|
539
|
+
span = {
|
|
540
|
+
"run_start": "run_start", "probe": f"c{cycle}.m1",
|
|
541
|
+
"analysis": f"c{cycle}.m2", "diagnosis": f"c{cycle}.m3",
|
|
542
|
+
"explore": f"c{cycle}.explore", "surgery": f"c{cycle}.{stage.lower()}",
|
|
543
|
+
"fix": "fix", "agent_decision": f"s{record.get('step')}.decision",
|
|
544
|
+
"agent_tool": f"s{record.get('step')}.tool",
|
|
545
|
+
"stage_skipped": f"{stage.lower()}.skipped",
|
|
546
|
+
}.get(event, f"{stage.lower()}.{event}.{self._event_seq}")
|
|
547
|
+
record.update(event=event, schema_version=RUN_LOG_SCHEMA_VERSION,
|
|
548
|
+
trace_id=self.trace_id, event_seq=self._event_seq,
|
|
549
|
+
stage=stage, span_id=span)
|
|
550
|
+
record.setdefault("ts", self._ts())
|
|
551
|
+
if self._validate_events:
|
|
552
|
+
try:
|
|
553
|
+
from evalrx.eval_agent.log_schema import _validator, build_schema
|
|
554
|
+
|
|
555
|
+
if self._event_validator is None:
|
|
556
|
+
self._event_validator = _validator(build_schema())
|
|
557
|
+
self._event_validator.validate(record)
|
|
558
|
+
except ImportError:
|
|
559
|
+
pass
|
|
560
|
+
except Exception as exc: # warn-only
|
|
561
|
+
warnings.warn(f"RunLoggerV2: event {event!r} violates log schema: {exc}")
|
|
562
|
+
|
|
563
|
+
def _append_stage(self, tag: "str | None", key: str, record: "dict[str, Any]") -> str:
|
|
564
|
+
"""Route *record* by *tag* into the right stage bucket; flush; return the stage."""
|
|
565
|
+
stage = _resolve_stage(tag)
|
|
566
|
+
with self._lock:
|
|
567
|
+
if stage is None:
|
|
568
|
+
warnings.warn(
|
|
569
|
+
f"RunLoggerV2: could not route event {key!r} (tag={tag!r}) to a "
|
|
570
|
+
"stage — filed under run.json['unrouted'] instead of being lost.",
|
|
571
|
+
stacklevel=3,
|
|
572
|
+
)
|
|
573
|
+
record = {"key": key, "tag": tag, **record}
|
|
574
|
+
self._stamp_event("unrouted", record, "RUN")
|
|
575
|
+
self._run_doc["unrouted"].append(record)
|
|
576
|
+
self._flush_run()
|
|
577
|
+
return "unrouted"
|
|
578
|
+
self._stamp_event(key, record, stage)
|
|
579
|
+
self._bucket(stage, key).append(record)
|
|
580
|
+
self._flush_stage(stage)
|
|
581
|
+
if self._narrator is not None:
|
|
582
|
+
self._narrator.on_event(stage, key, record)
|
|
583
|
+
elif self.verbose:
|
|
584
|
+
print(_V2JsonFormatter.line(stage, key, record))
|
|
585
|
+
return stage
|
|
586
|
+
|
|
587
|
+
def _append_run(self, key: str, record: "dict[str, Any]") -> None:
|
|
588
|
+
"""Append *record* to the (always list-valued) run.json bucket *key*.
|
|
589
|
+
|
|
590
|
+
``run_start`` is the one run.json field that isn't a list — it's set
|
|
591
|
+
directly by ``log_run_start``, never through here.
|
|
592
|
+
"""
|
|
593
|
+
with self._lock:
|
|
594
|
+
self._stamp_event(key, record, "RUN")
|
|
595
|
+
self._run_doc[key].append(record)
|
|
596
|
+
self._flush_run()
|
|
597
|
+
if self._narrator is not None:
|
|
598
|
+
self._narrator.on_run_event(key, record)
|
|
599
|
+
elif self.verbose:
|
|
600
|
+
print(_V2JsonFormatter.line(None, key, record))
|
|
601
|
+
|
|
602
|
+
@property
|
|
603
|
+
def managed_json_paths(self) -> "tuple[Path, ...]":
|
|
604
|
+
"""Atomic JSON documents that may be rewritten while quarantine runs."""
|
|
605
|
+
return (self.run_json_path, *(self.run_dir / s / "log.json" for s in _STAGES))
|
|
606
|
+
|
|
607
|
+
@staticmethod
|
|
608
|
+
def _ts() -> str:
|
|
609
|
+
return datetime.now(timezone.utc).isoformat(timespec="microseconds")
|
|
610
|
+
|
|
611
|
+
def _save_media(self, stage: str, stem: str, artifact: Any) -> "str | None":
|
|
612
|
+
"""Save a numeric artifact (tensor/array) + a rendered figure, if any.
|
|
613
|
+
|
|
614
|
+
Writes under this stage's ``artifacts/`` dir. Returns the ``.npy``
|
|
615
|
+
path (run-relative) or ``None`` when *artifact* isn't a recognised
|
|
616
|
+
numeric type.
|
|
617
|
+
"""
|
|
618
|
+
try:
|
|
619
|
+
import numpy as np
|
|
620
|
+
|
|
621
|
+
arr = _artifact_to_numpy(artifact)
|
|
622
|
+
if arr is not None:
|
|
623
|
+
art_dir = self._stage_artifacts_dir(stage)
|
|
624
|
+
path = art_dir / f"{stem}.npy"
|
|
625
|
+
np.save(path, arr)
|
|
626
|
+
_save_artifact_figure(art_dir, stem, arr)
|
|
627
|
+
return str(path.relative_to(self.run_dir))
|
|
628
|
+
return None
|
|
629
|
+
except Exception as exc: # noqa: BLE001
|
|
630
|
+
warnings.warn(f"RunLoggerV2: could not save artifact {stem!r}: {exc}")
|
|
631
|
+
return None
|
|
632
|
+
|
|
633
|
+
def _save_case_media(self, path: Path) -> "str | None":
|
|
634
|
+
"""Copy external case media into ``media/`` (content-hash-deduped); return rel path."""
|
|
635
|
+
import hashlib
|
|
636
|
+
import shutil
|
|
637
|
+
|
|
638
|
+
media_dir = self.run_dir / "media"
|
|
639
|
+
media_dir.mkdir(parents=True, exist_ok=True)
|
|
640
|
+
digest = hashlib.sha256()
|
|
641
|
+
try:
|
|
642
|
+
with path.open("rb") as handle:
|
|
643
|
+
for chunk in iter(lambda: handle.read(1024 * 1024), b""):
|
|
644
|
+
digest.update(chunk)
|
|
645
|
+
except OSError:
|
|
646
|
+
return None
|
|
647
|
+
copied = media_dir / f"{digest.hexdigest()[:16]}_{path.name}"
|
|
648
|
+
if not copied.exists():
|
|
649
|
+
shutil.copy2(path, copied)
|
|
650
|
+
return str(copied.relative_to(self.run_dir))
|
|
651
|
+
|
|
652
|
+
def _portable_path(self, value: "str | Path") -> str:
|
|
653
|
+
path = Path(value)
|
|
654
|
+
try:
|
|
655
|
+
return str(path.resolve().relative_to(self.run_dir.resolve()))
|
|
656
|
+
except (OSError, ValueError):
|
|
657
|
+
return str(value)
|
|
658
|
+
|
|
659
|
+
# ------------------------------------------------------------------
|
|
660
|
+
# Run provenance
|
|
661
|
+
# ------------------------------------------------------------------
|
|
662
|
+
|
|
663
|
+
def log_run_start(self, config: "dict[str, Any] | None" = None) -> None:
|
|
664
|
+
import platform
|
|
665
|
+
|
|
666
|
+
entry: dict[str, Any] = {"ts": self._ts(), "trace_id": self.trace_id}
|
|
667
|
+
if config:
|
|
668
|
+
entry.update(config)
|
|
669
|
+
entry.setdefault("python_version", platform.python_version())
|
|
670
|
+
try:
|
|
671
|
+
from evalrx import __version__ as _ver # type: ignore
|
|
672
|
+
entry.setdefault("evalrx_version", _ver)
|
|
673
|
+
except Exception: # noqa: BLE001
|
|
674
|
+
pass
|
|
675
|
+
commit = self._git_commit()
|
|
676
|
+
if commit:
|
|
677
|
+
entry.setdefault("git_commit", commit)
|
|
678
|
+
with self._lock:
|
|
679
|
+
self._stamp_event("run_start", entry, "RUN")
|
|
680
|
+
self._run_doc["run_start"] = entry
|
|
681
|
+
self._flush_run()
|
|
682
|
+
if self._narrator is not None:
|
|
683
|
+
self._narrator.on_run_start(entry)
|
|
684
|
+
elif self.verbose:
|
|
685
|
+
print(_V2JsonFormatter.line(None, "run_start", entry))
|
|
686
|
+
|
|
687
|
+
model_name = str(entry.get("model") or "Target Model")
|
|
688
|
+
proto = entry.get("protocol") or {}
|
|
689
|
+
proto_desc = proto.get("description", "") if isinstance(proto, dict) else str(proto)
|
|
690
|
+
bench_name = str(entry.get("benchmark_name") or proto_desc or "Benchmark")
|
|
691
|
+
self.tracer.start_trace(
|
|
692
|
+
model=model_name, benchmark=bench_name,
|
|
693
|
+
n_cases=int(entry.get("n_cases", 0) or 0), metadata=entry,
|
|
694
|
+
)
|
|
695
|
+
|
|
696
|
+
@staticmethod
|
|
697
|
+
def _git_commit() -> "str | None":
|
|
698
|
+
import subprocess
|
|
699
|
+
|
|
700
|
+
try:
|
|
701
|
+
out = subprocess.run(
|
|
702
|
+
["git", "rev-parse", "--short", "HEAD"],
|
|
703
|
+
capture_output=True, text=True, timeout=3, check=False,
|
|
704
|
+
)
|
|
705
|
+
commit = out.stdout.strip()
|
|
706
|
+
if commit:
|
|
707
|
+
return commit
|
|
708
|
+
except Exception: # noqa: BLE001
|
|
709
|
+
pass
|
|
710
|
+
return os.environ.get("EVALRX_GIT_COMMIT") or None
|
|
711
|
+
|
|
712
|
+
def log_cases(self, cases: "Any", *, split: "str | None" = None) -> None:
|
|
713
|
+
"""Persist complete case I/O; media is copied into ``media/`` (rule 4's exception).
|
|
714
|
+
|
|
715
|
+
``split`` is the partition the loop assigned (``explore`` / ``confirm`` /
|
|
716
|
+
``test``), recorded on the row so a report can group the cases the way
|
|
717
|
+
the run actually used them.
|
|
718
|
+
"""
|
|
719
|
+
for case in cases:
|
|
720
|
+
case_id = str(getattr(case, "id", "") or "")
|
|
721
|
+
if not case_id or case_id in self._logged_case_ids:
|
|
722
|
+
continue
|
|
723
|
+
if hasattr(case, "to_dict"):
|
|
724
|
+
payload = case.to_dict()
|
|
725
|
+
elif isinstance(case, dict):
|
|
726
|
+
payload = dict(case)
|
|
727
|
+
else:
|
|
728
|
+
payload = {"id": case_id, "value": str(case)}
|
|
729
|
+
payload = json.loads(json.dumps(payload, ensure_ascii=False, default=str))
|
|
730
|
+
media_paths: list[str] = []
|
|
731
|
+
inputs = getattr(case, "inputs", None)
|
|
732
|
+
for kind in ("image", "audio", "video"):
|
|
733
|
+
value = getattr(inputs, kind, None)
|
|
734
|
+
if not isinstance(value, (str, Path)):
|
|
735
|
+
continue
|
|
736
|
+
path = Path(value)
|
|
737
|
+
if not path.is_absolute():
|
|
738
|
+
run_relative = self.run_dir / path
|
|
739
|
+
path = run_relative if run_relative.is_file() else path.resolve()
|
|
740
|
+
if not path.is_file():
|
|
741
|
+
continue
|
|
742
|
+
saved = self._save_case_media(path)
|
|
743
|
+
if saved:
|
|
744
|
+
media_paths.append(saved)
|
|
745
|
+
record: dict[str, Any] = {
|
|
746
|
+
"ts": self._ts(), "case_id": case_id, "case": payload, "media_paths": media_paths,
|
|
747
|
+
}
|
|
748
|
+
if split:
|
|
749
|
+
record["split"] = str(split)
|
|
750
|
+
self._append_run("cases", record)
|
|
751
|
+
self._logged_case_ids.add(case_id)
|
|
752
|
+
|
|
753
|
+
def log_report_published(self, envelope: "dict[str, Any]") -> None:
|
|
754
|
+
generated = envelope.get("generated_by") or {}
|
|
755
|
+
self._append_run("report_published", {
|
|
756
|
+
"ts": self._ts(),
|
|
757
|
+
"report_schema_version": int(envelope.get("schema_version") or 1),
|
|
758
|
+
"catalog_version": str(envelope.get("catalog_version") or ""),
|
|
759
|
+
"json_render_version": str(envelope.get("json_render_version") or ""),
|
|
760
|
+
"source_event_seq": int(envelope.get("source_event_seq") or 0),
|
|
761
|
+
"sha256": str(envelope.get("sha256") or ""),
|
|
762
|
+
"generated_by": generated,
|
|
763
|
+
"report_paths": ["report/report_data.json", "report/report_spec.json"],
|
|
764
|
+
})
|
|
765
|
+
|
|
766
|
+
def log_diagnose_report(
|
|
767
|
+
self,
|
|
768
|
+
report: Any,
|
|
769
|
+
cases: "list[Any]",
|
|
770
|
+
*,
|
|
771
|
+
discovery: "list[dict[str, Any]] | None" = None,
|
|
772
|
+
) -> None:
|
|
773
|
+
"""Inline the standard post-diagnosis report into ``run.json`` as one
|
|
774
|
+
detailed machine-readable record; a UI renders prose from this data
|
|
775
|
+
rather than reading separate JSON/Markdown siblings."""
|
|
776
|
+
hyps_src = getattr(report, "all_hypotheses", None)
|
|
777
|
+
if hyps_src is None:
|
|
778
|
+
hyps_src = getattr(report, "final_hypotheses", [])
|
|
779
|
+
hypotheses = [
|
|
780
|
+
{
|
|
781
|
+
"statement": h.statement,
|
|
782
|
+
"plain_statement": getattr(h, "plain_statement", ""),
|
|
783
|
+
"failure_mode": h.predicted_failure_mode,
|
|
784
|
+
"status": h.status.value if h.status else None,
|
|
785
|
+
}
|
|
786
|
+
for h in hyps_src
|
|
787
|
+
]
|
|
788
|
+
m4_results = [
|
|
789
|
+
{
|
|
790
|
+
"hypothesis": tr.hypothesis.statement,
|
|
791
|
+
"failure_mode": tr.hypothesis.predicted_failure_mode,
|
|
792
|
+
"status": tr.status.value,
|
|
793
|
+
"effect_size": tr.effect_size,
|
|
794
|
+
"confidence": tr.confidence,
|
|
795
|
+
"protocol_consistent": tr.is_consistent_with_protocol,
|
|
796
|
+
"verdict": tr.verdict,
|
|
797
|
+
"evidence": tr.evidence,
|
|
798
|
+
}
|
|
799
|
+
for tr in getattr(report, "all_test_results", [])
|
|
800
|
+
]
|
|
801
|
+
self._append_run("diagnose_reports", {
|
|
802
|
+
"ts": self._ts(),
|
|
803
|
+
"cycles": report.cycles,
|
|
804
|
+
"stopped_by": getattr(report, "stopped_by", None),
|
|
805
|
+
"resolved": getattr(report, "resolved", None),
|
|
806
|
+
"n_cases": len(cases),
|
|
807
|
+
"n_hypotheses": len(hypotheses),
|
|
808
|
+
"n_verified": len(getattr(report, "verified_hypotheses", [])),
|
|
809
|
+
"hypotheses": hypotheses,
|
|
810
|
+
"m4_results": m4_results,
|
|
811
|
+
"discovery": list(discovery or []),
|
|
812
|
+
})
|
|
813
|
+
|
|
814
|
+
def log_manifest(self, *, run_id: str, config: "dict[str, Any]") -> None:
|
|
815
|
+
"""Record final run provenance and a compact file index in ``run.json``."""
|
|
816
|
+
files = [
|
|
817
|
+
str(path.relative_to(self.run_dir))
|
|
818
|
+
for path in sorted(self.run_dir.rglob("*"))
|
|
819
|
+
if path.is_file() and not path.name.startswith(".")
|
|
820
|
+
]
|
|
821
|
+
with self._lock:
|
|
822
|
+
self._run_doc["manifest"] = {
|
|
823
|
+
"ts": self._ts(), "run_id": run_id, "config": dict(config), "files": files,
|
|
824
|
+
}
|
|
825
|
+
self._flush_run()
|
|
826
|
+
|
|
827
|
+
def log_runtime_snapshot(self, root: Path) -> None:
|
|
828
|
+
"""Preserve execution files, including discarded trials, before cleanup."""
|
|
829
|
+
snapshot = _inline_workspace(
|
|
830
|
+
root, self._stage_artifacts_dir("M5"), run_dir=self.run_dir,
|
|
831
|
+
max_bytes=None, preserve_all=True,
|
|
832
|
+
)
|
|
833
|
+
with self._lock:
|
|
834
|
+
self._run_doc["runtime_snapshot"] = snapshot
|
|
835
|
+
self._flush_run()
|
|
836
|
+
|
|
837
|
+
def log_model_exchange(
|
|
838
|
+
self,
|
|
839
|
+
stage: str,
|
|
840
|
+
*,
|
|
841
|
+
role: str,
|
|
842
|
+
operation: str,
|
|
843
|
+
inputs: Any,
|
|
844
|
+
output: Any = None,
|
|
845
|
+
error: "str | None" = None,
|
|
846
|
+
duration_sec: "float | None" = None,
|
|
847
|
+
metadata: "dict[str, Any] | None" = None,
|
|
848
|
+
cycle: "int | None" = None,
|
|
849
|
+
) -> None:
|
|
850
|
+
"""Persist one exact model/agent input-output exchange in its stage."""
|
|
851
|
+
def json_safe(value: Any) -> Any:
|
|
852
|
+
import dataclasses
|
|
853
|
+
|
|
854
|
+
if dataclasses.is_dataclass(value):
|
|
855
|
+
value = dataclasses.asdict(value)
|
|
856
|
+
elif hasattr(value, "to_dict") and callable(value.to_dict):
|
|
857
|
+
try:
|
|
858
|
+
value = value.to_dict()
|
|
859
|
+
except Exception: # noqa: BLE001
|
|
860
|
+
pass
|
|
861
|
+
return json.loads(json.dumps(value, ensure_ascii=False, default=str))
|
|
862
|
+
|
|
863
|
+
entry: dict[str, Any] = {
|
|
864
|
+
"ts": self._ts(),
|
|
865
|
+
"cycle": self.current_cycle if cycle is None else cycle,
|
|
866
|
+
"role": role,
|
|
867
|
+
"operation": operation, "inputs": json_safe(inputs), "output": json_safe(output),
|
|
868
|
+
"error": error, "metadata": json_safe(dict(metadata or {})),
|
|
869
|
+
}
|
|
870
|
+
if duration_sec is not None:
|
|
871
|
+
entry["duration_sec"] = round(duration_sec, 4)
|
|
872
|
+
self._append_stage(stage, "model_calls", entry)
|
|
873
|
+
|
|
874
|
+
def save_artifact_json(self, stem: str, obj: Any) -> "str | None":
|
|
875
|
+
"""Write *obj* as JSON under the run-global ``artifacts/`` dir; return rel path."""
|
|
876
|
+
try:
|
|
877
|
+
d = self.run_dir / "artifacts"
|
|
878
|
+
d.mkdir(parents=True, exist_ok=True)
|
|
879
|
+
path = d / stem
|
|
880
|
+
path.write_text(json.dumps(obj, indent=2, default=str), encoding="utf-8")
|
|
881
|
+
return str(path.relative_to(self.run_dir))
|
|
882
|
+
except Exception as exc: # noqa: BLE001
|
|
883
|
+
warnings.warn(f"RunLoggerV2: could not save artifact {stem!r}: {exc}")
|
|
884
|
+
return None
|
|
885
|
+
|
|
886
|
+
# ------------------------------------------------------------------
|
|
887
|
+
# M1 — target-model calls (see model_instrumentation.InstrumentedModel)
|
|
888
|
+
# ------------------------------------------------------------------
|
|
889
|
+
|
|
890
|
+
def log_model_call(
|
|
891
|
+
self,
|
|
892
|
+
*,
|
|
893
|
+
cycle: int,
|
|
894
|
+
analyzer: str,
|
|
895
|
+
call_index: int,
|
|
896
|
+
method: str,
|
|
897
|
+
inputs: Any,
|
|
898
|
+
kwargs: "dict[str, Any]",
|
|
899
|
+
output: Any,
|
|
900
|
+
duration_sec: float,
|
|
901
|
+
error: "str | None",
|
|
902
|
+
case_id: "str | None" = None,
|
|
903
|
+
batch_case_ids: "list[str] | None" = None,
|
|
904
|
+
n_batch_cases: "int | None" = None,
|
|
905
|
+
) -> None:
|
|
906
|
+
record: dict[str, Any] = {
|
|
907
|
+
"ts": self._ts(), "cycle": cycle, "analyzer": analyzer,
|
|
908
|
+
"call_index": call_index, "method": method,
|
|
909
|
+
"case_id": case_id, "batch_case_ids": batch_case_ids or [],
|
|
910
|
+
"n_batch_cases": n_batch_cases if n_batch_cases is not None else len(batch_case_ids or []),
|
|
911
|
+
"inputs": inputs, "kwargs": kwargs, "output": output,
|
|
912
|
+
"duration_sec": round(duration_sec, 4),
|
|
913
|
+
}
|
|
914
|
+
if error is not None:
|
|
915
|
+
record["error"] = error
|
|
916
|
+
with self._lock:
|
|
917
|
+
self._model_call_seq += 1
|
|
918
|
+
record["seq"] = self._model_call_seq
|
|
919
|
+
self._stamp_event("model_calls", record, "M1")
|
|
920
|
+
self._bucket("M1", "model_calls").append(record)
|
|
921
|
+
self._flush_stage("M1")
|
|
922
|
+
self._pending_model_calls.setdefault(cycle, []).append(record)
|
|
923
|
+
if self.verbose:
|
|
924
|
+
print(_V2JsonFormatter.line("M1", "model_call", record))
|
|
925
|
+
|
|
926
|
+
# ------------------------------------------------------------------
|
|
927
|
+
# M1 — probe
|
|
928
|
+
# ------------------------------------------------------------------
|
|
929
|
+
|
|
930
|
+
def log_probe(
|
|
931
|
+
self,
|
|
932
|
+
cycle: int,
|
|
933
|
+
results: "dict[str, Result]",
|
|
934
|
+
schema: "Any | None" = None,
|
|
935
|
+
*,
|
|
936
|
+
cases: "Any | None" = None,
|
|
937
|
+
judge_prompt: "str | None" = None,
|
|
938
|
+
judge_raw: "str | None" = None,
|
|
939
|
+
duration_sec: "float | None" = None,
|
|
940
|
+
failed_analyzers: "dict[str, str] | None" = None,
|
|
941
|
+
) -> "list[Path]":
|
|
942
|
+
"""M1: one entry in M1/log.json's "probe" list. ``artifact_paths``/
|
|
943
|
+
results are inlined here instead of living in separate
|
|
944
|
+
``.result.json`` files."""
|
|
945
|
+
artifact_paths: dict[str, str] = {}
|
|
946
|
+
overlay_pngs: list[Path] = []
|
|
947
|
+
result_docs: dict[str, Any] = {}
|
|
948
|
+
inline_artifacts: dict[str, Any] = {}
|
|
949
|
+
for name, result in results.items():
|
|
950
|
+
for art_name, artifact in getattr(result, "artifacts", {}).items():
|
|
951
|
+
stem = f"c{cycle}_{name}_{art_name}"
|
|
952
|
+
rel = self._save_media("M1", stem, artifact)
|
|
953
|
+
if rel is not None:
|
|
954
|
+
artifact_paths[f"{name}/{art_name}"] = rel
|
|
955
|
+
elif isinstance(artifact, (dict, list)):
|
|
956
|
+
inline_artifacts[f"{name}/{art_name}"] = json.loads(
|
|
957
|
+
json.dumps(artifact, ensure_ascii=False, default=str)
|
|
958
|
+
)
|
|
959
|
+
image_overlays = getattr(result, "image_overlays", None)
|
|
960
|
+
if image_overlays is not None:
|
|
961
|
+
try:
|
|
962
|
+
overlay_pngs.extend(
|
|
963
|
+
image_overlays(self._stage_artifacts_dir("M1"), f"c{cycle}_{name}")
|
|
964
|
+
)
|
|
965
|
+
except Exception as exc: # noqa: BLE001 - viz must never break the probe
|
|
966
|
+
warnings.warn(f"RunLoggerV2: image_overlays failed for {name}: {exc}")
|
|
967
|
+
to_dict = getattr(result, "to_dict", None)
|
|
968
|
+
if callable(to_dict):
|
|
969
|
+
try:
|
|
970
|
+
doc = to_dict()
|
|
971
|
+
summary = getattr(result, "summary", None)
|
|
972
|
+
if callable(summary):
|
|
973
|
+
doc["summary"] = summary()
|
|
974
|
+
result_docs[name] = doc
|
|
975
|
+
except Exception as exc: # noqa: BLE001
|
|
976
|
+
warnings.warn(f"RunLoggerV2: could not serialise result {name!r}: {exc}")
|
|
977
|
+
|
|
978
|
+
with self._lock:
|
|
979
|
+
pending_calls = [c for bucket in self._pending_model_calls.values() for c in bucket]
|
|
980
|
+
self._pending_model_calls.clear()
|
|
981
|
+
|
|
982
|
+
entry: dict[str, Any] = {
|
|
983
|
+
"ts": self._ts(), "cycle": cycle,
|
|
984
|
+
"analyzers": list(results),
|
|
985
|
+
"findings": {name: r.findings for name, r in results.items()},
|
|
986
|
+
"results": result_docs,
|
|
987
|
+
"artifact_paths": artifact_paths,
|
|
988
|
+
"artifacts": inline_artifacts,
|
|
989
|
+
"n_model_calls": len(pending_calls),
|
|
990
|
+
}
|
|
991
|
+
examples = _probe_examples(results, cases)
|
|
992
|
+
if examples:
|
|
993
|
+
entry["examples"] = examples
|
|
994
|
+
if failed_analyzers:
|
|
995
|
+
entry["failed_analyzers"] = dict(failed_analyzers)
|
|
996
|
+
if schema is not None:
|
|
997
|
+
entry["selection_rationale"] = getattr(schema, "rationale", "")
|
|
998
|
+
selected = getattr(schema, "selected_analyzers", None)
|
|
999
|
+
if selected is not None:
|
|
1000
|
+
entry["selected_analyzers"] = list(selected)
|
|
1001
|
+
if judge_prompt:
|
|
1002
|
+
entry["judge_prompt"] = judge_prompt
|
|
1003
|
+
if judge_raw:
|
|
1004
|
+
entry["judge_response"] = judge_raw
|
|
1005
|
+
if duration_sec is not None:
|
|
1006
|
+
entry["duration_sec"] = round(duration_sec, 3)
|
|
1007
|
+
self._append_stage("M1", "probe", entry)
|
|
1008
|
+
if judge_prompt or judge_raw:
|
|
1009
|
+
self.log_model_exchange(
|
|
1010
|
+
"M1", role="analyzer_selection_judge", operation="generate",
|
|
1011
|
+
inputs=judge_prompt or "", output=judge_raw or "", cycle=cycle,
|
|
1012
|
+
duration_sec=duration_sec,
|
|
1013
|
+
)
|
|
1014
|
+
|
|
1015
|
+
png_figures: list[Path] = list(overlay_pngs)
|
|
1016
|
+
for rel_npy in artifact_paths.values():
|
|
1017
|
+
if not rel_npy.endswith(".npy"):
|
|
1018
|
+
continue
|
|
1019
|
+
png = self.run_dir / (rel_npy[: -len(".npy")] + ".png")
|
|
1020
|
+
if png.exists():
|
|
1021
|
+
png_figures.append(png)
|
|
1022
|
+
|
|
1023
|
+
m1_span = self.tracer.start_span(
|
|
1024
|
+
name=f"M1: Multi-Dimensional Checkup (Cycle {cycle})",
|
|
1025
|
+
stage="M1",
|
|
1026
|
+
input_data={"analyzers": list(results.keys()), "selected_analyzers": entry.get("selected_analyzers", [])},
|
|
1027
|
+
metadata={"duration_sec": duration_sec, "artifacts": {
|
|
1028
|
+
"artifact_paths": artifact_paths, "figures": [str(p) for p in png_figures],
|
|
1029
|
+
}},
|
|
1030
|
+
)
|
|
1031
|
+
if judge_prompt or judge_raw:
|
|
1032
|
+
self.tracer.log_generation(
|
|
1033
|
+
name="M1 Analyzer Selection", model="judge",
|
|
1034
|
+
prompt=judge_prompt or "", completion=judge_raw or "", span_id=m1_span,
|
|
1035
|
+
)
|
|
1036
|
+
for name, r in results.items():
|
|
1037
|
+
findings = getattr(r, "findings", {}) or {}
|
|
1038
|
+
per_case = findings.get("per_case") or []
|
|
1039
|
+
probe_span = self.tracer.start_span(
|
|
1040
|
+
name=f"Probe: {name}", stage=f"M1_{name}", input_data={"probe": name},
|
|
1041
|
+
parent_id=m1_span,
|
|
1042
|
+
metadata={
|
|
1043
|
+
"n_scored": len(per_case) if per_case else (findings.get("n_cases") or findings.get("n_scored")),
|
|
1044
|
+
},
|
|
1045
|
+
)
|
|
1046
|
+
calls_for_analyzer = [c for c in pending_calls if c.get("analyzer") == name]
|
|
1047
|
+
for call in calls_for_analyzer[:50]:
|
|
1048
|
+
self.tracer.log_generation(
|
|
1049
|
+
name=f"{name} · {call.get('method')} #{call.get('call_index')}",
|
|
1050
|
+
model="target_model", prompt=call.get("inputs"),
|
|
1051
|
+
completion=call.get("error") or call.get("output"),
|
|
1052
|
+
span_id=probe_span,
|
|
1053
|
+
metadata={"duration_sec": call.get("duration_sec"), "error": call.get("error")},
|
|
1054
|
+
)
|
|
1055
|
+
self.tracer.end_span(probe_span, output_data={
|
|
1056
|
+
"findings": findings, "n_model_calls": len(calls_for_analyzer),
|
|
1057
|
+
})
|
|
1058
|
+
orphaned = {c["analyzer"] for c in pending_calls} - set(results)
|
|
1059
|
+
for name in orphaned:
|
|
1060
|
+
failed_span = self.tracer.start_span(
|
|
1061
|
+
name=f"Probe: {name} (failed)", stage=f"M1_{name}", input_data={"probe": name},
|
|
1062
|
+
parent_id=m1_span, metadata={"note": "analyzer raised before producing a Result"},
|
|
1063
|
+
)
|
|
1064
|
+
calls_for_analyzer = [c for c in pending_calls if c.get("analyzer") == name]
|
|
1065
|
+
for call in calls_for_analyzer[:50]:
|
|
1066
|
+
self.tracer.log_generation(
|
|
1067
|
+
name=f"{name} · {call.get('method')} #{call.get('call_index')}",
|
|
1068
|
+
model="target_model", prompt=call.get("inputs"),
|
|
1069
|
+
completion=call.get("error") or call.get("output"),
|
|
1070
|
+
span_id=failed_span,
|
|
1071
|
+
metadata={"duration_sec": call.get("duration_sec"), "error": call.get("error")},
|
|
1072
|
+
)
|
|
1073
|
+
self.tracer.end_span(failed_span, status="failed", output_data={
|
|
1074
|
+
"n_model_calls": len(calls_for_analyzer),
|
|
1075
|
+
})
|
|
1076
|
+
self.tracer.end_span(m1_span, output_data={"n_probes": len(results)})
|
|
1077
|
+
return png_figures
|
|
1078
|
+
|
|
1079
|
+
# ------------------------------------------------------------------
|
|
1080
|
+
# M2 — analysis + explore
|
|
1081
|
+
# ------------------------------------------------------------------
|
|
1082
|
+
|
|
1083
|
+
def log_analysis(
|
|
1084
|
+
self, cycle: int, report: "AnalysisReport", *, duration_sec: "float | None" = None,
|
|
1085
|
+
) -> None:
|
|
1086
|
+
entry: dict[str, Any] = {
|
|
1087
|
+
"ts": self._ts(), "cycle": cycle,
|
|
1088
|
+
"severity": report.severity,
|
|
1089
|
+
"n_findings": len(report.findings),
|
|
1090
|
+
"findings": [str(f) for f in report.findings],
|
|
1091
|
+
"narrative": report.narrative,
|
|
1092
|
+
"descriptive_only": bool(getattr(report, "descriptive_only", False)),
|
|
1093
|
+
}
|
|
1094
|
+
stats_tool = getattr(report, "stats_tool", None)
|
|
1095
|
+
if stats_tool:
|
|
1096
|
+
entry["stats_tool"] = stats_tool
|
|
1097
|
+
fallback_reason = getattr(report, "llm_fallback_reason", None)
|
|
1098
|
+
if fallback_reason:
|
|
1099
|
+
entry["llm_fallback_reason"] = fallback_reason
|
|
1100
|
+
conclusion = getattr(report, "conclusion", None)
|
|
1101
|
+
if conclusion:
|
|
1102
|
+
entry["conclusion"] = conclusion
|
|
1103
|
+
evidence_chain = getattr(report, "evidence_chain", None)
|
|
1104
|
+
if evidence_chain:
|
|
1105
|
+
entry["evidence_chain"] = list(evidence_chain)
|
|
1106
|
+
stats_tool_results = getattr(report, "stats_tool_results", None)
|
|
1107
|
+
if stats_tool_results:
|
|
1108
|
+
entry["stats_tool_results"] = list(stats_tool_results)
|
|
1109
|
+
visualizations = getattr(report, "visualizations", None)
|
|
1110
|
+
if visualizations:
|
|
1111
|
+
entry["visualizations"] = list(visualizations)
|
|
1112
|
+
stats_plan = getattr(report, "stats_plan", None)
|
|
1113
|
+
if stats_plan:
|
|
1114
|
+
entry["stats_plan"] = stats_plan
|
|
1115
|
+
stats_results = getattr(report, "stats_results", None)
|
|
1116
|
+
if stats_results:
|
|
1117
|
+
entry["stats_results"] = [r.to_dict() for r in stats_results]
|
|
1118
|
+
corrected = getattr(report, "corrected_rejections", None)
|
|
1119
|
+
if corrected:
|
|
1120
|
+
entry["corrected_rejections"] = corrected
|
|
1121
|
+
figures = getattr(report, "figures", None)
|
|
1122
|
+
if figures:
|
|
1123
|
+
entry["figures"] = [self._portable_path(f) for f in figures]
|
|
1124
|
+
llm_prompt = getattr(report, "llm_prompt", None)
|
|
1125
|
+
llm_raw = getattr(report, "llm_raw", None)
|
|
1126
|
+
if llm_prompt:
|
|
1127
|
+
entry["judge_prompt"] = llm_prompt
|
|
1128
|
+
if llm_raw:
|
|
1129
|
+
entry["judge_response"] = llm_raw
|
|
1130
|
+
if duration_sec is not None:
|
|
1131
|
+
entry["duration_sec"] = round(duration_sec, 3)
|
|
1132
|
+
self._append_stage("M2", "analysis", entry)
|
|
1133
|
+
if llm_prompt or llm_raw:
|
|
1134
|
+
self.log_model_exchange(
|
|
1135
|
+
"M2", role="statistics_judge", operation="generate",
|
|
1136
|
+
inputs=llm_prompt or "", output=llm_raw or "", cycle=cycle,
|
|
1137
|
+
duration_sec=duration_sec,
|
|
1138
|
+
)
|
|
1139
|
+
|
|
1140
|
+
m2_span = self.tracer.start_span(
|
|
1141
|
+
name=f"M2: Screening & Confirmatory Signals (Cycle {cycle})", stage="M2",
|
|
1142
|
+
input_data={"severity": report.severity, "n_findings": len(report.findings)},
|
|
1143
|
+
metadata={"duration_sec": duration_sec, "stats_tool": stats_tool,
|
|
1144
|
+
"artifacts": {"figures": [str(f) for f in (figures or [])]}},
|
|
1145
|
+
)
|
|
1146
|
+
if llm_prompt or llm_raw:
|
|
1147
|
+
self.tracer.log_generation(
|
|
1148
|
+
name="M2 Statistical Screening Analysis", model="judge",
|
|
1149
|
+
prompt=llm_prompt or "", completion=llm_raw or "", span_id=m2_span,
|
|
1150
|
+
)
|
|
1151
|
+
for s in (stats_results or []):
|
|
1152
|
+
s_dict = s.to_dict() if hasattr(s, "to_dict") else (s if isinstance(s, dict) else {})
|
|
1153
|
+
sig_name = s_dict.get("config", {}).get("signal") or s_dict.get("tool") or "signal"
|
|
1154
|
+
eff = s_dict.get("effect")
|
|
1155
|
+
if eff is not None:
|
|
1156
|
+
self.tracer.log_score(
|
|
1157
|
+
name=f"m2_effect_{sig_name}", value=float(eff),
|
|
1158
|
+
comment=f"p={s_dict.get('p_value')}", span_id=m2_span,
|
|
1159
|
+
)
|
|
1160
|
+
self.tracer.end_span(m2_span, output_data={"conclusion": conclusion or ""})
|
|
1161
|
+
|
|
1162
|
+
def log_explore(
|
|
1163
|
+
self, cycle: int, report: "Any | None", *,
|
|
1164
|
+
out_dir: "Path | str | None" = None, duration_sec: "float | None" = None,
|
|
1165
|
+
) -> None:
|
|
1166
|
+
ok = bool(getattr(report, "ok", False)) if report is not None else False
|
|
1167
|
+
charts = list(getattr(report, "charts", None) or []) if report is not None else []
|
|
1168
|
+
rendered = [
|
|
1169
|
+
str(c.get("figure_path")) for c in charts
|
|
1170
|
+
if isinstance(c, dict) and c.get("figure_path")
|
|
1171
|
+
]
|
|
1172
|
+
workspace = None
|
|
1173
|
+
if out_dir is not None:
|
|
1174
|
+
workspace = _inline_workspace(
|
|
1175
|
+
out_dir, self._stage_artifacts_dir("M2"), run_dir=self.run_dir,
|
|
1176
|
+
)
|
|
1177
|
+
if workspace and workspace.get("media"):
|
|
1178
|
+
rendered = [
|
|
1179
|
+
path for path in workspace["media"]
|
|
1180
|
+
if Path(path).suffix.lower() in {".png", ".jpg", ".jpeg", ".gif", ".webp", ".bmp"}
|
|
1181
|
+
]
|
|
1182
|
+
tables = getattr(report, "tables", None) or {}
|
|
1183
|
+
adjudication = dict(getattr(report, "adjudication", None) or {}) if report is not None else {}
|
|
1184
|
+
entry: dict[str, Any] = {
|
|
1185
|
+
"ts": self._ts(), "cycle": cycle, "ok": ok,
|
|
1186
|
+
"n_observations": len(getattr(report, "observations", None) or []) if report is not None else 0,
|
|
1187
|
+
"n_charts": len(charts), "n_charts_rendered": len(rendered),
|
|
1188
|
+
"n_tables": len(tables) if isinstance(tables, dict) else len(list(tables or [])),
|
|
1189
|
+
"n_candidate_signals": len(getattr(report, "candidate_signals", None) or []) if report is not None else 0,
|
|
1190
|
+
"n_hypotheses": len(getattr(report, "hypotheses", None) or []) if report is not None else 0,
|
|
1191
|
+
"adjudication": {
|
|
1192
|
+
k: adjudication[k] for k in (
|
|
1193
|
+
"method", "alpha", "split", "n_host_adjudicated", "n_rejected",
|
|
1194
|
+
"n_in_family", "n_descriptive_only",
|
|
1195
|
+
) if k in adjudication
|
|
1196
|
+
},
|
|
1197
|
+
"observations": [str(o) for o in (getattr(report, "observations", None) or [])[:12]] if report is not None else [],
|
|
1198
|
+
"caveats": [str(c) for c in (getattr(report, "caveats", None) or [])[:8]] if report is not None else [],
|
|
1199
|
+
"figures": rendered,
|
|
1200
|
+
"workspace_snapshot": workspace,
|
|
1201
|
+
"attempts": int(getattr(report, "attempts", 0) or 0) if report is not None else 0,
|
|
1202
|
+
}
|
|
1203
|
+
error = str(getattr(report, "error", "") or "") if report is not None else "explorer produced no report"
|
|
1204
|
+
if error:
|
|
1205
|
+
entry["error"] = error
|
|
1206
|
+
if out_dir is not None and workspace is None:
|
|
1207
|
+
entry["out_dir"] = str(out_dir)
|
|
1208
|
+
report_path = Path(out_dir) / "exploratory_report.json"
|
|
1209
|
+
if report_path.exists():
|
|
1210
|
+
entry["report_path"] = str(report_path)
|
|
1211
|
+
if report is not None and getattr(report, "code", None):
|
|
1212
|
+
entry["code"] = str(report.code)
|
|
1213
|
+
if report is not None and getattr(report, "raw_outputs", None):
|
|
1214
|
+
entry["raw_outputs"] = [str(r) for r in report.raw_outputs]
|
|
1215
|
+
if duration_sec is not None:
|
|
1216
|
+
entry["duration_sec"] = round(duration_sec, 3)
|
|
1217
|
+
self._append_stage("M2", "explore", entry)
|
|
1218
|
+
for call in list(getattr(report, "model_calls", None) or []):
|
|
1219
|
+
self.log_model_exchange(
|
|
1220
|
+
"M2",
|
|
1221
|
+
role=str(call.get("role") or "explore_coder"),
|
|
1222
|
+
operation=str(call.get("operation") or "generate"),
|
|
1223
|
+
inputs=call.get("inputs"), output=call.get("output"),
|
|
1224
|
+
error=call.get("error"), duration_sec=call.get("duration_sec"),
|
|
1225
|
+
metadata=dict(call.get("metadata") or {}), cycle=cycle,
|
|
1226
|
+
)
|
|
1227
|
+
|
|
1228
|
+
exp_span = self.tracer.start_span(
|
|
1229
|
+
name=f"Explore: Free-form EDA (Cycle {cycle})", stage="EXPLORE",
|
|
1230
|
+
input_data={"ok": ok, "attempts": entry.get("attempts", 0)},
|
|
1231
|
+
metadata={"duration_sec": duration_sec, "artifacts": {
|
|
1232
|
+
"out_dir": str(out_dir) if out_dir is not None else None,
|
|
1233
|
+
"report_path": entry.get("report_path"), "figures": rendered,
|
|
1234
|
+
}},
|
|
1235
|
+
)
|
|
1236
|
+
if report is not None:
|
|
1237
|
+
for i, raw in enumerate(getattr(report, "raw_outputs", None) or []):
|
|
1238
|
+
self.tracer.log_generation(
|
|
1239
|
+
name=f"Explore Coder Agent (attempt {i + 1})", model="coder_agent",
|
|
1240
|
+
prompt=None, completion=str(raw), span_id=exp_span,
|
|
1241
|
+
)
|
|
1242
|
+
if getattr(report, "code", None):
|
|
1243
|
+
self.tracer.log_generation(
|
|
1244
|
+
name="Explore Analysis Code (analysis.py)", model="coder_agent",
|
|
1245
|
+
prompt=None, completion=str(report.code), span_id=exp_span,
|
|
1246
|
+
)
|
|
1247
|
+
self.tracer.end_span(exp_span, output_data={
|
|
1248
|
+
"n_observations": entry.get("n_observations", 0),
|
|
1249
|
+
"n_candidate_signals": entry.get("n_candidate_signals", 0),
|
|
1250
|
+
})
|
|
1251
|
+
|
|
1252
|
+
# ------------------------------------------------------------------
|
|
1253
|
+
# M3 — diagnosis
|
|
1254
|
+
# ------------------------------------------------------------------
|
|
1255
|
+
|
|
1256
|
+
def log_diagnosis(
|
|
1257
|
+
self, cycle: int, diag: "DiagnosisResult", *,
|
|
1258
|
+
duration_sec: "float | None" = None, explore_figures: "list[str] | None" = None,
|
|
1259
|
+
) -> None:
|
|
1260
|
+
entry: dict[str, Any] = {
|
|
1261
|
+
"ts": self._ts(), "cycle": cycle,
|
|
1262
|
+
"model_name": diag.model_name,
|
|
1263
|
+
"n_hypotheses": len(diag.hypotheses),
|
|
1264
|
+
"hypotheses": [
|
|
1265
|
+
{
|
|
1266
|
+
# Join key for M4/M5 entries logged against this same
|
|
1267
|
+
# hypothesis later (log_surgery/log_experiment/log_fix) —
|
|
1268
|
+
# see evalrx.eval_agent.hypothesis.hypothesis_id.
|
|
1269
|
+
"id": hypothesis_id(h),
|
|
1270
|
+
"statement": h.statement, "plain_statement": h.plain_statement,
|
|
1271
|
+
"failure_mode": h.predicted_failure_mode,
|
|
1272
|
+
"status": h.status.value if h.status else None,
|
|
1273
|
+
"test_design": h.test_design,
|
|
1274
|
+
"critic": (h.metadata or {}).get("critic"),
|
|
1275
|
+
"critic_reason": (h.metadata or {}).get("critic_reason"),
|
|
1276
|
+
}
|
|
1277
|
+
for h in diag.hypotheses
|
|
1278
|
+
],
|
|
1279
|
+
"raw_judge_output": diag.raw_judge_output,
|
|
1280
|
+
"n_critic_kept": int(getattr(diag, "n_critic_kept", 0) or 0),
|
|
1281
|
+
"n_critic_rejected": int(getattr(diag, "n_critic_rejected", 0) or 0),
|
|
1282
|
+
}
|
|
1283
|
+
critic_raw = getattr(diag, "critic_raw_output", "") or ""
|
|
1284
|
+
if critic_raw:
|
|
1285
|
+
entry["critic_raw_output"] = critic_raw
|
|
1286
|
+
entry["critic_prompt"] = getattr(diag, "critic_prompt", "") or ""
|
|
1287
|
+
proposed = list(getattr(diag, "proposed_hypotheses", None) or [])
|
|
1288
|
+
if proposed:
|
|
1289
|
+
entry["proposed_hypotheses"] = [
|
|
1290
|
+
{"statement": h.statement, "plain_statement": h.plain_statement,
|
|
1291
|
+
"failure_mode": h.predicted_failure_mode, "test_design": h.test_design}
|
|
1292
|
+
for h in proposed
|
|
1293
|
+
]
|
|
1294
|
+
review_decisions = list(getattr(diag, "review_decisions", None) or [])
|
|
1295
|
+
if review_decisions:
|
|
1296
|
+
entry["review"] = {
|
|
1297
|
+
"n_kept": sum(d.get("decision") == "keep" for d in review_decisions),
|
|
1298
|
+
"n_rejected": sum(d.get("decision") == "reject" for d in review_decisions),
|
|
1299
|
+
"decisions": review_decisions,
|
|
1300
|
+
}
|
|
1301
|
+
referenced = getattr(diag, "referenced_charts", None)
|
|
1302
|
+
if referenced:
|
|
1303
|
+
entry["referenced_charts"] = list(referenced)
|
|
1304
|
+
if getattr(diag, "explore_context_used", False):
|
|
1305
|
+
entry["explore_context_used"] = True
|
|
1306
|
+
if getattr(diag, "failure_modes_used", False):
|
|
1307
|
+
entry["failure_modes_used"] = True
|
|
1308
|
+
if explore_figures:
|
|
1309
|
+
entry["explore_figures"] = list(explore_figures)
|
|
1310
|
+
m3_prompt = getattr(diag, "prompt", None) or ""
|
|
1311
|
+
if m3_prompt:
|
|
1312
|
+
entry["judge_prompt"] = m3_prompt
|
|
1313
|
+
review_prompt = getattr(diag, "review_prompt", None) or ""
|
|
1314
|
+
review_raw = getattr(diag, "review_raw", None) or ""
|
|
1315
|
+
if review_prompt or review_raw:
|
|
1316
|
+
entry["review_prompt"] = review_prompt
|
|
1317
|
+
entry["review_response"] = review_raw
|
|
1318
|
+
if duration_sec is not None:
|
|
1319
|
+
entry["duration_sec"] = round(duration_sec, 3)
|
|
1320
|
+
self._append_stage("M3", "diagnosis", entry)
|
|
1321
|
+
detailed_calls = list(getattr(diag, "model_calls", None) or [])
|
|
1322
|
+
if detailed_calls:
|
|
1323
|
+
for call in detailed_calls:
|
|
1324
|
+
self.log_model_exchange(
|
|
1325
|
+
"M3",
|
|
1326
|
+
role=str(call.get("role") or "diagnosis_judge"),
|
|
1327
|
+
operation=str(call.get("operation") or "generate"),
|
|
1328
|
+
inputs=call.get("inputs"), output=call.get("output"),
|
|
1329
|
+
error=call.get("error"), duration_sec=call.get("duration_sec"),
|
|
1330
|
+
metadata=dict(call.get("metadata") or {}), cycle=cycle,
|
|
1331
|
+
)
|
|
1332
|
+
else:
|
|
1333
|
+
# Compatibility for DiagnosisResult values created by older callers.
|
|
1334
|
+
if m3_prompt or diag.raw_judge_output:
|
|
1335
|
+
self.log_model_exchange(
|
|
1336
|
+
"M3", role="diagnosis_judge", operation="generate",
|
|
1337
|
+
inputs=m3_prompt, output=diag.raw_judge_output or "", cycle=cycle,
|
|
1338
|
+
duration_sec=duration_sec,
|
|
1339
|
+
)
|
|
1340
|
+
if critic_raw or review_prompt or review_raw:
|
|
1341
|
+
self.log_model_exchange(
|
|
1342
|
+
"M3", role="hypothesis_critic", operation="generate",
|
|
1343
|
+
inputs=entry.get("critic_prompt") or review_prompt,
|
|
1344
|
+
output=critic_raw or review_raw, cycle=cycle,
|
|
1345
|
+
)
|
|
1346
|
+
|
|
1347
|
+
m3_span = self.tracer.start_span(
|
|
1348
|
+
name=f"M3: Root-Cause Diagnosis (Cycle {cycle})", stage="M3",
|
|
1349
|
+
input_data={"model_name": diag.model_name, "n_hypotheses": len(diag.hypotheses)},
|
|
1350
|
+
metadata={"duration_sec": duration_sec},
|
|
1351
|
+
)
|
|
1352
|
+
self.tracer.log_generation(
|
|
1353
|
+
name="AI Doctor Diagnostician",
|
|
1354
|
+
model=str(self.tracer.trace_metadata.get("judge") or "diagnosis_judge"),
|
|
1355
|
+
prompt=m3_prompt, completion=diag.raw_judge_output or "", span_id=m3_span,
|
|
1356
|
+
metadata={"hypotheses": [h.statement for h in diag.hypotheses]},
|
|
1357
|
+
)
|
|
1358
|
+
if review_prompt or review_raw:
|
|
1359
|
+
self.tracer.log_generation(
|
|
1360
|
+
name="M3 Adversarial Evidence Review",
|
|
1361
|
+
model=str(self.tracer.trace_metadata.get("judge") or "diagnosis_judge"),
|
|
1362
|
+
prompt=review_prompt, completion=review_raw, span_id=m3_span,
|
|
1363
|
+
metadata={"decisions": review_decisions},
|
|
1364
|
+
)
|
|
1365
|
+
self.tracer.end_span(m3_span, output_data={
|
|
1366
|
+
"n_hypotheses": len(diag.hypotheses),
|
|
1367
|
+
"n_proposed": len(proposed) or len(diag.hypotheses),
|
|
1368
|
+
"review": entry.get("review"),
|
|
1369
|
+
})
|
|
1370
|
+
|
|
1371
|
+
# ------------------------------------------------------------------
|
|
1372
|
+
# M4/M5 — hypothesis verification & intervention
|
|
1373
|
+
# ------------------------------------------------------------------
|
|
1374
|
+
|
|
1375
|
+
def log_surgery(
|
|
1376
|
+
self, cycle: int, hypothesis: "Hypothesis", iv: "InterventionResult", *,
|
|
1377
|
+
validation_cases: "Any | None" = None, duration_sec: "float | None" = None,
|
|
1378
|
+
judge_prompt: "str | None" = None, judge_raw: "str | None" = None,
|
|
1379
|
+
) -> None:
|
|
1380
|
+
"""M4 (hypothesis verification) or M5 (intervention) — split on
|
|
1381
|
+
whether "m4_test_name" is present in ``iv.evidence``."""
|
|
1382
|
+
is_m4 = "m4_test_name" in (iv.evidence or {})
|
|
1383
|
+
stage = "M4" if is_m4 else "M5"
|
|
1384
|
+
entry: dict[str, Any] = {
|
|
1385
|
+
"ts": self._ts(), "cycle": cycle,
|
|
1386
|
+
"module": stage.lower(),
|
|
1387
|
+
|
|
1388
|
+
# Joins this verdict back to its M3 hypotheses[] entry (same id).
|
|
1389
|
+
"hypothesis_id": hypothesis_id(hypothesis),
|
|
1390
|
+
"hypothesis": hypothesis.statement,
|
|
1391
|
+
"failure_mode": hypothesis.predicted_failure_mode,
|
|
1392
|
+
"status": iv.status.value, "fixed": iv.fixed,
|
|
1393
|
+
"confidence_score": iv.confidence_score,
|
|
1394
|
+
"evidence_dimensions": iv.evidence_dimensions,
|
|
1395
|
+
"evidence": iv.evidence,
|
|
1396
|
+
"n_refocused_cases": len(iv.new_data) if iv.new_data else None,
|
|
1397
|
+
}
|
|
1398
|
+
if is_m4:
|
|
1399
|
+
candidates = _iter_cases(validation_cases)
|
|
1400
|
+
if candidates:
|
|
1401
|
+
snapshots = [_case_snapshot(case) for case in candidates]
|
|
1402
|
+
snapshot = next(
|
|
1403
|
+
(item for item in snapshots if str(item.get("outcome", "")).lower() == "fail"),
|
|
1404
|
+
snapshots[0],
|
|
1405
|
+
)
|
|
1406
|
+
entry["validation_examples"] = [{
|
|
1407
|
+
"id": f"m4-{snapshot.get('id')}", "kind": "validation_case",
|
|
1408
|
+
"case_id": snapshot.get("id"), **snapshot,
|
|
1409
|
+
"plain_reading": "This is one case in the independent validation pool. The verdict is determined from the full pool, not this case alone.",
|
|
1410
|
+
"evidence_scope": "one case in the independent validation pool",
|
|
1411
|
+
}]
|
|
1412
|
+
if judge_prompt:
|
|
1413
|
+
entry["judge_prompt"] = judge_prompt
|
|
1414
|
+
if judge_raw:
|
|
1415
|
+
entry["judge_response"] = judge_raw
|
|
1416
|
+
if duration_sec is not None:
|
|
1417
|
+
entry["duration_sec"] = round(duration_sec, 3)
|
|
1418
|
+
self._append_stage(stage, "surgery", entry)
|
|
1419
|
+
if judge_prompt or judge_raw:
|
|
1420
|
+
self.log_model_exchange(
|
|
1421
|
+
stage, role="protocol_consistency_judge", operation="generate",
|
|
1422
|
+
inputs=judge_prompt or "", output=judge_raw or "", cycle=cycle,
|
|
1423
|
+
duration_sec=duration_sec,
|
|
1424
|
+
)
|
|
1425
|
+
|
|
1426
|
+
stage_title = "M4 Adjudication" if is_m4 else "M5 Intervention"
|
|
1427
|
+
surg_span = self.tracer.start_span(
|
|
1428
|
+
name=f"{stage_title}: {hypothesis.statement[:60]}",
|
|
1429
|
+
stage="M4" if is_m4 else "M5_SURGERY",
|
|
1430
|
+
input_data={"hypothesis": hypothesis.statement, "failure_mode": hypothesis.predicted_failure_mode},
|
|
1431
|
+
metadata={"status": iv.status.value, "fixed": iv.fixed, "confidence_score": iv.confidence_score},
|
|
1432
|
+
)
|
|
1433
|
+
if judge_prompt or judge_raw:
|
|
1434
|
+
self.tracer.log_generation(
|
|
1435
|
+
name=f"{stage_title}: Protocol Consistency Judge", model="judge",
|
|
1436
|
+
prompt=judge_prompt or "", completion=judge_raw or "", span_id=surg_span,
|
|
1437
|
+
)
|
|
1438
|
+
if iv.confidence_score is not None:
|
|
1439
|
+
self.tracer.log_score(
|
|
1440
|
+
name="adjudication_confidence", value=float(iv.confidence_score),
|
|
1441
|
+
comment=f"status={iv.status.value}, fixed={iv.fixed}", span_id=surg_span,
|
|
1442
|
+
)
|
|
1443
|
+
self.tracer.end_span(surg_span, output_data={"evidence": iv.evidence or {}})
|
|
1444
|
+
|
|
1445
|
+
def log_experiment(
|
|
1446
|
+
self, cycle: int, hypothesis: "Hypothesis", iv: "InterventionResult", *,
|
|
1447
|
+
module: str = "m5",
|
|
1448
|
+
) -> None:
|
|
1449
|
+
"""The experiment the agent wrote and ran to test *hypothesis*.
|
|
1450
|
+
|
|
1451
|
+
No separate ``experiments/``/``workspace/`` files: the generated
|
|
1452
|
+
source (``exp["files"]``/``exp["code"]``), stdout/stderr, the coder
|
|
1453
|
+
agent's raw narration, and the validation log are already plain
|
|
1454
|
+
strings in *iv.experiment* and go straight into the entry as JSON
|
|
1455
|
+
string values.
|
|
1456
|
+
"""
|
|
1457
|
+
exp = getattr(iv, "experiment", None) or {}
|
|
1458
|
+
files = exp.get("files") or {}
|
|
1459
|
+
if not files and exp.get("code"):
|
|
1460
|
+
files = {"main.py": exp["code"]}
|
|
1461
|
+
|
|
1462
|
+
# An "experiment" is always M4/M5-shaped content by definition, so it
|
|
1463
|
+
# gets the same resolve-with-M5-floor treatment as its workspace
|
|
1464
|
+
# media, rather than trusting `_append_stage`'s generic unroutable
|
|
1465
|
+
# fallback (which would file a genuinely M5-ish record under
|
|
1466
|
+
# run.json["unrouted"] if *module* is ever something the M1-M5 regex
|
|
1467
|
+
# and alias table don't recognize).
|
|
1468
|
+
stage = _resolve_stage(module) or "M5"
|
|
1469
|
+
|
|
1470
|
+
workspace = None
|
|
1471
|
+
workdir = exp.get("workdir")
|
|
1472
|
+
if workdir:
|
|
1473
|
+
workspace = _inline_workspace(
|
|
1474
|
+
workdir, self._stage_artifacts_dir(stage), run_dir=self.run_dir,
|
|
1475
|
+
)
|
|
1476
|
+
|
|
1477
|
+
entry: dict[str, Any] = {
|
|
1478
|
+
"ts": self._ts(), "cycle": cycle, "module": module,
|
|
1479
|
+
# Joins this experiment back to its M3 hypotheses[] entry (same id).
|
|
1480
|
+
"hypothesis_id": hypothesis_id(hypothesis),
|
|
1481
|
+
"hypothesis": hypothesis.statement,
|
|
1482
|
+
"failure_mode": hypothesis.predicted_failure_mode,
|
|
1483
|
+
"status": iv.status.value if iv.status else None,
|
|
1484
|
+
"fixed": iv.fixed,
|
|
1485
|
+
"provider": exp.get("provider"), "verdict": exp.get("verdict"),
|
|
1486
|
+
"metrics": exp.get("metrics"), "returncode": exp.get("returncode"),
|
|
1487
|
+
"timed_out": exp.get("timed_out"), "cli_usage": exp.get("cli_usage"),
|
|
1488
|
+
"llm_calls": exp.get("llm_calls"), "sandbox_runs": exp.get("sandbox_runs"),
|
|
1489
|
+
"code": files,
|
|
1490
|
+
"stdout": exp.get("stdout"), "stderr": exp.get("stderr"),
|
|
1491
|
+
"blueprint": exp.get("blueprint"), "cli_raw_output": exp.get("cli_raw_output"),
|
|
1492
|
+
"validation_log": list(exp.get("validation_log") or []) or None,
|
|
1493
|
+
"workspace_snapshot": workspace,
|
|
1494
|
+
"trial_root": exp.get("trial_root"),
|
|
1495
|
+
}
|
|
1496
|
+
self._append_stage(stage, "experiment", entry)
|
|
1497
|
+
for call in exp.get("model_calls") or []:
|
|
1498
|
+
if isinstance(call, dict):
|
|
1499
|
+
self.log_model_exchange(
|
|
1500
|
+
stage,
|
|
1501
|
+
role=str(call.get("role") or "experiment_writer"),
|
|
1502
|
+
operation=str(call.get("operation") or "generate"),
|
|
1503
|
+
inputs=call.get("inputs"), output=call.get("output"),
|
|
1504
|
+
error=call.get("error"), duration_sec=call.get("duration_sec"),
|
|
1505
|
+
metadata=call.get("metadata"),
|
|
1506
|
+
)
|
|
1507
|
+
|
|
1508
|
+
exp_span = self.tracer.start_span(
|
|
1509
|
+
name=f"{stage}: Experiment — {hypothesis.statement[:60]}",
|
|
1510
|
+
stage=f"{stage}_EXPERIMENT",
|
|
1511
|
+
input_data={"hypothesis": hypothesis.statement, "failure_mode": hypothesis.predicted_failure_mode},
|
|
1512
|
+
metadata={"provider": exp.get("provider"), "returncode": exp.get("returncode")},
|
|
1513
|
+
)
|
|
1514
|
+
cli_raw = exp.get("cli_raw_output")
|
|
1515
|
+
if cli_raw:
|
|
1516
|
+
self.tracer.log_generation(
|
|
1517
|
+
name=f"{module.upper()} Coder Agent", model=str(exp.get("provider") or "coder_agent"),
|
|
1518
|
+
prompt=None, completion=str(cli_raw), span_id=exp_span,
|
|
1519
|
+
)
|
|
1520
|
+
vlog = exp.get("validation_log")
|
|
1521
|
+
if vlog:
|
|
1522
|
+
self.tracer.log_generation(
|
|
1523
|
+
name=f"{module.upper()} Validation Log", model=str(exp.get("provider") or "coder_agent"),
|
|
1524
|
+
prompt=None, completion="\n".join(str(x) for x in vlog), span_id=exp_span,
|
|
1525
|
+
)
|
|
1526
|
+
self.tracer.end_span(exp_span, output_data={
|
|
1527
|
+
"status": entry.get("status"), "fixed": entry.get("fixed"), "verdict": entry.get("verdict"),
|
|
1528
|
+
})
|
|
1529
|
+
|
|
1530
|
+
def log_fix(self, outcome: "Any") -> None:
|
|
1531
|
+
"""Post-loop fix module: the tiered repair attempt + recommendation.
|
|
1532
|
+
|
|
1533
|
+
Per-case outputs are NOT popped out to a sibling ``outputs.jsonl`` —
|
|
1534
|
+
they stay inline in ``M5/log.json`` under each candidate's own
|
|
1535
|
+
``"outputs"`` key. Fewer files was the whole point; a bulkier single
|
|
1536
|
+
JSON is the intended trade for that.
|
|
1537
|
+
"""
|
|
1538
|
+
d = json.loads(json.dumps(outcome.to_dict(), ensure_ascii=False, default=str))
|
|
1539
|
+
for attempt in [*(d.get("attempted") or []), *(d.get("selection_attempted") or [])]:
|
|
1540
|
+
trial_root = attempt.get("trial_root")
|
|
1541
|
+
if trial_root:
|
|
1542
|
+
attempt["workspace_snapshot"] = _inline_workspace(
|
|
1543
|
+
trial_root, self._stage_artifacts_dir("M5"),
|
|
1544
|
+
run_dir=self.run_dir, max_bytes=None, preserve_all=True,
|
|
1545
|
+
)
|
|
1546
|
+
best_ref = d.get("best")
|
|
1547
|
+
if isinstance(best_ref, dict):
|
|
1548
|
+
best = best_ref
|
|
1549
|
+
elif isinstance(best_ref, str):
|
|
1550
|
+
best = next((a for a in d.get("attempted") or [] if a.get("name") == best_ref), {})
|
|
1551
|
+
else:
|
|
1552
|
+
best = {}
|
|
1553
|
+
entry: dict[str, Any] = {"ts": self._ts(), "cycle": -1, "module": "fix"}
|
|
1554
|
+
entry.update(d)
|
|
1555
|
+
entry["best"] = best
|
|
1556
|
+
self._append_stage("M5", "fix", entry)
|
|
1557
|
+
|
|
1558
|
+
fix_span = self.tracer.start_span(
|
|
1559
|
+
name="M5: Targeted Repair & Confirmation", stage="M5_FIX",
|
|
1560
|
+
input_data={"candidates_evaluated": len(d.get("attempted", []))},
|
|
1561
|
+
)
|
|
1562
|
+
if best.get("effect") is not None:
|
|
1563
|
+
self.tracer.log_score(
|
|
1564
|
+
name="repair_net_accuracy_gain", value=float(best.get("effect", 0.0)),
|
|
1565
|
+
comment=f"cured={best.get('n_fixed')}, broken={best.get('n_broken')}",
|
|
1566
|
+
span_id=fix_span,
|
|
1567
|
+
)
|
|
1568
|
+
if (best.get("payload") or {}).get("prompt_template"):
|
|
1569
|
+
self.tracer.log_generation(
|
|
1570
|
+
name="Winning Repair Patch", model="evalrx_repair",
|
|
1571
|
+
prompt="Repair Candidate Search",
|
|
1572
|
+
completion=best["payload"]["prompt_template"], span_id=fix_span,
|
|
1573
|
+
)
|
|
1574
|
+
self.tracer.end_span(fix_span, output_data={"selected": best.get("name")})
|
|
1575
|
+
|
|
1576
|
+
# ------------------------------------------------------------------
|
|
1577
|
+
# AgenticDiagnoseLoop dispatch layer — run-level, not stage content
|
|
1578
|
+
# ------------------------------------------------------------------
|
|
1579
|
+
|
|
1580
|
+
def log_agent_decision(
|
|
1581
|
+
self, step: int, *, action: str, params: "dict[str, Any] | None" = None,
|
|
1582
|
+
rationale: str = "", valid: bool = True, repair_attempts: int = 0,
|
|
1583
|
+
fallback_used: bool = False, judge_prompt: "str | None" = None,
|
|
1584
|
+
judge_raw: "str | None" = None, duration_sec: "float | None" = None,
|
|
1585
|
+
judge_calls: "list[dict[str, Any]] | None" = None,
|
|
1586
|
+
) -> None:
|
|
1587
|
+
entry: dict[str, Any] = {
|
|
1588
|
+
"ts": self._ts(), "step": step, "action": action, "params": params or {},
|
|
1589
|
+
"rationale": rationale, "valid": valid, "repair_attempts": repair_attempts,
|
|
1590
|
+
"fallback_used": fallback_used,
|
|
1591
|
+
}
|
|
1592
|
+
if judge_prompt:
|
|
1593
|
+
entry["judge_prompt"] = judge_prompt
|
|
1594
|
+
if judge_raw:
|
|
1595
|
+
entry["judge_response"] = judge_raw
|
|
1596
|
+
if judge_calls:
|
|
1597
|
+
entry["model_calls"] = json.loads(json.dumps(judge_calls, default=str))
|
|
1598
|
+
if duration_sec is not None:
|
|
1599
|
+
entry["duration_sec"] = round(duration_sec, 3)
|
|
1600
|
+
self._append_run("agent_decisions", entry)
|
|
1601
|
+
|
|
1602
|
+
decision_span = self.tracer.start_span(
|
|
1603
|
+
name=f"Agent Decision: step {step}", stage="AGENT_DECISION",
|
|
1604
|
+
input_data={"action": action, "params": params or {}},
|
|
1605
|
+
metadata={"valid": valid, "repair_attempts": repair_attempts, "fallback_used": fallback_used},
|
|
1606
|
+
)
|
|
1607
|
+
if judge_prompt or judge_raw:
|
|
1608
|
+
self.tracer.log_generation(
|
|
1609
|
+
name="Agent Decision Judge", model="judge",
|
|
1610
|
+
prompt=judge_prompt or "", completion=judge_raw or "", span_id=decision_span,
|
|
1611
|
+
)
|
|
1612
|
+
self.tracer.end_span(decision_span, output_data={"action": action, "rationale": rationale})
|
|
1613
|
+
|
|
1614
|
+
def log_agent_tool(
|
|
1615
|
+
self, step: int, *, tool: str, ok: bool, summary: str = "",
|
|
1616
|
+
error: "str | None" = None, duration_sec: "float | None" = None,
|
|
1617
|
+
) -> None:
|
|
1618
|
+
entry: dict[str, Any] = {
|
|
1619
|
+
"ts": self._ts(), "step": step, "tool": tool, "ok": ok, "summary": summary,
|
|
1620
|
+
}
|
|
1621
|
+
if error is not None:
|
|
1622
|
+
entry["error"] = error
|
|
1623
|
+
if duration_sec is not None:
|
|
1624
|
+
entry["duration_sec"] = round(duration_sec, 3)
|
|
1625
|
+
self._append_run("agent_tool_calls", entry)
|
|
1626
|
+
|
|
1627
|
+
def log_stage_skipped(self, stage: str, reason_code: str, *, cycle: int = -1, detail: str = "") -> None:
|
|
1628
|
+
entry = {"ts": self._ts(), "stage": stage, "cycle": cycle, "reason_code": reason_code, "detail": detail}
|
|
1629
|
+
self._append_stage(stage, "stage_skipped", entry)
|
|
1630
|
+
span = self.tracer.start_span(
|
|
1631
|
+
name=f"{stage}: skipped", stage=stage,
|
|
1632
|
+
input_data={"reason_code": reason_code}, metadata={"detail": detail},
|
|
1633
|
+
)
|
|
1634
|
+
self.tracer.end_span(span, output_data={"reason_code": reason_code}, status="skipped")
|
|
1635
|
+
|
|
1636
|
+
def log_loop_end(
|
|
1637
|
+
self, report: "AutoDiagnoseReport", *,
|
|
1638
|
+
tokens_used: "int | None" = None, timings: "dict[str, float] | None" = None,
|
|
1639
|
+
) -> None:
|
|
1640
|
+
entry: dict[str, Any] = {"ts": self._ts(), "cycles": report.cycles}
|
|
1641
|
+
if tokens_used is not None:
|
|
1642
|
+
entry["tokens_used"] = tokens_used
|
|
1643
|
+
if timings:
|
|
1644
|
+
entry["timings_sec"] = {k: round(v, 3) for k, v in timings.items()}
|
|
1645
|
+
entry["total_duration_sec"] = round(sum(timings.values()), 3)
|
|
1646
|
+
if hasattr(report, "resolved"):
|
|
1647
|
+
entry["resolved"] = report.resolved
|
|
1648
|
+
hyps = getattr(report, "final_hypotheses", [])
|
|
1649
|
+
entry["n_hypotheses"] = len(hyps)
|
|
1650
|
+
entry["final_hypotheses"] = [
|
|
1651
|
+
{"statement": h.statement, "plain_statement": h.plain_statement,
|
|
1652
|
+
"failure_mode": h.predicted_failure_mode, "status": h.status.value if h.status else None}
|
|
1653
|
+
for h in hyps
|
|
1654
|
+
]
|
|
1655
|
+
if hasattr(report, "stopped_by"):
|
|
1656
|
+
entry["stopped_by"] = report.stopped_by
|
|
1657
|
+
all_hyps = getattr(report, "all_hypotheses", [])
|
|
1658
|
+
verified = getattr(report, "verified_hypotheses", [])
|
|
1659
|
+
entry["n_hypotheses"] = len(all_hyps)
|
|
1660
|
+
entry["n_verified"] = len(verified)
|
|
1661
|
+
entry["verified_hypotheses"] = [
|
|
1662
|
+
{"statement": tr.hypothesis.statement,
|
|
1663
|
+
"failure_mode": tr.hypothesis.predicted_failure_mode,
|
|
1664
|
+
"status": tr.status.value,
|
|
1665
|
+
"confidence": tr.confidence,
|
|
1666
|
+
"protocol_consistent": tr.is_consistent_with_protocol,
|
|
1667
|
+
"verdict": getattr(tr, "verdict", None)}
|
|
1668
|
+
for tr in verified
|
|
1669
|
+
]
|
|
1670
|
+
self._append_run("loop_end", entry)
|
|
1671
|
+
|
|
1672
|
+
# ------------------------------------------------------------------
|
|
1673
|
+
# Tool synthesis (M1/M2 probes+stats tools, generated on demand)
|
|
1674
|
+
# ------------------------------------------------------------------
|
|
1675
|
+
|
|
1676
|
+
def log_tool_codegen(
|
|
1677
|
+
self, *, module: str, name: str, need: str, source: str, ok: bool,
|
|
1678
|
+
code: str = "", prompt: str = "", raw_output: str = "", raw_stream: str = "",
|
|
1679
|
+
error: str = "", stdout: str = "", cycle: "int | None" = None,
|
|
1680
|
+
extra: "dict[str, Any] | None" = None,
|
|
1681
|
+
) -> None:
|
|
1682
|
+
cyc = self.current_cycle if cycle is None else cycle
|
|
1683
|
+
entry: dict[str, Any] = {
|
|
1684
|
+
"ts": self._ts(), "cycle": cyc, "module": module, "tool_name": name,
|
|
1685
|
+
"need": need, "source": source, "ok": ok, "error": error or None,
|
|
1686
|
+
"code": code or None, "prompt": prompt or None,
|
|
1687
|
+
"raw_output": raw_output or None, "raw_stream": raw_stream or None,
|
|
1688
|
+
"stdout": stdout or None,
|
|
1689
|
+
}
|
|
1690
|
+
if extra:
|
|
1691
|
+
entry.update(extra)
|
|
1692
|
+
self._append_stage(module, "tool_codegen", entry)
|
|
1693
|
+
if prompt or raw_stream or raw_output:
|
|
1694
|
+
self.log_model_exchange(
|
|
1695
|
+
_resolve_stage(module) or "M5", role="tool_codegen",
|
|
1696
|
+
operation=source, inputs=prompt or "",
|
|
1697
|
+
output=raw_stream or raw_output or code,
|
|
1698
|
+
error=error or None, cycle=cyc,
|
|
1699
|
+
metadata={"tool_name": name, "ok": ok},
|
|
1700
|
+
)
|
|
1701
|
+
|
|
1702
|
+
cg_span = self.tracer.start_span(
|
|
1703
|
+
name=f"Tool Codegen: {module}/{name}", stage=f"CODEGEN_{module}",
|
|
1704
|
+
input_data={"need": need, "source": source},
|
|
1705
|
+
metadata={"ok": ok, "error": error or None},
|
|
1706
|
+
)
|
|
1707
|
+
if prompt or raw_output:
|
|
1708
|
+
self.tracer.log_generation(
|
|
1709
|
+
name=f"Tool Synthesis: {name}", model=source,
|
|
1710
|
+
prompt=prompt or "", completion=raw_stream or raw_output or code, span_id=cg_span,
|
|
1711
|
+
)
|
|
1712
|
+
self.tracer.end_span(cg_span, output_data={"ok": ok, "code_chars": len(code or "")})
|
|
1713
|
+
|
|
1714
|
+
def log_tool_registry(self, cycle: int, module: str, generated: "list[Any]") -> None:
|
|
1715
|
+
if not generated:
|
|
1716
|
+
return
|
|
1717
|
+
tools = [
|
|
1718
|
+
{
|
|
1719
|
+
"name": getattr(g, "name", "tool"), "need": getattr(g, "need", ""),
|
|
1720
|
+
"source": getattr(g, "source", ""), "code": getattr(g, "code", "") or None,
|
|
1721
|
+
}
|
|
1722
|
+
for g in generated
|
|
1723
|
+
]
|
|
1724
|
+
self._append_stage(module, "tool_registry", {
|
|
1725
|
+
"ts": self._ts(), "cycle": cycle, "module": module, "n_tools": len(tools), "tools": tools,
|
|
1726
|
+
})
|
|
1727
|
+
|
|
1728
|
+
# ------------------------------------------------------------------
|
|
1729
|
+
# Lifecycle
|
|
1730
|
+
# ------------------------------------------------------------------
|
|
1731
|
+
|
|
1732
|
+
def close(self) -> None:
|
|
1733
|
+
"""Flush every doc one last time and persist the Langfuse trace bundle."""
|
|
1734
|
+
if self._closed:
|
|
1735
|
+
return
|
|
1736
|
+
with self._lock:
|
|
1737
|
+
self._flush_run()
|
|
1738
|
+
for stage in _STAGES:
|
|
1739
|
+
self._flush_stage(stage)
|
|
1740
|
+
self.tracer.end_trace({
|
|
1741
|
+
"spans": len(self.tracer.spans),
|
|
1742
|
+
"generations": len(self.tracer.generations),
|
|
1743
|
+
"scores": len(self.tracer.scores),
|
|
1744
|
+
})
|
|
1745
|
+
self.tracer.flush()
|
|
1746
|
+
try:
|
|
1747
|
+
self.tracer.export_bundle(self.run_dir / "langfuse_trace.json")
|
|
1748
|
+
except Exception: # noqa: BLE001
|
|
1749
|
+
pass
|
|
1750
|
+
# Offline runs need no retry queue once the complete JSON trace bundle
|
|
1751
|
+
# has been exported. A live run keeps a non-empty queue for retry.
|
|
1752
|
+
if self.tracer.mode == "offline" or self.tracer.outbox.pending_count() == 0:
|
|
1753
|
+
try:
|
|
1754
|
+
self._outbox_path.unlink(missing_ok=True)
|
|
1755
|
+
self._outbox_path.parent.rmdir()
|
|
1756
|
+
except OSError:
|
|
1757
|
+
pass
|
|
1758
|
+
self._closed = True
|
|
1759
|
+
|
|
1760
|
+
def __enter__(self) -> "RunLoggerV2":
|
|
1761
|
+
return self
|
|
1762
|
+
|
|
1763
|
+
def __exit__(self, *exc_info: Any) -> None:
|
|
1764
|
+
self.close()
|