evalrx 0.1.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- evalrx/__init__.py +139 -0
- evalrx/agent_assets/__init__.py +2 -0
- evalrx/agent_assets/skills/README.md +28 -0
- evalrx/agent_assets/skills/eval-chart-style/SKILL.md +172 -0
- evalrx/agent_assets/skills/evalrx-report-ui/SKILL.md +116 -0
- evalrx/agent_assets/skills/nature-figure/LICENSE +201 -0
- evalrx/agent_assets/skills/nature-figure/README.md +412 -0
- evalrx/agent_assets/skills/nature-figure/SKILL.md +60 -0
- evalrx/agent_assets/skills/nature-figure/manifest.yaml +59 -0
- evalrx/agent_assets/skills/nature-figure/references/api.md +436 -0
- evalrx/agent_assets/skills/nature-figure/references/backend-selection.md +100 -0
- evalrx/agent_assets/skills/nature-figure/references/chart-types.md +281 -0
- evalrx/agent_assets/skills/nature-figure/references/common-patterns.md +350 -0
- evalrx/agent_assets/skills/nature-figure/references/demos.md +65 -0
- evalrx/agent_assets/skills/nature-figure/references/design-theory.md +439 -0
- evalrx/agent_assets/skills/nature-figure/references/figure-contract.md +93 -0
- evalrx/agent_assets/skills/nature-figure/references/figure-legend-conventions.md +71 -0
- evalrx/agent_assets/skills/nature-figure/references/nature-2026-observations.md +112 -0
- evalrx/agent_assets/skills/nature-figure/references/qa-contract.md +119 -0
- evalrx/agent_assets/skills/nature-figure/references/r-template-index.md +66 -0
- evalrx/agent_assets/skills/nature-figure/references/r-workflow.md +161 -0
- evalrx/agent_assets/skills/nature-figure/references/tutorials.md +251 -0
- evalrx/agent_assets/skills/nature-figure/static/core/contract.md +29 -0
- evalrx/agent_assets/skills/nature-figure/static/core/stance.md +37 -0
- evalrx/agent_assets/skills/nature-figure/static/fragments/backend/python.md +37 -0
- evalrx/agent_assets/skills/nature-figure/static/fragments/backend/r.md +44 -0
- evalrx/agent_assets/skills/outcome-driver-analysis/SKILL.md +213 -0
- evalrx/agent_assets/skills/outcome-driver-analysis/assets/analysis_report_template.md +53 -0
- evalrx/agent_assets/skills/outcome-driver-analysis/references/model_selection.md +72 -0
- evalrx/agent_assets/skills/outcome-driver-analysis/scripts/explanatory_var_eda.R +130 -0
- evalrx/agent_assets/skills/outcome-driver-analysis/scripts/explanatory_var_eda.py +150 -0
- evalrx/agent_assets/skills/outcome-driver-analysis/scripts/fit_outcome_model.R +181 -0
- evalrx/agent_assets/skills/outcome-driver-analysis/scripts/fit_outcome_model.py +186 -0
- evalrx/agent_assets/skills/outcome-driver-analysis/scripts/univariate_eda.R +149 -0
- evalrx/agent_assets/skills/outcome-driver-analysis/scripts/univariate_eda.py +177 -0
- evalrx/agent_assets/skills.py +27 -0
- evalrx/agent_runtime/__init__.py +78 -0
- evalrx/agent_runtime/_docker_runner.py +89 -0
- evalrx/agent_runtime/cli_runtime.py +103 -0
- evalrx/agent_runtime/cli_transcript.py +138 -0
- evalrx/agent_runtime/cli_types.py +68 -0
- evalrx/agent_runtime/codegen/__init__.py +5 -0
- evalrx/agent_runtime/codegen/runner.py +94 -0
- evalrx/agent_runtime/experiment_harness.py +117 -0
- evalrx/agent_runtime/factory.py +102 -0
- evalrx/agent_runtime/json_shape.py +44 -0
- evalrx/agent_runtime/judges/__init__.py +28 -0
- evalrx/agent_runtime/judges/agy.py +179 -0
- evalrx/agent_runtime/judges/autodetect.py +135 -0
- evalrx/agent_runtime/judges/claude.py +159 -0
- evalrx/agent_runtime/judges/codex.py +120 -0
- evalrx/agent_runtime/providers/__init__.py +21 -0
- evalrx/agent_runtime/providers/antigravity.py +31 -0
- evalrx/agent_runtime/providers/base.py +145 -0
- evalrx/agent_runtime/providers/claude_code.py +49 -0
- evalrx/agent_runtime/providers/codex.py +37 -0
- evalrx/agent_runtime/providers/gemini_cli.py +26 -0
- evalrx/agent_runtime/providers/kimi_cli.py +27 -0
- evalrx/agent_runtime/providers/opencode.py +27 -0
- evalrx/agent_runtime/providers/registry.py +58 -0
- evalrx/agent_runtime/sandbox.py +517 -0
- evalrx/agent_runtime/skill_audit.py +143 -0
- evalrx/agent_runtime/skills/__init__.py +19 -0
- evalrx/agent_runtime/skills/installer.py +68 -0
- evalrx/agent_runtime/skills/prompt_policy.py +86 -0
- evalrx/agent_runtime/skills/resolver.py +19 -0
- evalrx/analysis/__init__.py +132 -0
- evalrx/analysis/adjudicate.py +154 -0
- evalrx/analysis/analysis_module.py +361 -0
- evalrx/analysis/api.py +171 -0
- evalrx/analysis/case_studio.py +651 -0
- evalrx/analysis/cli.py +114 -0
- evalrx/analysis/dashboard.py +350 -0
- evalrx/analysis/eval_case_matrix.py +118 -0
- evalrx/analysis/eval_viz_theme.py +833 -0
- evalrx/analysis/explore_run.py +333 -0
- evalrx/analysis/explorer.py +1276 -0
- evalrx/analysis/failure_modes.py +607 -0
- evalrx/analysis/fused_pipeline.py +489 -0
- evalrx/analysis/holdout.py +300 -0
- evalrx/analysis/hypothesis_agent.py +230 -0
- evalrx/analysis/narration.py +177 -0
- evalrx/analysis/operationalize.py +442 -0
- evalrx/analysis/plain_language.py +42 -0
- evalrx/analysis/planner.py +283 -0
- evalrx/analysis/probe_search.py +203 -0
- evalrx/analysis/profile.py +268 -0
- evalrx/analysis/prompts/__init__.py +0 -0
- evalrx/analysis/prompts/explorer.py +417 -0
- evalrx/analysis/prompts/failure_modes.py +33 -0
- evalrx/analysis/prompts/holdout.py +27 -0
- evalrx/analysis/prompts/hypothesis_agent.py +78 -0
- evalrx/analysis/prompts/run_codebase.py +47 -0
- evalrx/analysis/prompts/stats_agent.py +72 -0
- evalrx/analysis/prompts/stats_tool_generator.py +43 -0
- evalrx/analysis/result_marker.py +47 -0
- evalrx/analysis/run_codebase.py +242 -0
- evalrx/analysis/run_view.py +205 -0
- evalrx/analysis/stage_views.py +93 -0
- evalrx/analysis/stats_agent.py +944 -0
- evalrx/analysis/stats_tool_agent.py +261 -0
- evalrx/analysis/stats_tool_generator.py +415 -0
- evalrx/analysis/stats_tools.py +1153 -0
- evalrx/analysis/trajectory_records.py +193 -0
- evalrx/analysis/workbench.py +431 -0
- evalrx/analyzers/__init__.py +42 -0
- evalrx/analyzers/agent/__init__.py +25 -0
- evalrx/analyzers/agent/counterfactual.py +84 -0
- evalrx/analyzers/agent/first_error_judge.py +96 -0
- evalrx/analyzers/agent/ignored_obs.py +81 -0
- evalrx/analyzers/agent/loop_detect.py +79 -0
- evalrx/analyzers/agent/reliability.py +165 -0
- evalrx/analyzers/agent/tool_shap.py +225 -0
- evalrx/analyzers/agent/trajectory_rubric.py +168 -0
- evalrx/analyzers/attention/__init__.py +19 -0
- evalrx/analyzers/attention/relative_attn.py +610 -0
- evalrx/analyzers/attention/rollout.py +73 -0
- evalrx/analyzers/attention/sink.py +56 -0
- evalrx/analyzers/attention/summary.py +190 -0
- evalrx/analyzers/attribution/__init__.py +6 -0
- evalrx/analyzers/attribution/generic_attn.py +31 -0
- evalrx/analyzers/attribution/gradcam.py +30 -0
- evalrx/analyzers/base.py +12 -0
- evalrx/analyzers/geometry/__init__.py +6 -0
- evalrx/analyzers/geometry/cka.py +70 -0
- evalrx/analyzers/geometry/linear_probe.py +157 -0
- evalrx/analyzers/hallucination/__init__.py +9 -0
- evalrx/analyzers/hallucination/chair.py +78 -0
- evalrx/analyzers/hallucination/opera.py +29 -0
- evalrx/analyzers/hallucination/pope.py +119 -0
- evalrx/analyzers/hallucination/selfcheck.py +155 -0
- evalrx/analyzers/hallucination/vcd.py +29 -0
- evalrx/analyzers/lens/__init__.py +7 -0
- evalrx/analyzers/lens/layer_contrast.py +133 -0
- evalrx/analyzers/lens/logit_lens.py +138 -0
- evalrx/analyzers/lens/tuned_lens.py +30 -0
- evalrx/analyzers/patching/__init__.py +5 -0
- evalrx/analyzers/patching/causal_trace.py +30 -0
- evalrx/analyzers/perturbation/__init__.py +23 -0
- evalrx/analyzers/perturbation/_shapley.py +54 -0
- evalrx/analyzers/perturbation/context_shap.py +174 -0
- evalrx/analyzers/perturbation/cot_faithfulness.py +239 -0
- evalrx/analyzers/perturbation/format_sensitivity.py +237 -0
- evalrx/analyzers/perturbation/mm_shap.py +146 -0
- evalrx/analyzers/perturbation/modality_ablation.py +196 -0
- evalrx/analyzers/perturbation/perturbation_battery.py +274 -0
- evalrx/analyzers/perturbation/prompt_contrast.py +265 -0
- evalrx/analyzers/perturbation/rise.py +94 -0
- evalrx/analyzers/perturbation/vl_shap.py +102 -0
- evalrx/analyzers/reasoning/__init__.py +33 -0
- evalrx/analyzers/reasoning/_text.py +328 -0
- evalrx/analyzers/reasoning/answer_extraction_audit.py +327 -0
- evalrx/analyzers/reasoning/arith_audit.py +226 -0
- evalrx/analyzers/reasoning/contamination.py +214 -0
- evalrx/analyzers/reasoning/knowledge_split.py +253 -0
- evalrx/analyzers/reasoning/self_repair.py +246 -0
- evalrx/analyzers/reasoning/step_rollout_value.py +216 -0
- evalrx/analyzers/reasoning/termination_audit.py +258 -0
- evalrx/analyzers/uncertainty/__init__.py +18 -0
- evalrx/analyzers/uncertainty/calibration.py +174 -0
- evalrx/analyzers/uncertainty/coverage_gap.py +199 -0
- evalrx/analyzers/uncertainty/entropy.py +90 -0
- evalrx/analyzers/uncertainty/logprob_entropy.py +69 -0
- evalrx/analyzers/uncertainty/self_consistency.py +204 -0
- evalrx/analyzers/uncertainty/verbalized_conf.py +64 -0
- evalrx/cli.py +411 -0
- evalrx/config.py +77 -0
- evalrx/contract/__init__.py +179 -0
- evalrx/contract/common.py +452 -0
- evalrx/contract/emit.py +948 -0
- evalrx/contract/export.py +237 -0
- evalrx/contract/m1.py +325 -0
- evalrx/contract/m2.py +317 -0
- evalrx/contract/m3.py +165 -0
- evalrx/contract/m4.py +130 -0
- evalrx/contract/m5.py +292 -0
- evalrx/contract/methodology.py +76 -0
- evalrx/contract/pre_m1.py +58 -0
- evalrx/contract/typescript.py +140 -0
- evalrx/core/__init__.py +85 -0
- evalrx/core/analyzer.py +174 -0
- evalrx/core/capability.py +54 -0
- evalrx/core/case.py +443 -0
- evalrx/core/experiment.py +106 -0
- evalrx/core/model.py +198 -0
- evalrx/core/pipeline.py +42 -0
- evalrx/core/registry.py +142 -0
- evalrx/core/result.py +64 -0
- evalrx/core/spec.py +173 -0
- evalrx/core/tokentype.py +165 -0
- evalrx/core/tool.py +92 -0
- evalrx/datasets/__init__.py +41 -0
- evalrx/datasets/base.py +68 -0
- evalrx/datasets/gui_os.py +52 -0
- evalrx/datasets/llm_qa.py +57 -0
- evalrx/datasets/pure_qa.py +12 -0
- evalrx/datasets/vlm_qa.py +695 -0
- evalrx/datasets/web_search_qa.py +52 -0
- evalrx/eval_agent/__init__.py +341 -0
- evalrx/eval_agent/_tools.py +81 -0
- evalrx/eval_agent/ab_runner.py +50 -0
- evalrx/eval_agent/agentic/__init__.py +43 -0
- evalrx/eval_agent/agentic/actions.py +216 -0
- evalrx/eval_agent/agentic/board.py +107 -0
- evalrx/eval_agent/agentic/loop.py +190 -0
- evalrx/eval_agent/agentic/tools.py +538 -0
- evalrx/eval_agent/checkpoint.py +57 -0
- evalrx/eval_agent/cli_agent.py +59 -0
- evalrx/eval_agent/cli_skills.py +5 -0
- evalrx/eval_agent/evolution.py +396 -0
- evalrx/eval_agent/git_manager.py +215 -0
- evalrx/eval_agent/hypothesis.py +172 -0
- evalrx/eval_agent/label_quarantine.py +209 -0
- evalrx/eval_agent/legacy.py +530 -0
- evalrx/eval_agent/log_schema.py +497 -0
- evalrx/eval_agent/loop.py +2159 -0
- evalrx/eval_agent/loop_reports.py +116 -0
- evalrx/eval_agent/model_instrumentation.py +282 -0
- evalrx/eval_agent/narration.py +193 -0
- evalrx/eval_agent/nl_runner.py +460 -0
- evalrx/eval_agent/orchestrator.py +61 -0
- evalrx/eval_agent/preregister.py +93 -0
- evalrx/eval_agent/prompts/__init__.py +1 -0
- evalrx/eval_agent/prompts/agentic.py +46 -0
- evalrx/eval_agent/prompts/case_discovery.py +25 -0
- evalrx/eval_agent/prompts/diagnosis.py +125 -0
- evalrx/eval_agent/prompts/experiment_writer.py +265 -0
- evalrx/eval_agent/prompts/explore_step.py +37 -0
- evalrx/eval_agent/prompts/fix_agent.py +257 -0
- evalrx/eval_agent/prompts/hypothesis_tester.py +15 -0
- evalrx/eval_agent/prompts/nl_runner.py +38 -0
- evalrx/eval_agent/prompts/probe_agent.py +25 -0
- evalrx/eval_agent/prompts/probe_candidate_generator.py +14 -0
- evalrx/eval_agent/prompts/probe_generator.py +35 -0
- evalrx/eval_agent/prompts/whitebox_probe_generator.py +38 -0
- evalrx/eval_agent/report.py +58 -0
- evalrx/eval_agent/run_context.py +354 -0
- evalrx/eval_agent/run_log.schema.json +1215 -0
- evalrx/eval_agent/run_logger_v2.py +1764 -0
- evalrx/eval_agent/run_metadata.py +208 -0
- evalrx/eval_agent/stages/__init__.py +56 -0
- evalrx/eval_agent/stages/case_discovery.py +293 -0
- evalrx/eval_agent/stages/diagnosis.py +1017 -0
- evalrx/eval_agent/stages/experiment_writer.py +1634 -0
- evalrx/eval_agent/stages/fix_agent.py +3916 -0
- evalrx/eval_agent/stages/fix_internals.py +499 -0
- evalrx/eval_agent/stages/fix_pipeline.py +725 -0
- evalrx/eval_agent/stages/fix_tiers.py +187 -0
- evalrx/eval_agent/stages/fix_tools.py +1034 -0
- evalrx/eval_agent/stages/hypothesis_tester.py +1014 -0
- evalrx/eval_agent/stages/probe.py +439 -0
- evalrx/eval_agent/stages/probe_agent.py +1079 -0
- evalrx/eval_agent/stages/probe_candidate_generator.py +128 -0
- evalrx/eval_agent/stages/probe_generator.py +326 -0
- evalrx/eval_agent/stages/probe_search_agent.py +106 -0
- evalrx/eval_agent/stages/protocol.py +112 -0
- evalrx/eval_agent/stages/repair_catalog.py +273 -0
- evalrx/eval_agent/stages/surgery.py +524 -0
- evalrx/eval_agent/stages/whitebox_probe_generator.py +351 -0
- evalrx/eval_agent/store.py +231 -0
- evalrx/logging_utils.py +112 -0
- evalrx/models/__init__.py +161 -0
- evalrx/models/_discover.py +101 -0
- evalrx/models/agent.py +380 -0
- evalrx/models/backends/__init__.py +58 -0
- evalrx/models/backends/api.py +169 -0
- evalrx/models/backends/base.py +57 -0
- evalrx/models/backends/gemini_compat.py +579 -0
- evalrx/models/backends/hf_local.py +2074 -0
- evalrx/models/backends/openai_compat.py +301 -0
- evalrx/models/backends/vllm_offline.py +116 -0
- evalrx/models/base.py +24 -0
- evalrx/models/blackbox/__init__.py +4 -0
- evalrx/models/blackbox/agent.py +31 -0
- evalrx/models/blackbox/base.py +29 -0
- evalrx/models/blackbox/gemini.py +279 -0
- evalrx/models/blackbox/llm_api.py +17 -0
- evalrx/models/blackbox/vlm_api.py +17 -0
- evalrx/models/compose.py +66 -0
- evalrx/models/inference.py +88 -0
- evalrx/models/paper_methods/__init__.py +8 -0
- evalrx/models/paper_methods/aad.py +53 -0
- evalrx/models/paper_methods/ifcd.py +204 -0
- evalrx/models/paper_methods/pai.py +164 -0
- evalrx/models/paper_methods/tcd.py +202 -0
- evalrx/models/paper_methods/vcd.py +45 -0
- evalrx/models/paper_methods/vicrop.py +137 -0
- evalrx/models/toolcodec.py +143 -0
- evalrx/models/tools/__init__.py +20 -0
- evalrx/models/tools/perception.py +300 -0
- evalrx/models/tools/visual.py +174 -0
- evalrx/models/whitebox/__init__.py +26 -0
- evalrx/models/whitebox/agent.py +31 -0
- evalrx/models/whitebox/base.py +24 -0
- evalrx/models/whitebox/qwen.py +61 -0
- evalrx/models/whitebox/qwen2_5_omni.py +29 -0
- evalrx/models/whitebox/qwen2_audio.py +25 -0
- evalrx/models/whitebox/qwen_omni.py +53 -0
- evalrx/models/whitebox/qwen_vl.py +62 -0
- evalrx/observability/__init__.py +21 -0
- evalrx/observability/envelope.py +122 -0
- evalrx/observability/outbox.py +111 -0
- evalrx/observability/tracer.py +882 -0
- evalrx/reporting/__init__.py +28 -0
- evalrx/reporting/case_study.py +947 -0
- evalrx/reporting/compiler.py +587 -0
- evalrx/reporting/dynamic.py +1882 -0
- evalrx/reporting/html_report.py +2225 -0
- evalrx/reporting/langfuse_exporter.py +38 -0
- evalrx/reporting/langfuse_source.py +155 -0
- evalrx/reporting/model.py +151 -0
- evalrx/reporting/run_events.py +184 -0
- evalrx/reporting/server.py +557 -0
- evalrx/reporting/stages.py +58 -0
- evalrx/reporting/static_export.py +142 -0
- evalrx/reporting/web_dist/index.html +146 -0
- evalrx/specs.py +727 -0
- evalrx/stats/__init__.py +47 -0
- evalrx/stats/api.py +192 -0
- evalrx/stats/bootstrap.py +86 -0
- evalrx/stats/ebh.py +27 -0
- evalrx/stats/evalue.py +98 -0
- evalrx/stats/friedman.py +138 -0
- evalrx/stats/mcnemar.py +40 -0
- evalrx/stats/multiplicity.py +159 -0
- evalrx/stats/subset_sampling.py +55 -0
- evalrx/term_links.py +43 -0
- evalrx/viz/__init__.py +7 -0
- evalrx/viz/labels.py +77 -0
- evalrx/viz/prompts.py +39 -0
- evalrx/viz/renderer.py +590 -0
- evalrx/viz/schema.py +36 -0
- evalrx/viz/style.py +134 -0
- evalrx-0.1.2.dist-info/METADATA +532 -0
- evalrx-0.1.2.dist-info/RECORD +339 -0
- evalrx-0.1.2.dist-info/WHEEL +5 -0
- evalrx-0.1.2.dist-info/entry_points.txt +3 -0
- evalrx-0.1.2.dist-info/licenses/LICENSE +121 -0
- evalrx-0.1.2.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,882 @@
|
|
|
1
|
+
"""DiagnosticTracer — Native Langfuse & OpenTelemetry Tracing Engine for EvalRX.
|
|
2
|
+
|
|
3
|
+
This module provides first-class observability for model evaluations and agentic diagnostics:
|
|
4
|
+
1. Live streaming to a Langfuse server (self-hosted or cloud) via the Langfuse Python SDK (v3/v4 API).
|
|
5
|
+
2. Standardized JSON Trace Bundle export (OpenTelemetry-compatible) for offline auditing —
|
|
6
|
+
this file is the durable source of truth and works with no SDK installed.
|
|
7
|
+
3. Multi-tier hierarchical span recording:
|
|
8
|
+
- Root Trace: Benchmark run session (model metadata, benchmark parameters, dataset fingerprint).
|
|
9
|
+
- Pipeline Stage Spans: PRE-M1, M1 Checkup, M2 Screening, M3 Diagnosis, M4 Adjudication, M5 Repair,
|
|
10
|
+
plus Explore and Tool-Codegen (the coder-agent trajectories).
|
|
11
|
+
- Probe Execution Spans: Granular probe runs with findings and artifact paths.
|
|
12
|
+
- AI Doctor Generations: LLM reasoning, screening evidence digest, and falsifiable hypothesis generation.
|
|
13
|
+
- Repair Generations: Candidate patch searches, prompt templates, and McNemar confirmation scores.
|
|
14
|
+
|
|
15
|
+
Live-sync contract:
|
|
16
|
+
- Set ``EVALRX_LANGFUSE_MODE=live`` plus LANGFUSE_PUBLIC_KEY /
|
|
17
|
+
LANGFUSE_SECRET_KEY (and LANGFUSE_HOST for self-hosted) for a user-facing
|
|
18
|
+
run. Missing credentials then fail fast. ``auto`` retains compatibility for
|
|
19
|
+
local development, and ``offline`` is an explicit no-upload choice.
|
|
20
|
+
- The Langfuse trace id is the logger's trace_id (a UUID, dashes stripped) so
|
|
21
|
+
the live trace, the exported bundle and run.json all agree on identity.
|
|
22
|
+
- Every live failure is reported ONCE on stderr instead of being silently swallowed — an
|
|
23
|
+
observability layer that fails quietly is worse than none.
|
|
24
|
+
"""
|
|
25
|
+
|
|
26
|
+
from __future__ import annotations
|
|
27
|
+
|
|
28
|
+
import json
|
|
29
|
+
import os
|
|
30
|
+
import sys
|
|
31
|
+
import time
|
|
32
|
+
import uuid
|
|
33
|
+
from pathlib import Path
|
|
34
|
+
from typing import Any
|
|
35
|
+
|
|
36
|
+
from evalrx.observability.envelope import artifact_manifests_for_event, make_event_envelope
|
|
37
|
+
from evalrx.observability.outbox import ObservabilityOutbox
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
class DiagnosticTracer:
|
|
41
|
+
"""Manages active tracing context across the diagnostic pipeline.
|
|
42
|
+
|
|
43
|
+
All records are accumulated in memory and exported verbatim via
|
|
44
|
+
:meth:`export_bundle`; the optional Langfuse client mirrors them live.
|
|
45
|
+
"""
|
|
46
|
+
|
|
47
|
+
def __init__(
|
|
48
|
+
self,
|
|
49
|
+
run_dir: str | Path | None = None,
|
|
50
|
+
auto_sync: bool = False,
|
|
51
|
+
mode: str | None = None,
|
|
52
|
+
outbox_path: str | Path | None = None,
|
|
53
|
+
):
|
|
54
|
+
self.run_dir = Path(run_dir).resolve() if run_dir else None
|
|
55
|
+
self.auto_sync = auto_sync
|
|
56
|
+
self.trace_id = f"evalrx_{uuid.uuid4().hex[:12]}"
|
|
57
|
+
self.spans: list[dict[str, Any]] = []
|
|
58
|
+
self.generations: list[dict[str, Any]] = []
|
|
59
|
+
self.scores: list[dict[str, Any]] = []
|
|
60
|
+
self.events: list[dict[str, Any]] = []
|
|
61
|
+
self.trace_metadata: dict[str, Any] = {}
|
|
62
|
+
self._start_time = time.time()
|
|
63
|
+
requested_mode = (mode or os.getenv("EVALRX_LANGFUSE_MODE", "auto")).strip().lower()
|
|
64
|
+
if requested_mode not in {"auto", "live", "offline"}:
|
|
65
|
+
raise ValueError("EVALRX_LANGFUSE_MODE must be one of: auto, live, offline")
|
|
66
|
+
self.mode = requested_mode
|
|
67
|
+
self._langfuse_client = None
|
|
68
|
+
self._live_root = None # root "chain" observation for the run
|
|
69
|
+
self._live_obs: dict[str, Any] = {} # our span_id -> live observation wrapper
|
|
70
|
+
self._warned: set[str] = set()
|
|
71
|
+
# Every structured run-log event enters this durable queue before live
|
|
72
|
+
# delivery. It is intentionally kept separate from the run's own
|
|
73
|
+
# log documents so delivery retries never mutate the run record.
|
|
74
|
+
default_outbox = (
|
|
75
|
+
(self.run_dir / ".evalrx" / "langfuse_outbox.sqlite3")
|
|
76
|
+
if self.run_dir is not None
|
|
77
|
+
else Path(".evalrx-langfuse-outbox.sqlite3")
|
|
78
|
+
)
|
|
79
|
+
self.outbox = ObservabilityOutbox(outbox_path or default_outbox)
|
|
80
|
+
|
|
81
|
+
has_credentials = bool(os.getenv("LANGFUSE_PUBLIC_KEY") and os.getenv("LANGFUSE_SECRET_KEY"))
|
|
82
|
+
if requested_mode == "live" and not has_credentials:
|
|
83
|
+
raise RuntimeError(
|
|
84
|
+
"Live Langfuse was requested but LANGFUSE_PUBLIC_KEY and "
|
|
85
|
+
"LANGFUSE_SECRET_KEY are not configured. Set EVALRX_LANGFUSE_MODE=offline "
|
|
86
|
+
"only for an explicitly offline run."
|
|
87
|
+
)
|
|
88
|
+
if requested_mode != "offline" and has_credentials:
|
|
89
|
+
try:
|
|
90
|
+
from langfuse import Langfuse
|
|
91
|
+
|
|
92
|
+
self._langfuse_client = Langfuse()
|
|
93
|
+
except Exception as exc:
|
|
94
|
+
self._warn(
|
|
95
|
+
"langfuse_init",
|
|
96
|
+
f"LANGFUSE keys are set but the Langfuse client failed to initialize; "
|
|
97
|
+
f"continuing with local-only tracing. Error: {exc}",
|
|
98
|
+
)
|
|
99
|
+
|
|
100
|
+
# ------------------------------------------------------------------
|
|
101
|
+
# Helpers
|
|
102
|
+
# ------------------------------------------------------------------
|
|
103
|
+
|
|
104
|
+
def _warn(self, key: str, msg: str) -> None:
|
|
105
|
+
"""Report a live-sync problem once per key; silent afterwards."""
|
|
106
|
+
if key in self._warned:
|
|
107
|
+
return
|
|
108
|
+
self._warned.add(key)
|
|
109
|
+
print(f"[langfuse] {msg}", file=sys.stderr)
|
|
110
|
+
|
|
111
|
+
@property
|
|
112
|
+
def langfuse_trace_id(self) -> str:
|
|
113
|
+
"""Langfuse requires a 32-hex trace id; our trace_id is a UUID."""
|
|
114
|
+
return self.trace_id.replace("-", "")
|
|
115
|
+
|
|
116
|
+
def _apply_tags(self, tags: list[str]) -> None:
|
|
117
|
+
"""Best-effort trace tags (private SDK helper; cosmetic if unavailable)."""
|
|
118
|
+
if self._langfuse_client is None:
|
|
119
|
+
return
|
|
120
|
+
try:
|
|
121
|
+
fn = getattr(self._langfuse_client, "_create_trace_tags_via_ingestion", None)
|
|
122
|
+
if fn is not None:
|
|
123
|
+
fn(trace_id=self.langfuse_trace_id, tags=[str(t) for t in tags if t])
|
|
124
|
+
except Exception:
|
|
125
|
+
pass # tags are cosmetic; never fail the run on them
|
|
126
|
+
|
|
127
|
+
def record_event(self, event: dict[str, Any], *, event_seq: int) -> None:
|
|
128
|
+
"""Queue one complete EvalRX event for durable Langfuse delivery.
|
|
129
|
+
|
|
130
|
+
This is called after the JSONL write has received its timestamp and
|
|
131
|
+
trace id. The deterministic event id makes repeated process startup
|
|
132
|
+
and manual backfill safe.
|
|
133
|
+
"""
|
|
134
|
+
artifact_refs = (
|
|
135
|
+
artifact_manifests_for_event(event, run_dir=self.run_dir)
|
|
136
|
+
if self.run_dir is not None
|
|
137
|
+
else []
|
|
138
|
+
)
|
|
139
|
+
envelope = make_event_envelope(
|
|
140
|
+
event, trace_id=self.trace_id, event_seq=event_seq, artifact_refs=artifact_refs,
|
|
141
|
+
)
|
|
142
|
+
self.events.append(envelope)
|
|
143
|
+
self.outbox.enqueue(envelope)
|
|
144
|
+
# A live diagnostic should be inspectable while it is running. The
|
|
145
|
+
# SQLite outbox remains the durability boundary: a transient upload
|
|
146
|
+
# failure simply leaves this event pending for the next stage/close.
|
|
147
|
+
if self.auto_sync and self._langfuse_client is not None:
|
|
148
|
+
self.flush()
|
|
149
|
+
|
|
150
|
+
def _publish_event(self, envelope: dict[str, Any]) -> None:
|
|
151
|
+
"""Publish a queued event as a native Langfuse EVENT observation."""
|
|
152
|
+
if self._langfuse_client is None:
|
|
153
|
+
raise RuntimeError("Langfuse client is not configured")
|
|
154
|
+
# Attach audit events to the run chain rather than emitting a flat trace
|
|
155
|
+
# row. The deterministic event id still makes retry/backfill safe.
|
|
156
|
+
input_data: dict[str, Any] = {"event": envelope["payload"]}
|
|
157
|
+
media = self._media_payload(envelope)
|
|
158
|
+
if media:
|
|
159
|
+
input_data["artifacts"] = media
|
|
160
|
+
metadata = {
|
|
161
|
+
"evalrx_schema_version": envelope["schema_version"],
|
|
162
|
+
"event_id": envelope["event_id"],
|
|
163
|
+
"event_seq": envelope["event_seq"],
|
|
164
|
+
"stage": envelope["stage"],
|
|
165
|
+
"cycle": envelope.get("cycle"),
|
|
166
|
+
"span_id": envelope["payload"].get("span_id"),
|
|
167
|
+
"artifact_refs": envelope["artifact_refs"],
|
|
168
|
+
}
|
|
169
|
+
parent = self._live_root
|
|
170
|
+
if parent is not None:
|
|
171
|
+
observation = parent.start_observation(
|
|
172
|
+
name=f"EvalRX {envelope['stage']}: {envelope['event_type']}",
|
|
173
|
+
as_type="event", input=input_data, metadata=metadata,
|
|
174
|
+
)
|
|
175
|
+
observation.end()
|
|
176
|
+
return
|
|
177
|
+
self._langfuse_client.create_event(
|
|
178
|
+
trace_context={"trace_id": self.langfuse_trace_id},
|
|
179
|
+
name=f"EvalRX {envelope['stage']}: {envelope['event_type']}",
|
|
180
|
+
input=input_data, metadata=metadata,
|
|
181
|
+
)
|
|
182
|
+
|
|
183
|
+
def _media_payload(self, envelope: dict[str, Any]) -> list[dict[str, Any]]:
|
|
184
|
+
"""Attach supported files as Langfuse Media while retaining every manifest.
|
|
185
|
+
|
|
186
|
+
Non-displayable files (for example ``.npy`` tensors) are uploaded as
|
|
187
|
+
``application/octet-stream``. They still remain downloadable and
|
|
188
|
+
checksum-addressable even when the Langfuse UI cannot preview them.
|
|
189
|
+
"""
|
|
190
|
+
if self.run_dir is None:
|
|
191
|
+
return []
|
|
192
|
+
try:
|
|
193
|
+
from langfuse import LangfuseMedia
|
|
194
|
+
from langfuse.api.media.types.media_content_type import MediaContentType
|
|
195
|
+
except ImportError:
|
|
196
|
+
return []
|
|
197
|
+
result: list[dict[str, Any]] = []
|
|
198
|
+
for manifest in envelope["artifact_refs"]:
|
|
199
|
+
path = self.run_dir / str(manifest["path"])
|
|
200
|
+
if path.is_file():
|
|
201
|
+
try:
|
|
202
|
+
content_type = MediaContentType(str(manifest["mime_type"]))
|
|
203
|
+
except ValueError:
|
|
204
|
+
content_type = MediaContentType.APPLICATION_OCTET_STREAM
|
|
205
|
+
result.append({
|
|
206
|
+
"role": manifest["role"],
|
|
207
|
+
"artifact_id": manifest["artifact_id"],
|
|
208
|
+
"content": LangfuseMedia(file_path=str(path), content_type=content_type),
|
|
209
|
+
})
|
|
210
|
+
return result
|
|
211
|
+
|
|
212
|
+
# ------------------------------------------------------------------
|
|
213
|
+
# Trace / span / generation / score recording
|
|
214
|
+
# ------------------------------------------------------------------
|
|
215
|
+
|
|
216
|
+
def start_trace(self, model: str, benchmark: str, n_cases: int = 0, metadata: dict[str, Any] | None = None) -> None:
|
|
217
|
+
"""Initialize the root trace."""
|
|
218
|
+
self.trace_metadata = {
|
|
219
|
+
"model": model,
|
|
220
|
+
"benchmark": benchmark,
|
|
221
|
+
"n_cases": n_cases,
|
|
222
|
+
"created_at": time.time(),
|
|
223
|
+
**(metadata or {}),
|
|
224
|
+
}
|
|
225
|
+
if self._langfuse_client is not None:
|
|
226
|
+
try:
|
|
227
|
+
self._live_root = self._langfuse_client.start_observation(
|
|
228
|
+
trace_context={"trace_id": self.langfuse_trace_id},
|
|
229
|
+
name=f"EvalRX: {model} on {benchmark}",
|
|
230
|
+
as_type="chain",
|
|
231
|
+
input=self.trace_metadata,
|
|
232
|
+
metadata=self.trace_metadata,
|
|
233
|
+
)
|
|
234
|
+
self._apply_tags(["evalrx", model, benchmark])
|
|
235
|
+
except Exception as exc:
|
|
236
|
+
self._warn("start_trace", f"failed to create Langfuse root trace: {exc}")
|
|
237
|
+
|
|
238
|
+
def start_span(
|
|
239
|
+
self,
|
|
240
|
+
name: str,
|
|
241
|
+
stage: str,
|
|
242
|
+
input_data: Any = None,
|
|
243
|
+
metadata: dict[str, Any] | None = None,
|
|
244
|
+
parent_id: str | None = None,
|
|
245
|
+
) -> str:
|
|
246
|
+
"""Create a new pipeline or probe span (nested under *parent_id* when given)."""
|
|
247
|
+
span_id = f"{self.trace_id}_{stage.lower()}_{uuid.uuid4().hex[:6]}"
|
|
248
|
+
span_rec = {
|
|
249
|
+
"id": span_id,
|
|
250
|
+
"trace_id": self.trace_id,
|
|
251
|
+
"parent_id": parent_id,
|
|
252
|
+
"name": name,
|
|
253
|
+
"stage": stage,
|
|
254
|
+
"start_time": time.time(),
|
|
255
|
+
"input": input_data,
|
|
256
|
+
"metadata": metadata or {},
|
|
257
|
+
"status": "running",
|
|
258
|
+
}
|
|
259
|
+
self.spans.append(span_rec)
|
|
260
|
+
if self._langfuse_client is not None:
|
|
261
|
+
try:
|
|
262
|
+
parent_obs = self._live_obs.get(parent_id) if parent_id else None
|
|
263
|
+
if parent_obs is None:
|
|
264
|
+
parent_obs = self._live_root
|
|
265
|
+
if parent_obs is not None:
|
|
266
|
+
obs = parent_obs.start_observation(
|
|
267
|
+
name=name, input=input_data, metadata=metadata or {}
|
|
268
|
+
)
|
|
269
|
+
else:
|
|
270
|
+
obs = self._langfuse_client.start_observation(
|
|
271
|
+
trace_context={"trace_id": self.langfuse_trace_id},
|
|
272
|
+
name=name,
|
|
273
|
+
input=input_data,
|
|
274
|
+
metadata=metadata or {},
|
|
275
|
+
)
|
|
276
|
+
self._live_obs[span_id] = obs
|
|
277
|
+
except Exception as exc:
|
|
278
|
+
self._warn("start_span", f"failed to create Langfuse span {name!r}: {exc}")
|
|
279
|
+
return span_id
|
|
280
|
+
|
|
281
|
+
def end_span(self, span_id: str, output_data: Any = None, status: str = "completed") -> None:
|
|
282
|
+
"""Mark a span as ended (locally and, when live, on Langfuse)."""
|
|
283
|
+
for s in self.spans:
|
|
284
|
+
if s["id"] == span_id:
|
|
285
|
+
s["end_time"] = time.time()
|
|
286
|
+
s["duration_sec"] = s["end_time"] - s["start_time"]
|
|
287
|
+
s["output"] = output_data
|
|
288
|
+
s["status"] = status
|
|
289
|
+
break
|
|
290
|
+
obs = self._live_obs.pop(span_id, None)
|
|
291
|
+
if obs is not None:
|
|
292
|
+
try:
|
|
293
|
+
if output_data is not None:
|
|
294
|
+
obs.update(output=output_data)
|
|
295
|
+
obs.end()
|
|
296
|
+
except Exception as exc:
|
|
297
|
+
self._warn("end_span", f"failed to end Langfuse span {span_id!r}: {exc}")
|
|
298
|
+
|
|
299
|
+
def log_generation(
|
|
300
|
+
self,
|
|
301
|
+
name: str,
|
|
302
|
+
model: str,
|
|
303
|
+
prompt: Any,
|
|
304
|
+
completion: Any,
|
|
305
|
+
span_id: str | None = None,
|
|
306
|
+
metadata: dict[str, Any] | None = None,
|
|
307
|
+
usage: dict[str, int] | None = None,
|
|
308
|
+
) -> str:
|
|
309
|
+
"""Record an LLM agent / diagnostician generation."""
|
|
310
|
+
gen_id = f"gen_{uuid.uuid4().hex[:8]}"
|
|
311
|
+
gen_rec = {
|
|
312
|
+
"id": gen_id,
|
|
313
|
+
"trace_id": self.trace_id,
|
|
314
|
+
"span_id": span_id,
|
|
315
|
+
"name": name,
|
|
316
|
+
"model": model,
|
|
317
|
+
"prompt": prompt,
|
|
318
|
+
"completion": completion,
|
|
319
|
+
"usage": usage or {},
|
|
320
|
+
"metadata": metadata or {},
|
|
321
|
+
"timestamp": time.time(),
|
|
322
|
+
}
|
|
323
|
+
self.generations.append(gen_rec)
|
|
324
|
+
if self._langfuse_client is not None:
|
|
325
|
+
try:
|
|
326
|
+
parent = self._live_obs.get(span_id) if span_id else None
|
|
327
|
+
if parent is None:
|
|
328
|
+
parent = self._live_root
|
|
329
|
+
if parent is not None:
|
|
330
|
+
gen = parent.start_observation(
|
|
331
|
+
name=name,
|
|
332
|
+
as_type="generation",
|
|
333
|
+
model=model,
|
|
334
|
+
input=prompt,
|
|
335
|
+
output=completion,
|
|
336
|
+
metadata=metadata or {},
|
|
337
|
+
)
|
|
338
|
+
else:
|
|
339
|
+
gen = self._langfuse_client.start_observation(
|
|
340
|
+
trace_context={"trace_id": self.langfuse_trace_id},
|
|
341
|
+
name=name,
|
|
342
|
+
as_type="generation",
|
|
343
|
+
model=model,
|
|
344
|
+
input=prompt,
|
|
345
|
+
output=completion,
|
|
346
|
+
metadata=metadata or {},
|
|
347
|
+
)
|
|
348
|
+
gen.end()
|
|
349
|
+
except Exception as exc:
|
|
350
|
+
self._warn("log_generation", f"failed to create Langfuse generation {name!r}: {exc}")
|
|
351
|
+
return gen_id
|
|
352
|
+
|
|
353
|
+
def log_score(
|
|
354
|
+
self,
|
|
355
|
+
name: str,
|
|
356
|
+
value: float,
|
|
357
|
+
comment: str = "",
|
|
358
|
+
span_id: str | None = None,
|
|
359
|
+
) -> None:
|
|
360
|
+
"""Attach a quantitative metric or evaluation score to trace/span."""
|
|
361
|
+
score_rec = {
|
|
362
|
+
"name": name,
|
|
363
|
+
"value": float(value),
|
|
364
|
+
"comment": comment,
|
|
365
|
+
"span_id": span_id,
|
|
366
|
+
"timestamp": time.time(),
|
|
367
|
+
}
|
|
368
|
+
self.scores.append(score_rec)
|
|
369
|
+
if self._langfuse_client is not None:
|
|
370
|
+
obs = self._live_obs.get(span_id) if span_id else None
|
|
371
|
+
try:
|
|
372
|
+
if obs is not None:
|
|
373
|
+
obs.score(name=name, value=float(value), comment=comment)
|
|
374
|
+
else:
|
|
375
|
+
self._langfuse_client.create_score(
|
|
376
|
+
trace_id=self.langfuse_trace_id,
|
|
377
|
+
name=name,
|
|
378
|
+
value=float(value),
|
|
379
|
+
comment=comment,
|
|
380
|
+
)
|
|
381
|
+
except Exception as exc:
|
|
382
|
+
self._warn("log_score", f"failed to create Langfuse score {name!r}: {exc}")
|
|
383
|
+
|
|
384
|
+
def export_bundle(self, out_path: str | Path | None = None) -> dict[str, Any]:
|
|
385
|
+
"""Export the full trace hierarchy to a clean JSON bundle."""
|
|
386
|
+
bundle = {
|
|
387
|
+
"trace": {
|
|
388
|
+
"id": self.trace_id,
|
|
389
|
+
"name": f"EvalRX: {self.trace_metadata.get('model', 'Model')}",
|
|
390
|
+
"metadata": self.trace_metadata,
|
|
391
|
+
"duration_sec": time.time() - self._start_time,
|
|
392
|
+
},
|
|
393
|
+
"spans": self.spans,
|
|
394
|
+
"generations": self.generations,
|
|
395
|
+
"scores": self.scores,
|
|
396
|
+
"events": self.events,
|
|
397
|
+
}
|
|
398
|
+
if out_path:
|
|
399
|
+
p = Path(out_path)
|
|
400
|
+
p.parent.mkdir(parents=True, exist_ok=True)
|
|
401
|
+
p.write_text(json.dumps(bundle, indent=2, ensure_ascii=False, default=str), encoding="utf-8")
|
|
402
|
+
return bundle
|
|
403
|
+
|
|
404
|
+
def flush(self) -> None:
|
|
405
|
+
"""Flush any queued live observations to Langfuse."""
|
|
406
|
+
if self._langfuse_client is not None:
|
|
407
|
+
try:
|
|
408
|
+
published, failed = self.outbox.drain(self._publish_event)
|
|
409
|
+
if failed:
|
|
410
|
+
self._warn(
|
|
411
|
+
"outbox_delivery",
|
|
412
|
+
f"{failed} Langfuse event(s) remain in the local outbox for retry.",
|
|
413
|
+
)
|
|
414
|
+
self._langfuse_client.flush()
|
|
415
|
+
except Exception as exc:
|
|
416
|
+
self._warn("flush", f"failed to flush Langfuse client: {exc}")
|
|
417
|
+
|
|
418
|
+
@property
|
|
419
|
+
def live_enabled(self) -> bool:
|
|
420
|
+
"""Whether this run has an active Langfuse client (safe for UI provenance)."""
|
|
421
|
+
return self._langfuse_client is not None
|
|
422
|
+
|
|
423
|
+
def end_trace(self, output_data: Any = None) -> None:
|
|
424
|
+
"""Finish the live root observation, if live mirroring was enabled."""
|
|
425
|
+
root, self._live_root = self._live_root, None
|
|
426
|
+
if root is None:
|
|
427
|
+
return
|
|
428
|
+
try:
|
|
429
|
+
if output_data is not None:
|
|
430
|
+
root.update(output=output_data)
|
|
431
|
+
root.end()
|
|
432
|
+
except Exception as exc:
|
|
433
|
+
self._warn("end_trace", f"failed to end Langfuse root trace: {exc}")
|
|
434
|
+
|
|
435
|
+
|
|
436
|
+
# ---------------------------------------------------------------------------
|
|
437
|
+
# Offline (run-directory) bundle export & batch sync
|
|
438
|
+
# ---------------------------------------------------------------------------
|
|
439
|
+
|
|
440
|
+
def _to_float(value: Any) -> "float | None":
|
|
441
|
+
"""Parse a headline value like "12.3%", 12.3 or "1,234" into a float."""
|
|
442
|
+
if value is None:
|
|
443
|
+
return None
|
|
444
|
+
try:
|
|
445
|
+
if isinstance(value, (int, float)):
|
|
446
|
+
return float(value)
|
|
447
|
+
s = str(value).strip().rstrip("%").replace(",", "")
|
|
448
|
+
return float(s)
|
|
449
|
+
except (TypeError, ValueError):
|
|
450
|
+
return None
|
|
451
|
+
|
|
452
|
+
|
|
453
|
+
def _resolve_trace_id(run_dir: Path, fingerprint: str) -> str:
|
|
454
|
+
"""The run's trace id — the SAME uuid used by run.json and langfuse_trace.json.
|
|
455
|
+
|
|
456
|
+
Resolution order: langfuse_trace.json (written live by the logger) → the
|
|
457
|
+
run_start event in run.json → a deterministic 32-hex id derived from the
|
|
458
|
+
data fingerprint (Langfuse requires uuid-shaped trace ids).
|
|
459
|
+
"""
|
|
460
|
+
import hashlib
|
|
461
|
+
|
|
462
|
+
bundle_path = run_dir / "langfuse_trace.json"
|
|
463
|
+
if bundle_path.exists():
|
|
464
|
+
try:
|
|
465
|
+
bundle = json.loads(bundle_path.read_text(encoding="utf-8"))
|
|
466
|
+
tid = (bundle.get("trace") or {}).get("id")
|
|
467
|
+
if tid:
|
|
468
|
+
return str(tid)
|
|
469
|
+
except Exception:
|
|
470
|
+
pass
|
|
471
|
+
try:
|
|
472
|
+
from evalrx.reporting.run_events import read_v2_events
|
|
473
|
+
|
|
474
|
+
start = next(
|
|
475
|
+
(event for event in read_v2_events(run_dir) if event.get("event") == "run_start"),
|
|
476
|
+
{},
|
|
477
|
+
)
|
|
478
|
+
if start.get("trace_id"):
|
|
479
|
+
return str(start["trace_id"])
|
|
480
|
+
except Exception:
|
|
481
|
+
pass
|
|
482
|
+
return hashlib.md5((fingerprint or "evalrx").encode("utf-8")).hexdigest()
|
|
483
|
+
|
|
484
|
+
|
|
485
|
+
def export_to_langfuse_bundle(run_dir: str | Path, out_json: str | Path | None = None) -> dict[str, Any]:
|
|
486
|
+
"""Convert an EvalRX run artifacts directory into a rich Langfuse bundle."""
|
|
487
|
+
from evalrx.reporting.html_report import extract_run_data
|
|
488
|
+
|
|
489
|
+
run_dir = Path(run_dir).resolve()
|
|
490
|
+
data = extract_run_data(run_dir)
|
|
491
|
+
run = data["run"]
|
|
492
|
+
m1 = data["m1"]
|
|
493
|
+
m2 = data["m2"]
|
|
494
|
+
m3 = data["m3"]
|
|
495
|
+
m4 = data["m4"]
|
|
496
|
+
m5_s = data["m5_surgery"]
|
|
497
|
+
m5_f = data["m5_fix"]
|
|
498
|
+
|
|
499
|
+
trace_id = _resolve_trace_id(run_dir, run.get("data_fingerprint") or "")
|
|
500
|
+
trace_name = f"EvalRX: {run['model']} · {run['benchmark_name']}"
|
|
501
|
+
|
|
502
|
+
trace = {
|
|
503
|
+
"id": trace_id,
|
|
504
|
+
"name": trace_name,
|
|
505
|
+
"release": f"v{run['version']}",
|
|
506
|
+
"tags": ["evalrx", run["model"], run["benchmark_name"], run["stopped_by"]],
|
|
507
|
+
"metadata": {
|
|
508
|
+
"model": run["model"],
|
|
509
|
+
"raw_model": run["raw_model"],
|
|
510
|
+
"benchmark": run["benchmark_name"],
|
|
511
|
+
"n_cases": run["n_cases"],
|
|
512
|
+
"cycles": run["cycles"],
|
|
513
|
+
"protocol": run["protocol"],
|
|
514
|
+
"data_fingerprint": run["data_fingerprint"],
|
|
515
|
+
"logs_dir": run["logs_dir"],
|
|
516
|
+
},
|
|
517
|
+
}
|
|
518
|
+
|
|
519
|
+
spans = []
|
|
520
|
+
generations = []
|
|
521
|
+
scores = []
|
|
522
|
+
|
|
523
|
+
# PRE-M1 Span
|
|
524
|
+
if data["pre_m1"]["ran"]:
|
|
525
|
+
spans.append({
|
|
526
|
+
"id": f"{trace_id}_pre_m1",
|
|
527
|
+
"name": "PRE-M1: Case Synthesis & Adversarial Probing",
|
|
528
|
+
"type": "span",
|
|
529
|
+
"metadata": {"stage": "PRE-M1", "n_synthesized": data["pre_m1"]["n_cases"]},
|
|
530
|
+
"input": {"criteria": "adversarial_blindspots"},
|
|
531
|
+
"output": {"synthesized_cases_count": data["pre_m1"]["n_cases"]},
|
|
532
|
+
})
|
|
533
|
+
|
|
534
|
+
# M1 Span & Sub-spans for Probes
|
|
535
|
+
m1_span_id = f"{trace_id}_m1"
|
|
536
|
+
spans.append({
|
|
537
|
+
"id": m1_span_id,
|
|
538
|
+
"name": "M1: Multi-Dimensional Checkup & Signals",
|
|
539
|
+
"type": "span",
|
|
540
|
+
"metadata": {
|
|
541
|
+
"stage": "M1",
|
|
542
|
+
"analyzers": m1["analyzers"],
|
|
543
|
+
"duration_sec": m1["duration"],
|
|
544
|
+
"n_probes": len(m1["results"]),
|
|
545
|
+
},
|
|
546
|
+
"input": {"analyzers": m1["analyzers"]},
|
|
547
|
+
"output": {"summary": [f"{r['name']}: {r['display_name']}" for r in m1["results"]]},
|
|
548
|
+
})
|
|
549
|
+
|
|
550
|
+
# Individual Probe Sub-Spans
|
|
551
|
+
for r in m1["results"]:
|
|
552
|
+
probe_span_id = f"{m1_span_id}_{r['name']}"
|
|
553
|
+
findings = r.get("findings") or {}
|
|
554
|
+
spans.append({
|
|
555
|
+
"id": probe_span_id,
|
|
556
|
+
"parent_id": m1_span_id,
|
|
557
|
+
"name": f"Probe: {r['display_name']} ({r['name']})",
|
|
558
|
+
"type": "span",
|
|
559
|
+
"metadata": {
|
|
560
|
+
"probe_code": r["name"],
|
|
561
|
+
"question": r["question"],
|
|
562
|
+
"n_scored": r["n"],
|
|
563
|
+
"headlines": r["headline"],
|
|
564
|
+
},
|
|
565
|
+
"input": {"probe_name": r["name"], "question": r["question"]},
|
|
566
|
+
"output": findings,
|
|
567
|
+
})
|
|
568
|
+
# If findings contain scalar scores, log them
|
|
569
|
+
for h in r["headline"]:
|
|
570
|
+
value = _to_float(h["value"])
|
|
571
|
+
if value is not None:
|
|
572
|
+
scores.append({
|
|
573
|
+
"name": f"m1_{r['name']}_{h['label'].lower().replace(' ', '_')}",
|
|
574
|
+
"value": value,
|
|
575
|
+
"comment": f"{r['display_name']} - {h['label']}",
|
|
576
|
+
"span_id": probe_span_id,
|
|
577
|
+
})
|
|
578
|
+
|
|
579
|
+
# M2 Span
|
|
580
|
+
m2_span_id = f"{trace_id}_m2"
|
|
581
|
+
spans.append({
|
|
582
|
+
"id": m2_span_id,
|
|
583
|
+
"name": "M2: Screening & Confirmatory Signals (EDA)",
|
|
584
|
+
"type": "span",
|
|
585
|
+
"metadata": {
|
|
586
|
+
"stage": "M2",
|
|
587
|
+
"severity": m2["severity"],
|
|
588
|
+
"duration_sec": m2["duration"],
|
|
589
|
+
"n_tests": len(m2["stats"]),
|
|
590
|
+
"n_rejected": sum(1 for s in m2["stats"] if s["reject"]),
|
|
591
|
+
},
|
|
592
|
+
"input": {"n_signals_screened": len(m2["stats"])},
|
|
593
|
+
"output": {
|
|
594
|
+
"conclusion": m2["conclusion"],
|
|
595
|
+
"significant_signals": [
|
|
596
|
+
s.get("config", {}).get("signal") or s.get("tool")
|
|
597
|
+
for s in m2["stats"] if s["reject"]
|
|
598
|
+
],
|
|
599
|
+
},
|
|
600
|
+
})
|
|
601
|
+
for s in m2["stats"]:
|
|
602
|
+
sig_name = s.get("config", {}).get("signal") or s.get("tool") or "stat_test"
|
|
603
|
+
if s.get("p_value") is not None:
|
|
604
|
+
scores.append({
|
|
605
|
+
"name": f"m2_fdr_p_{sig_name}",
|
|
606
|
+
"value": float(s["p_value"]),
|
|
607
|
+
"comment": f"FDR p-value (reject={s['reject']})",
|
|
608
|
+
"span_id": m2_span_id,
|
|
609
|
+
})
|
|
610
|
+
|
|
611
|
+
# M3 Span & LLM Diagnostician Generation
|
|
612
|
+
m3_span_id = f"{trace_id}_m3"
|
|
613
|
+
spans.append({
|
|
614
|
+
"id": m3_span_id,
|
|
615
|
+
"name": "M3: Root-Cause Diagnosis (AI Doctor)",
|
|
616
|
+
"type": "span",
|
|
617
|
+
"metadata": {
|
|
618
|
+
"stage": "M3",
|
|
619
|
+
"duration_sec": m3["duration"],
|
|
620
|
+
"n_hypotheses": len(m3["hypotheses"]),
|
|
621
|
+
},
|
|
622
|
+
"input": {"screened_anomalies": m2["conclusion"]},
|
|
623
|
+
"output": {"hypotheses": m3["hypotheses"]},
|
|
624
|
+
})
|
|
625
|
+
for idx, h in enumerate(m3["hypotheses"], 1):
|
|
626
|
+
generations.append({
|
|
627
|
+
"id": f"gen_m3_hyp_{idx}",
|
|
628
|
+
"trace_id": trace_id,
|
|
629
|
+
"span_id": m3_span_id,
|
|
630
|
+
"name": f"AI Doctor Diagnosis #{idx}",
|
|
631
|
+
"model": "AI Diagnostician Agent",
|
|
632
|
+
"prompt": f"Analyze screened signals: {m2['conclusion']}",
|
|
633
|
+
"completion": json.dumps(h, indent=2, ensure_ascii=False),
|
|
634
|
+
"metadata": {"failure_mode": h.get("failure_mode")},
|
|
635
|
+
})
|
|
636
|
+
|
|
637
|
+
# M4 Span
|
|
638
|
+
if m4["ran"]:
|
|
639
|
+
spans.append({
|
|
640
|
+
"id": f"{trace_id}_m4",
|
|
641
|
+
"name": "M4: Independent Blind Adjudication",
|
|
642
|
+
"type": "span",
|
|
643
|
+
"metadata": {
|
|
644
|
+
"stage": "M4",
|
|
645
|
+
"results_count": len(m4["results"]),
|
|
646
|
+
},
|
|
647
|
+
"input": {"hypotheses_tested": [h.get("statement") for h in m3["hypotheses"]]},
|
|
648
|
+
"output": {"event": m4["event"], "results": m4["results"]},
|
|
649
|
+
})
|
|
650
|
+
|
|
651
|
+
# M5 Surgery Span
|
|
652
|
+
if m5_s["ran"]:
|
|
653
|
+
spans.append({
|
|
654
|
+
"id": f"{trace_id}_m5_surgery",
|
|
655
|
+
"name": "M5-SURGERY: Causal Mechanism Interventions",
|
|
656
|
+
"type": "span",
|
|
657
|
+
"metadata": {"stage": "M5-Surgery"},
|
|
658
|
+
"output": {"surgeries": m5_s["surgeries"]},
|
|
659
|
+
})
|
|
660
|
+
|
|
661
|
+
# M5 Fix Span & Generations
|
|
662
|
+
if m5_f["ran"]:
|
|
663
|
+
m5_span_id = f"{trace_id}_m5_fix"
|
|
664
|
+
spans.append({
|
|
665
|
+
"id": m5_span_id,
|
|
666
|
+
"name": "M5-FIX: Targeted Repair & Confirmation",
|
|
667
|
+
"type": "span",
|
|
668
|
+
"metadata": {
|
|
669
|
+
"stage": "M5-Fix",
|
|
670
|
+
"fixed": m5_f["fixed"],
|
|
671
|
+
"n_candidates_screened": len(m5_f["selection"]),
|
|
672
|
+
},
|
|
673
|
+
"input": {"candidates": [s.get("name") for s in m5_f["selection"]]},
|
|
674
|
+
"output": {
|
|
675
|
+
# Fix events written before PR #88 carry ``best`` as the candidate
|
|
676
|
+
# NAME, later ones as the candidate record: accept both.
|
|
677
|
+
"best_candidate": (
|
|
678
|
+
(m5_f.get("best") or {}).get("name")
|
|
679
|
+
if isinstance(m5_f.get("best"), dict) else (m5_f.get("best") or None)
|
|
680
|
+
) or (m5_f.get("confirm") or {}).get("name"),
|
|
681
|
+
"confirm": m5_f.get("confirm"),
|
|
682
|
+
},
|
|
683
|
+
})
|
|
684
|
+
if m5_f.get("prompt_template"):
|
|
685
|
+
generations.append({
|
|
686
|
+
"id": "gen_m5_patch",
|
|
687
|
+
"trace_id": trace_id,
|
|
688
|
+
"span_id": m5_span_id,
|
|
689
|
+
"name": "Winning Repair Patch",
|
|
690
|
+
"model": "Repair Search Agent",
|
|
691
|
+
"prompt": "Synthesize targeted prompt patch based on diagnosed mechanism",
|
|
692
|
+
"completion": m5_f["prompt_template"],
|
|
693
|
+
"metadata": {"candidate_name": (m5_f.get("confirm") or {}).get("name")},
|
|
694
|
+
})
|
|
695
|
+
|
|
696
|
+
# Global Key Evaluation Scores
|
|
697
|
+
cfm = m5_f.get("confirm") or {}
|
|
698
|
+
if cfm.get("n_baseline_correct") is not None and cfm.get("n_pairs"):
|
|
699
|
+
scores.append({
|
|
700
|
+
"name": "baseline_accuracy",
|
|
701
|
+
"value": cfm["n_baseline_correct"] / cfm["n_pairs"],
|
|
702
|
+
"comment": f"Baseline Correct: {cfm['n_baseline_correct']}/{cfm['n_pairs']}",
|
|
703
|
+
})
|
|
704
|
+
if cfm.get("effect") is not None:
|
|
705
|
+
scores.append({
|
|
706
|
+
"name": "repair_net_gain",
|
|
707
|
+
"value": float(cfm["effect"]),
|
|
708
|
+
"comment": f"Net accuracy shift: {float(cfm['effect']) * 100:+.2f}%",
|
|
709
|
+
})
|
|
710
|
+
if cfm.get("e_value") is not None:
|
|
711
|
+
scores.append({
|
|
712
|
+
"name": "evidence_strength_e_value",
|
|
713
|
+
"value": float(cfm["e_value"]),
|
|
714
|
+
"comment": "Multiplicity-corrected certainty",
|
|
715
|
+
})
|
|
716
|
+
|
|
717
|
+
bundle = {
|
|
718
|
+
"trace": trace,
|
|
719
|
+
"spans": spans,
|
|
720
|
+
"generations": generations,
|
|
721
|
+
"scores": scores,
|
|
722
|
+
}
|
|
723
|
+
|
|
724
|
+
if out_json:
|
|
725
|
+
out_p = Path(out_json)
|
|
726
|
+
out_p.parent.mkdir(parents=True, exist_ok=True)
|
|
727
|
+
out_p.write_text(json.dumps(bundle, indent=2, ensure_ascii=False, default=str), encoding="utf-8")
|
|
728
|
+
print(f"[✓] Exported Langfuse bundle to: {out_p}")
|
|
729
|
+
|
|
730
|
+
return bundle
|
|
731
|
+
|
|
732
|
+
|
|
733
|
+
def backfill_run_to_langfuse(run_dir: str | Path, *, dry_run: bool = False) -> dict[str, int | str]:
|
|
734
|
+
"""Queue an existing run for the same reliable Langfuse pipeline.
|
|
735
|
+
|
|
736
|
+
Existing ``event_seq`` values are preserved; older logs without one receive
|
|
737
|
+
their line order. Re-running the command is safe because envelope IDs are
|
|
738
|
+
deterministic and the outbox primary key de-duplicates them.
|
|
739
|
+
"""
|
|
740
|
+
from evalrx.reporting.run_events import read_v2_events, resolve_v2_root
|
|
741
|
+
|
|
742
|
+
root = Path(run_dir)
|
|
743
|
+
v2_root = resolve_v2_root(root)
|
|
744
|
+
if v2_root is None:
|
|
745
|
+
raise FileNotFoundError(f"No run bundle (run.json) found under {run_dir}")
|
|
746
|
+
root = v2_root
|
|
747
|
+
records: list[dict[str, Any]] = read_v2_events(root)
|
|
748
|
+
if not records:
|
|
749
|
+
return {"trace_id": "", "events": 0, "pending": 0, "published": 0}
|
|
750
|
+
|
|
751
|
+
trace_id = str(next((r.get("trace_id") for r in records if r.get("trace_id")), uuid.uuid4()))
|
|
752
|
+
if dry_run:
|
|
753
|
+
return {"trace_id": trace_id, "events": len(records), "pending": len(records), "published": 0}
|
|
754
|
+
|
|
755
|
+
tracer = DiagnosticTracer(run_dir=root)
|
|
756
|
+
tracer.trace_id = trace_id
|
|
757
|
+
run_start = next((r for r in records if r.get("event") == "run_start"), {})
|
|
758
|
+
tracer.start_trace(
|
|
759
|
+
model=str(run_start.get("model") or "Target Model"),
|
|
760
|
+
benchmark=str(run_start.get("benchmark_name") or "Benchmark"),
|
|
761
|
+
n_cases=int(run_start.get("n_cases") or 0),
|
|
762
|
+
metadata=run_start,
|
|
763
|
+
)
|
|
764
|
+
for fallback_seq, record in enumerate(records, start=1):
|
|
765
|
+
tracer.record_event(record, event_seq=int(record.get("event_seq") or fallback_seq))
|
|
766
|
+
pending_before = tracer.outbox.pending_count()
|
|
767
|
+
tracer.flush()
|
|
768
|
+
pending_after = tracer.outbox.pending_count()
|
|
769
|
+
tracer.end_trace({"backfilled_events": len(records)})
|
|
770
|
+
tracer.flush()
|
|
771
|
+
return {
|
|
772
|
+
"trace_id": trace_id,
|
|
773
|
+
"events": len(records),
|
|
774
|
+
"pending": pending_after,
|
|
775
|
+
"published": max(0, pending_before - pending_after),
|
|
776
|
+
}
|
|
777
|
+
|
|
778
|
+
|
|
779
|
+
def sync_to_langfuse_live(run_dir: str | Path) -> bool:
|
|
780
|
+
"""If langfuse SDK is installed and credentials exist, push live to Langfuse.
|
|
781
|
+
|
|
782
|
+
Uses the run's own trace id (from langfuse_trace.json / run.json) so a
|
|
783
|
+
batch re-sync lands on the SAME trace the live mirroring wrote to, rather
|
|
784
|
+
than creating a duplicate.
|
|
785
|
+
"""
|
|
786
|
+
try:
|
|
787
|
+
from langfuse import Langfuse
|
|
788
|
+
except ImportError:
|
|
789
|
+
print("[!] `langfuse` package is not installed. Install with `pip install langfuse` to enable live sync.")
|
|
790
|
+
return False
|
|
791
|
+
|
|
792
|
+
bundle = export_to_langfuse_bundle(run_dir)
|
|
793
|
+
trace_info = bundle["trace"]
|
|
794
|
+
lf_trace_id = str(trace_info["id"]).replace("-", "")
|
|
795
|
+
|
|
796
|
+
def _tags(client: "Langfuse", tags: list) -> None:
|
|
797
|
+
try:
|
|
798
|
+
fn = getattr(client, "_create_trace_tags_via_ingestion", None)
|
|
799
|
+
if fn is not None:
|
|
800
|
+
fn(trace_id=lf_trace_id, tags=[str(t) for t in tags if t])
|
|
801
|
+
except Exception:
|
|
802
|
+
pass
|
|
803
|
+
|
|
804
|
+
try:
|
|
805
|
+
langfuse = Langfuse()
|
|
806
|
+
root = langfuse.start_observation(
|
|
807
|
+
trace_context={"trace_id": lf_trace_id},
|
|
808
|
+
name=trace_info["name"],
|
|
809
|
+
as_type="chain",
|
|
810
|
+
input=trace_info.get("metadata"),
|
|
811
|
+
metadata=trace_info.get("metadata"),
|
|
812
|
+
)
|
|
813
|
+
_tags(langfuse, trace_info.get("tags") or [])
|
|
814
|
+
|
|
815
|
+
obs_map: dict[str, Any] = {}
|
|
816
|
+
for span in bundle["spans"]:
|
|
817
|
+
parent = obs_map.get(span.get("parent_id")) or root
|
|
818
|
+
try:
|
|
819
|
+
obs = parent.start_observation(
|
|
820
|
+
name=span["name"],
|
|
821
|
+
input=span.get("input"),
|
|
822
|
+
output=span.get("output"),
|
|
823
|
+
metadata=span.get("metadata"),
|
|
824
|
+
)
|
|
825
|
+
except AttributeError:
|
|
826
|
+
obs = langfuse.start_observation(
|
|
827
|
+
trace_context={"trace_id": lf_trace_id},
|
|
828
|
+
name=span["name"],
|
|
829
|
+
input=span.get("input"),
|
|
830
|
+
output=span.get("output"),
|
|
831
|
+
metadata=span.get("metadata"),
|
|
832
|
+
)
|
|
833
|
+
obs_map[span["id"]] = obs
|
|
834
|
+
obs.end()
|
|
835
|
+
|
|
836
|
+
for gen in bundle.get("generations", []):
|
|
837
|
+
parent = obs_map.get(gen.get("span_id")) or root
|
|
838
|
+
try:
|
|
839
|
+
g = parent.start_observation(
|
|
840
|
+
name=gen["name"],
|
|
841
|
+
as_type="generation",
|
|
842
|
+
model=gen.get("model"),
|
|
843
|
+
input=gen.get("prompt"),
|
|
844
|
+
output=gen.get("completion"),
|
|
845
|
+
metadata=gen.get("metadata"),
|
|
846
|
+
)
|
|
847
|
+
except AttributeError:
|
|
848
|
+
g = langfuse.start_observation(
|
|
849
|
+
trace_context={"trace_id": lf_trace_id},
|
|
850
|
+
name=gen["name"],
|
|
851
|
+
as_type="generation",
|
|
852
|
+
model=gen.get("model"),
|
|
853
|
+
input=gen.get("prompt"),
|
|
854
|
+
output=gen.get("completion"),
|
|
855
|
+
metadata=gen.get("metadata"),
|
|
856
|
+
)
|
|
857
|
+
g.end()
|
|
858
|
+
|
|
859
|
+
for score in bundle.get("scores", []):
|
|
860
|
+
obs = obs_map.get(score.get("span_id"))
|
|
861
|
+
if obs is not None:
|
|
862
|
+
obs.score(
|
|
863
|
+
name=score["name"],
|
|
864
|
+
value=score["value"],
|
|
865
|
+
comment=score.get("comment", ""),
|
|
866
|
+
)
|
|
867
|
+
else:
|
|
868
|
+
langfuse.create_score(
|
|
869
|
+
trace_id=lf_trace_id,
|
|
870
|
+
name=score["name"],
|
|
871
|
+
value=score["value"],
|
|
872
|
+
comment=score.get("comment", ""),
|
|
873
|
+
)
|
|
874
|
+
|
|
875
|
+
root.update(output={"spans": len(bundle["spans"]), "generations": len(bundle.get("generations", []))})
|
|
876
|
+
root.end()
|
|
877
|
+
langfuse.flush()
|
|
878
|
+
print(f"[✓] Successfully pushed trace to Langfuse dashboard: {trace_info['name']}")
|
|
879
|
+
return True
|
|
880
|
+
except Exception as e:
|
|
881
|
+
print(f"[!] Langfuse sync failed: {e}")
|
|
882
|
+
return False
|