clear-eval 2.0.1__tar.gz → 2.0.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- clear_eval-2.0.2/PKG-INFO +248 -0
- clear_eval-2.0.2/README.md +207 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/pyproject.toml +1 -1
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/agentic/dashboard/launch_dashboard.py +18 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/agentic/pipeline/full_traces_evaluation/run_trajectory_evaluation_pipeline.py +4 -9
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/agentic/pipeline/run_clear_agentic_eval.py +18 -13
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/agentic/pipeline/run_clear_step_analysis.py +4 -8
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/agentic/pipeline/setup/default_agentic_config.yaml +1 -1
- clear_eval-2.0.2/src/clear_eval.egg-info/PKG-INFO +248 -0
- clear_eval-2.0.1/PKG-INFO +0 -276
- clear_eval-2.0.1/README.md +0 -235
- clear_eval-2.0.1/src/clear_eval.egg-info/PKG-INFO +0 -276
- {clear_eval-2.0.1 → clear_eval-2.0.2}/LICENSE +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/setup.cfg +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/__init__.py +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/agentic/dashboard/agentic_workflow_dashboard.py +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/agentic/dashboard/generate_static_dashboard.py +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/agentic/dashboard/path_analysis.py +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/agentic/pipeline/argument_definitions.py +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/agentic/pipeline/build_json_results.py +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/agentic/pipeline/create_ui_input.py +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/agentic/pipeline/full_traces_evaluation/clear_analysis/base_clear_runner.py +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/agentic/pipeline/full_traces_evaluation/clear_analysis/issues_clear_runner.py +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/agentic/pipeline/full_traces_evaluation/clear_analysis/root_cause_clear_runner.py +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/agentic/pipeline/full_traces_evaluation/clear_analysis/run_clear_analysis.py +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/agentic/pipeline/full_traces_evaluation/trace_evaluation/base_evaluator.py +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/agentic/pipeline/full_traces_evaluation/trace_evaluation/full_trajectory_evaluator.py +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/agentic/pipeline/full_traces_evaluation/trace_evaluation/rubric_evaluator.py +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/agentic/pipeline/full_traces_evaluation/trace_evaluation/rubric_generator.py +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/agentic/pipeline/full_traces_evaluation/trace_evaluation/task_success_evaluator.py +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/agentic/pipeline/preprocess_traces/compact_trace_formatter.py +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/agentic/pipeline/preprocess_traces/preprocess_traces.py +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/agentic/pipeline/preprocess_traces/process_langfuse_traces.py +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/agentic/pipeline/preprocess_traces/process_mlflow_traces.py +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/agentic/pipeline/preprocess_traces/trace_utils.py +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/agentic/pipeline/setup/__init__.py +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/agentic/pipeline/setup/config_example.json +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/agentic/pipeline/setup/config_example.yaml +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/agentic/pipeline/setup/rh_config_example.yaml +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/agentic/pipeline/setup/test_tools_config_example.yaml +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/agentic/pipeline/utils.py +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/analysis_runner.py +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/args.py +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/cli.py +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/dashboard/__init__.py +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/dashboard/show_analysis_dashboard.py +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/load_ui.py +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/logging_config.py +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/pipeline/__init__.py +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/pipeline/caching_utils.py +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/pipeline/config_loader.py +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/pipeline/constants.py +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/pipeline/eval_utils.py +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/pipeline/evaluation_criteria.py +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/pipeline/external_judge.py +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/pipeline/full_pipeline.py +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/pipeline/inference_utils/__init__.py +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/pipeline/inference_utils/endpoint_backends.py +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/pipeline/inference_utils/langchain_chat_models.py +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/pipeline/inference_utils/llm_client.py +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/pipeline/propmts.py +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/pipeline/setup/__init__.py +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/pipeline/setup/default_config.yaml +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/pipeline/threading_utils.py +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/pipeline/use_cases/__init__.py +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/pipeline/use_cases/eval_use_case.py +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/pipeline/use_cases/external_judge_use_case.py +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/pipeline/use_cases/tool_call_use_case.py +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/pipeline/use_cases/use_case_utils.py +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/sample_data/__init__.py +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/sample_data/agentic/research_agent_results/mlflow/my_experiment/pipeline_summary.json +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/sample_data/agentic/research_agent_results/mlflow/my_experiment/traces_data/tr-006eee07eed4484592661b1bdba32a75.csv +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/sample_data/agentic/research_agent_results/mlflow/my_experiment/traces_data/tr-0e0148ed21eeebc046f03f63e4418723.csv +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/sample_data/agentic/research_agent_results/mlflow/my_experiment/traces_data/tr-0e1ef041647642c958a8aaa1892fdb88.csv +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/sample_data/agentic/research_agent_results/mlflow/my_experiment/traces_data/tr-4ae9a4305b7a0319ec516ef4eef4ae17.csv +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/sample_data/agentic/research_agent_results/mlflow/my_experiment/traces_data/tr-4eead8c00e900cc84f217a9a10724ecf.csv +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/sample_data/agentic/research_agent_results/mlflow/my_experiment/traces_data/tr-50adf266e9c1522dae57a6290a61596f.csv +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/sample_data/agentic/research_agent_results/mlflow/my_experiment/traces_data/tr-554015d661e4f69cfc629b6d46898421.csv +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/sample_data/agentic/research_agent_results/mlflow/my_experiment/traces_data/tr-55a6003784e34390f47c5c929d8f4db0.csv +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/sample_data/agentic/research_agent_results/mlflow/my_experiment/traces_data/tr-5818239708258c81fff3a4bb0b144802.csv +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/sample_data/agentic/research_agent_results/mlflow/my_experiment/traces_data/tr-5e63ae3886b225ae89a07b7fa4aafe0d.csv +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/sample_data/agentic/research_agent_results/mlflow/my_experiment/traces_data/tr-640f5750b635929a88547548f0cbb4da.csv +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/sample_data/agentic/research_agent_results/mlflow/my_experiment/traces_data/tr-6b2d02f30d4f0d3f4ca7d8f1469913ae.csv +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/sample_data/agentic/research_agent_results/mlflow/my_experiment/traces_data/tr-6c852575331f2e0a02a331d1ddaae59f.csv +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/sample_data/agentic/research_agent_results/mlflow/my_experiment/traces_data/tr-8e68becd37326e6fec6a7d3d3b47dbbb.csv +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/sample_data/agentic/research_agent_results/mlflow/my_experiment/traces_data/tr-b35ed52266fd023dd808950c6848f062.csv +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/sample_data/agentic/research_agent_results/mlflow/my_experiment/traces_data/tr-bccc245d6d47cdf730fd5db3028b52a3.csv +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/sample_data/agentic/research_agent_results/mlflow/my_experiment/traces_data/tr-d6010cdc5e36a50046fc10f33f0bc427.csv +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/sample_data/agentic/research_agent_results/mlflow/my_experiment/traces_data/tr-edadad8b3a67a24a43d145ae07c0c12a.csv +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/sample_data/agentic/research_agent_results/mlflow/my_experiment/traces_data/tr-f082801e1021022e0756e7d089f1558c.csv +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/sample_data/agentic/research_agent_results/mlflow/my_experiment/traces_data/tr-f34a9851b31fafffe527c5b2d6262de8.csv +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/sample_data/agentic/research_agent_results/mlflow/my_experiment/unified_ui_results.zip +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/sample_data/agentic/research_agent_traces/langfuse/434f3f96ac470e805428c4be45b4f2a3.json +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/sample_data/agentic/research_agent_traces/langfuse/4c354cfb5b17f61af073df6796f2b236.json +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/sample_data/agentic/research_agent_traces/langfuse/4deb34a4521fcedf92083d0a93fe11d4.json +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/sample_data/agentic/research_agent_traces/langfuse/c685b7eb7826a87149bd2c42a50e47c5.json +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/sample_data/agentic/research_agent_traces/langfuse/ebf1d6897941adb63bcc62aa7c50e8f8.json +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/sample_data/agentic/research_agent_traces/mlflow/tr-006eee07eed4484592661b1bdba32a75.json +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/sample_data/agentic/research_agent_traces/mlflow/tr-0e0148ed21eeebc046f03f63e4418723.json +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/sample_data/agentic/research_agent_traces/mlflow/tr-0e1ef041647642c958a8aaa1892fdb88.json +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/sample_data/agentic/research_agent_traces/mlflow/tr-4ae9a4305b7a0319ec516ef4eef4ae17.json +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/sample_data/agentic/research_agent_traces/mlflow/tr-4eead8c00e900cc84f217a9a10724ecf.json +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/sample_data/agentic/research_agent_traces/mlflow/tr-50adf266e9c1522dae57a6290a61596f.json +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/sample_data/agentic/research_agent_traces/mlflow/tr-554015d661e4f69cfc629b6d46898421.json +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/sample_data/agentic/research_agent_traces/mlflow/tr-55a6003784e34390f47c5c929d8f4db0.json +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/sample_data/agentic/research_agent_traces/mlflow/tr-5818239708258c81fff3a4bb0b144802.json +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/sample_data/agentic/research_agent_traces/mlflow/tr-5e63ae3886b225ae89a07b7fa4aafe0d.json +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/sample_data/agentic/research_agent_traces/mlflow/tr-640f5750b635929a88547548f0cbb4da.json +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/sample_data/agentic/research_agent_traces/mlflow/tr-6b2d02f30d4f0d3f4ca7d8f1469913ae.json +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/sample_data/agentic/research_agent_traces/mlflow/tr-6c852575331f2e0a02a331d1ddaae59f.json +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/sample_data/agentic/research_agent_traces/mlflow/tr-8e68becd37326e6fec6a7d3d3b47dbbb.json +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/sample_data/agentic/research_agent_traces/mlflow/tr-b35ed52266fd023dd808950c6848f062.json +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/sample_data/agentic/research_agent_traces/mlflow/tr-bccc245d6d47cdf730fd5db3028b52a3.json +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/sample_data/agentic/research_agent_traces/mlflow/tr-d6010cdc5e36a50046fc10f33f0bc427.json +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/sample_data/agentic/research_agent_traces/mlflow/tr-edadad8b3a67a24a43d145ae07c0c12a.json +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/sample_data/agentic/research_agent_traces/mlflow/tr-f082801e1021022e0756e7d089f1558c.json +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/sample_data/agentic/research_agent_traces/mlflow/tr-f34a9851b31fafffe527c5b2d6262de8.json +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/sample_data/gsm8k/__init__.py +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/sample_data/gsm8k/analysis_results_gsm8k_default.zip +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/sample_data/gsm8k/gsm8k_default_predictions.csv +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/sample_data/tool_calls/tool_calls_sample_data.csv +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval.egg-info/SOURCES.txt +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval.egg-info/dependency_links.txt +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval.egg-info/entry_points.txt +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval.egg-info/requires.txt +0 -0
- {clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval.egg-info/top_level.txt +0 -0
|
@@ -0,0 +1,248 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: clear_eval
|
|
3
|
+
Version: 2.0.2
|
|
4
|
+
Summary: A python API sdk facilitating Error Analysis via LLM-as-a-Judge
|
|
5
|
+
Author-email: Lilach Eden <lilache@il.ibm.com>, Asaf Yehudai <asaf.yehudai@ibm.com>
|
|
6
|
+
License: Apache-2.0
|
|
7
|
+
Classifier: Programming Language :: Python :: 3
|
|
8
|
+
Classifier: Operating System :: OS Independent
|
|
9
|
+
Requires-Python: >=3.10
|
|
10
|
+
Description-Content-Type: text/markdown
|
|
11
|
+
License-File: LICENSE
|
|
12
|
+
Requires-Dist: python-dotenv
|
|
13
|
+
Requires-Dist: langchain
|
|
14
|
+
Requires-Dist: langgraph
|
|
15
|
+
Requires-Dist: langchain_openai
|
|
16
|
+
Requires-Dist: langchain_ibm
|
|
17
|
+
Requires-Dist: ibm_watsonx_ai>=1.3.42
|
|
18
|
+
Requires-Dist: openai
|
|
19
|
+
Requires-Dist: pyyaml
|
|
20
|
+
Requires-Dist: pandas
|
|
21
|
+
Requires-Dist: tqdm
|
|
22
|
+
Requires-Dist: pyyaml
|
|
23
|
+
Requires-Dist: streamlit
|
|
24
|
+
Requires-Dist: matplotlib
|
|
25
|
+
Requires-Dist: seaborn
|
|
26
|
+
Requires-Dist: langchain_community
|
|
27
|
+
Requires-Dist: numpy
|
|
28
|
+
Requires-Dist: pyarrow>=22.0.0
|
|
29
|
+
Requires-Dist: nest-asyncio
|
|
30
|
+
Requires-Dist: nicegui>=3.7.1
|
|
31
|
+
Requires-Dist: plotly
|
|
32
|
+
Requires-Dist: networkx
|
|
33
|
+
Requires-Dist: litellm>=1.83.0
|
|
34
|
+
Requires-Dist: nest-asyncio
|
|
35
|
+
Requires-Dist: scipy
|
|
36
|
+
Requires-Dist: scikit-learn
|
|
37
|
+
Requires-Dist: botocore
|
|
38
|
+
Provides-Extra: tool-calls
|
|
39
|
+
Requires-Dist: agent-lifecycle-toolkit; extra == "tool-calls"
|
|
40
|
+
Dynamic: license-file
|
|
41
|
+
|
|
42
|
+
# CLEAR: Comprehensive LLM Error Analysis and Reporting
|
|
43
|
+
|
|
44
|
+
**CLEAR** is an open-source toolkit for **LLM error analysis** using an LLM-as-a-Judge approach.
|
|
45
|
+
|
|
46
|
+
[](https://www.python.org/downloads/)
|
|
47
|
+
[](LICENSE)
|
|
48
|
+
[](https://pypi.org/project/clear-eval/)
|
|
49
|
+
|
|
50
|
+
---
|
|
51
|
+
|
|
52
|
+
## What is CLEAR?
|
|
53
|
+
|
|
54
|
+
CLEAR provides systematic error analysis for LLM-based systems. It combines automated LLM-as-a-judge evaluation with interactive dashboards to help you:
|
|
55
|
+
- Identify recurring error patterns across your dataset
|
|
56
|
+
- Quantify issue frequencies and severity
|
|
57
|
+
- Drill down into specific failure cases
|
|
58
|
+
- Prioritize improvements based on data-driven insights
|
|
59
|
+
|
|
60
|
+
CLEAR operates in two phases:
|
|
61
|
+
|
|
62
|
+
1. **Analysis** — Generates per-instance textual feedback, identifies system-level error categories, and quantifies their frequencies.
|
|
63
|
+
2. **Interactive Dashboard** — Explore aggregate visualizations, apply dynamic filters, and drill down into individual failure examples.
|
|
64
|
+
|
|
65
|
+
---
|
|
66
|
+
|
|
67
|
+
## Two Analysis Modes
|
|
68
|
+
|
|
69
|
+
### LLM Analysis
|
|
70
|
+
|
|
71
|
+
Evaluate standard LLM outputs — generation quality, correctness, and recurring error patterns. Provide a CSV with prompts and responses, and CLEAR will score each instance, generate textual critiques, and surface system-level issues.
|
|
72
|
+
|
|
73
|
+
- **Input:** CSV with model inputs and responses
|
|
74
|
+
- **Output:** Per-record scores, evaluation text, aggregated issue categories
|
|
75
|
+
- **Dashboard:** Streamlit-based interactive explorer
|
|
76
|
+
|
|
77
|
+
> **[LLM Analysis Guide →](docs/llm-analysis.md)**
|
|
78
|
+
|
|
79
|
+
### Agentic Analysis
|
|
80
|
+
|
|
81
|
+
Evaluate multi-agent system trajectories — step-by-step agent interactions and full trajectory analysis.
|
|
82
|
+
Supports traces from LangGraph, CrewAI, and other frameworks via MLflow or Langfuse.
|
|
83
|
+
|
|
84
|
+
- **Input:** Raw JSON traces or preprocessed trajectory CSVs (each trace captures one complete agent task execution)
|
|
85
|
+
- **Output:** Per-step CLEAR analysis, trajectory-level scores, rubric evaluations
|
|
86
|
+
- **Dashboard:** NiceGUI-based workflow visualization with path and temporal analysis
|
|
87
|
+
|
|
88
|
+
> **[Agentic Workflows Guide →](src/clear_eval/agentic/README.md)**
|
|
89
|
+
|
|
90
|
+
---
|
|
91
|
+
|
|
92
|
+
## Installation
|
|
93
|
+
|
|
94
|
+
**Requires Python 3.10+**
|
|
95
|
+
|
|
96
|
+
#### Option 1: pip
|
|
97
|
+
|
|
98
|
+
```bash
|
|
99
|
+
pip install clear-eval
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
#### Option 2: From source (for development)
|
|
103
|
+
|
|
104
|
+
```bash
|
|
105
|
+
git clone https://github.com/IBM/CLEAR.git
|
|
106
|
+
cd CLEAR
|
|
107
|
+
python3 -m venv .venv
|
|
108
|
+
source .venv/bin/activate # On Windows: .venv\Scripts\activate
|
|
109
|
+
pip install -e .
|
|
110
|
+
```
|
|
111
|
+
|
|
112
|
+
Verify the installation:
|
|
113
|
+
|
|
114
|
+
```bash
|
|
115
|
+
run-clear-eval-analysis --help
|
|
116
|
+
```
|
|
117
|
+
|
|
118
|
+
---
|
|
119
|
+
|
|
120
|
+
## Quick Start
|
|
121
|
+
|
|
122
|
+
### 1. Set your provider credentials
|
|
123
|
+
|
|
124
|
+
CLEAR requires a supported LLM provider. Set the appropriate environment variables for your provider (e.g., `OPENAI_API_KEY` for OpenAI). Adjust `--provider` and `--eval-model-name` in the commands below to match your setup. See [Provider Configuration](#provider-configuration) for details.
|
|
125
|
+
|
|
126
|
+
### 2. LLM Analysis
|
|
127
|
+
|
|
128
|
+
This evaluates GSM8K math problem responses and surfaces recurring quality issues:
|
|
129
|
+
|
|
130
|
+
```bash
|
|
131
|
+
run-clear-eval-analysis --provider openai --eval-model-name gpt-4o
|
|
132
|
+
```
|
|
133
|
+
|
|
134
|
+
Results are saved to `results/gsm8k/sample_output/`. View them:
|
|
135
|
+
|
|
136
|
+
```bash
|
|
137
|
+
run-clear-eval-dashboard
|
|
138
|
+
```
|
|
139
|
+
|
|
140
|
+
> **[Full LLM Analysis Guide →](docs/llm-analysis.md)** — input formats, CLI arguments, configuration, Python API, and external judges.
|
|
141
|
+
|
|
142
|
+
### 3. Agentic Analysis
|
|
143
|
+
|
|
144
|
+
These two modes are independent — this section does not require step 2.
|
|
145
|
+
|
|
146
|
+
Run CLEAR on sample agent traces (3 traces, each capturing one complete agent task execution, ~2 minutes):
|
|
147
|
+
|
|
148
|
+
```bash
|
|
149
|
+
run-clear-agentic-eval \
|
|
150
|
+
--data-dir src/clear_eval/sample_data/agentic/research_agent_traces/mlflow \
|
|
151
|
+
--results-dir my_smoke_test_results \
|
|
152
|
+
--from-raw-traces true \
|
|
153
|
+
--agent-framework langgraph \
|
|
154
|
+
--observability-framework mlflow \
|
|
155
|
+
--run-name smoke_test \
|
|
156
|
+
--max-files 3 \
|
|
157
|
+
--eval-model-name gpt-4o \
|
|
158
|
+
--provider openai
|
|
159
|
+
```
|
|
160
|
+
|
|
161
|
+
View pre-computed results (all 20 traces) without re-running:
|
|
162
|
+
|
|
163
|
+
```bash
|
|
164
|
+
run-clear-agentic-dashboard
|
|
165
|
+
# Upload: src/clear_eval/sample_data/agentic/research_agent_results/mlflow/my_experiment/unified_ui_results.zip
|
|
166
|
+
```
|
|
167
|
+
|
|
168
|
+
> **[Full Agentic Guide →](src/clear_eval/agentic/README.md)** — trace generation, configuration, output structure, and dashboard features.
|
|
169
|
+
|
|
170
|
+
---
|
|
171
|
+
|
|
172
|
+
## Provider Configuration
|
|
173
|
+
|
|
174
|
+
CLEAR uses [LiteLLM](https://docs.litellm.ai/docs/providers) as its inference backend, supporting 100+ LLM providers (OpenAI, Anthropic, WatsonX, AWS Bedrock, Google Vertex AI, and more).
|
|
175
|
+
|
|
176
|
+
**Parameters:**
|
|
177
|
+
|
|
178
|
+
| Parameter | CLI Flag | Description |
|
|
179
|
+
|-----------|----------|-------------|
|
|
180
|
+
| `provider` | `--provider` | LiteLLM provider name (e.g., `openai`, `anthropic`, `bedrock`, `vertex_ai`) |
|
|
181
|
+
| `eval_model_name` | `--eval-model-name` | Model identifier (e.g., `gpt-4o`, `claude-3-5-sonnet-20241022`) |
|
|
182
|
+
| `eval_model_params` | `--eval-model-params` | Additional model parameters as JSON (e.g., `{"temperature": 0}`) |
|
|
183
|
+
| `endpoint_url` | `--endpoint-url` | Custom endpoint URL for local/self-hosted models |
|
|
184
|
+
|
|
185
|
+
**Cloud provider example:**
|
|
186
|
+
|
|
187
|
+
```bash
|
|
188
|
+
export OPENAI_API_KEY="..."
|
|
189
|
+
run-clear-eval-analysis --provider openai --eval-model-name gpt-4o
|
|
190
|
+
```
|
|
191
|
+
|
|
192
|
+
**Local model example (vLLM, llama.cpp, Ollama, etc.):**
|
|
193
|
+
|
|
194
|
+
```bash
|
|
195
|
+
run-clear-eval-analysis \
|
|
196
|
+
--provider openai \
|
|
197
|
+
--eval-model-name my-local-model \
|
|
198
|
+
--endpoint-url http://localhost:8000/v1
|
|
199
|
+
```
|
|
200
|
+
|
|
201
|
+
No credentials are needed when using `--endpoint-url` with a local server.
|
|
202
|
+
|
|
203
|
+
Set the required environment variables for your provider according to [LiteLLM's documentation](https://docs.litellm.ai/docs/providers).
|
|
204
|
+
|
|
205
|
+
---
|
|
206
|
+
|
|
207
|
+
## Documentation
|
|
208
|
+
|
|
209
|
+
| Guide | Description |
|
|
210
|
+
|-------|-------------|
|
|
211
|
+
| [**Agentic Workflows Guide**](src/clear_eval/agentic/README.md) | Multi-agent evaluation — trace preprocessing, step-by-step and trajectory analysis, configuration reference |
|
|
212
|
+
| [**Agentic Dashboard Guide**](docs/agentic/dashboard.md) | Dashboard features — workflow view, node analysis, trajectory explorer, path and temporal analysis |
|
|
213
|
+
| [**LLM Analysis Guide**](docs/llm-analysis.md) | Full pipeline reference — input formats, CLI arguments, configuration, and external judges |
|
|
214
|
+
|
|
215
|
+
---
|
|
216
|
+
|
|
217
|
+
## Citation
|
|
218
|
+
|
|
219
|
+
If you use CLEAR in your research, please cite the relevant paper(s):
|
|
220
|
+
|
|
221
|
+
**LLM Analysis** ([AAAI 2026](https://ojs.aaai.org/index.php/AAAI/article/view/42398)):
|
|
222
|
+
```bibtex
|
|
223
|
+
@inproceedings{yehudai2026clear,
|
|
224
|
+
title={CLEAR: Error analysis via llm-as-a-judge made easy},
|
|
225
|
+
author={Yehudai, Asaf and Eden, Lilach and Perlitz, Yotam and Bar-Haim, Roy and Shmueli-Scheuer, Michal},
|
|
226
|
+
booktitle={Proceedings of the AAAI Conference on Artificial Intelligence},
|
|
227
|
+
volume={40},
|
|
228
|
+
number={48},
|
|
229
|
+
pages={41736--41738},
|
|
230
|
+
year={2026}
|
|
231
|
+
}
|
|
232
|
+
```
|
|
233
|
+
|
|
234
|
+
**Agentic Analysis** (ACL 2026, to appear — [preprint](https://arxiv.org/abs/2605.22608)):
|
|
235
|
+
```bibtex
|
|
236
|
+
@article{yehudai2026agentic,
|
|
237
|
+
title={Agentic CLEAR: Automating Multi-Level Evaluation of LLM Agents},
|
|
238
|
+
author={Yehudai, Asaf and Eden, Lilach and Shmueli-Scheuer, Michal},
|
|
239
|
+
journal={arXiv preprint arXiv:2605.22608},
|
|
240
|
+
year={2026}
|
|
241
|
+
}
|
|
242
|
+
```
|
|
243
|
+
|
|
244
|
+
---
|
|
245
|
+
|
|
246
|
+
## License
|
|
247
|
+
|
|
248
|
+
Apache 2.0 — see [LICENSE](LICENSE) for details.
|
|
@@ -0,0 +1,207 @@
|
|
|
1
|
+
# CLEAR: Comprehensive LLM Error Analysis and Reporting
|
|
2
|
+
|
|
3
|
+
**CLEAR** is an open-source toolkit for **LLM error analysis** using an LLM-as-a-Judge approach.
|
|
4
|
+
|
|
5
|
+
[](https://www.python.org/downloads/)
|
|
6
|
+
[](LICENSE)
|
|
7
|
+
[](https://pypi.org/project/clear-eval/)
|
|
8
|
+
|
|
9
|
+
---
|
|
10
|
+
|
|
11
|
+
## What is CLEAR?
|
|
12
|
+
|
|
13
|
+
CLEAR provides systematic error analysis for LLM-based systems. It combines automated LLM-as-a-judge evaluation with interactive dashboards to help you:
|
|
14
|
+
- Identify recurring error patterns across your dataset
|
|
15
|
+
- Quantify issue frequencies and severity
|
|
16
|
+
- Drill down into specific failure cases
|
|
17
|
+
- Prioritize improvements based on data-driven insights
|
|
18
|
+
|
|
19
|
+
CLEAR operates in two phases:
|
|
20
|
+
|
|
21
|
+
1. **Analysis** — Generates per-instance textual feedback, identifies system-level error categories, and quantifies their frequencies.
|
|
22
|
+
2. **Interactive Dashboard** — Explore aggregate visualizations, apply dynamic filters, and drill down into individual failure examples.
|
|
23
|
+
|
|
24
|
+
---
|
|
25
|
+
|
|
26
|
+
## Two Analysis Modes
|
|
27
|
+
|
|
28
|
+
### LLM Analysis
|
|
29
|
+
|
|
30
|
+
Evaluate standard LLM outputs — generation quality, correctness, and recurring error patterns. Provide a CSV with prompts and responses, and CLEAR will score each instance, generate textual critiques, and surface system-level issues.
|
|
31
|
+
|
|
32
|
+
- **Input:** CSV with model inputs and responses
|
|
33
|
+
- **Output:** Per-record scores, evaluation text, aggregated issue categories
|
|
34
|
+
- **Dashboard:** Streamlit-based interactive explorer
|
|
35
|
+
|
|
36
|
+
> **[LLM Analysis Guide →](docs/llm-analysis.md)**
|
|
37
|
+
|
|
38
|
+
### Agentic Analysis
|
|
39
|
+
|
|
40
|
+
Evaluate multi-agent system trajectories — step-by-step agent interactions and full trajectory analysis.
|
|
41
|
+
Supports traces from LangGraph, CrewAI, and other frameworks via MLflow or Langfuse.
|
|
42
|
+
|
|
43
|
+
- **Input:** Raw JSON traces or preprocessed trajectory CSVs (each trace captures one complete agent task execution)
|
|
44
|
+
- **Output:** Per-step CLEAR analysis, trajectory-level scores, rubric evaluations
|
|
45
|
+
- **Dashboard:** NiceGUI-based workflow visualization with path and temporal analysis
|
|
46
|
+
|
|
47
|
+
> **[Agentic Workflows Guide →](src/clear_eval/agentic/README.md)**
|
|
48
|
+
|
|
49
|
+
---
|
|
50
|
+
|
|
51
|
+
## Installation
|
|
52
|
+
|
|
53
|
+
**Requires Python 3.10+**
|
|
54
|
+
|
|
55
|
+
#### Option 1: pip
|
|
56
|
+
|
|
57
|
+
```bash
|
|
58
|
+
pip install clear-eval
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
#### Option 2: From source (for development)
|
|
62
|
+
|
|
63
|
+
```bash
|
|
64
|
+
git clone https://github.com/IBM/CLEAR.git
|
|
65
|
+
cd CLEAR
|
|
66
|
+
python3 -m venv .venv
|
|
67
|
+
source .venv/bin/activate # On Windows: .venv\Scripts\activate
|
|
68
|
+
pip install -e .
|
|
69
|
+
```
|
|
70
|
+
|
|
71
|
+
Verify the installation:
|
|
72
|
+
|
|
73
|
+
```bash
|
|
74
|
+
run-clear-eval-analysis --help
|
|
75
|
+
```
|
|
76
|
+
|
|
77
|
+
---
|
|
78
|
+
|
|
79
|
+
## Quick Start
|
|
80
|
+
|
|
81
|
+
### 1. Set your provider credentials
|
|
82
|
+
|
|
83
|
+
CLEAR requires a supported LLM provider. Set the appropriate environment variables for your provider (e.g., `OPENAI_API_KEY` for OpenAI). Adjust `--provider` and `--eval-model-name` in the commands below to match your setup. See [Provider Configuration](#provider-configuration) for details.
|
|
84
|
+
|
|
85
|
+
### 2. LLM Analysis
|
|
86
|
+
|
|
87
|
+
This evaluates GSM8K math problem responses and surfaces recurring quality issues:
|
|
88
|
+
|
|
89
|
+
```bash
|
|
90
|
+
run-clear-eval-analysis --provider openai --eval-model-name gpt-4o
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
Results are saved to `results/gsm8k/sample_output/`. View them:
|
|
94
|
+
|
|
95
|
+
```bash
|
|
96
|
+
run-clear-eval-dashboard
|
|
97
|
+
```
|
|
98
|
+
|
|
99
|
+
> **[Full LLM Analysis Guide →](docs/llm-analysis.md)** — input formats, CLI arguments, configuration, Python API, and external judges.
|
|
100
|
+
|
|
101
|
+
### 3. Agentic Analysis
|
|
102
|
+
|
|
103
|
+
These two modes are independent — this section does not require step 2.
|
|
104
|
+
|
|
105
|
+
Run CLEAR on sample agent traces (3 traces, each capturing one complete agent task execution, ~2 minutes):
|
|
106
|
+
|
|
107
|
+
```bash
|
|
108
|
+
run-clear-agentic-eval \
|
|
109
|
+
--data-dir src/clear_eval/sample_data/agentic/research_agent_traces/mlflow \
|
|
110
|
+
--results-dir my_smoke_test_results \
|
|
111
|
+
--from-raw-traces true \
|
|
112
|
+
--agent-framework langgraph \
|
|
113
|
+
--observability-framework mlflow \
|
|
114
|
+
--run-name smoke_test \
|
|
115
|
+
--max-files 3 \
|
|
116
|
+
--eval-model-name gpt-4o \
|
|
117
|
+
--provider openai
|
|
118
|
+
```
|
|
119
|
+
|
|
120
|
+
View pre-computed results (all 20 traces) without re-running:
|
|
121
|
+
|
|
122
|
+
```bash
|
|
123
|
+
run-clear-agentic-dashboard
|
|
124
|
+
# Upload: src/clear_eval/sample_data/agentic/research_agent_results/mlflow/my_experiment/unified_ui_results.zip
|
|
125
|
+
```
|
|
126
|
+
|
|
127
|
+
> **[Full Agentic Guide →](src/clear_eval/agentic/README.md)** — trace generation, configuration, output structure, and dashboard features.
|
|
128
|
+
|
|
129
|
+
---
|
|
130
|
+
|
|
131
|
+
## Provider Configuration
|
|
132
|
+
|
|
133
|
+
CLEAR uses [LiteLLM](https://docs.litellm.ai/docs/providers) as its inference backend, supporting 100+ LLM providers (OpenAI, Anthropic, WatsonX, AWS Bedrock, Google Vertex AI, and more).
|
|
134
|
+
|
|
135
|
+
**Parameters:**
|
|
136
|
+
|
|
137
|
+
| Parameter | CLI Flag | Description |
|
|
138
|
+
|-----------|----------|-------------|
|
|
139
|
+
| `provider` | `--provider` | LiteLLM provider name (e.g., `openai`, `anthropic`, `bedrock`, `vertex_ai`) |
|
|
140
|
+
| `eval_model_name` | `--eval-model-name` | Model identifier (e.g., `gpt-4o`, `claude-3-5-sonnet-20241022`) |
|
|
141
|
+
| `eval_model_params` | `--eval-model-params` | Additional model parameters as JSON (e.g., `{"temperature": 0}`) |
|
|
142
|
+
| `endpoint_url` | `--endpoint-url` | Custom endpoint URL for local/self-hosted models |
|
|
143
|
+
|
|
144
|
+
**Cloud provider example:**
|
|
145
|
+
|
|
146
|
+
```bash
|
|
147
|
+
export OPENAI_API_KEY="..."
|
|
148
|
+
run-clear-eval-analysis --provider openai --eval-model-name gpt-4o
|
|
149
|
+
```
|
|
150
|
+
|
|
151
|
+
**Local model example (vLLM, llama.cpp, Ollama, etc.):**
|
|
152
|
+
|
|
153
|
+
```bash
|
|
154
|
+
run-clear-eval-analysis \
|
|
155
|
+
--provider openai \
|
|
156
|
+
--eval-model-name my-local-model \
|
|
157
|
+
--endpoint-url http://localhost:8000/v1
|
|
158
|
+
```
|
|
159
|
+
|
|
160
|
+
No credentials are needed when using `--endpoint-url` with a local server.
|
|
161
|
+
|
|
162
|
+
Set the required environment variables for your provider according to [LiteLLM's documentation](https://docs.litellm.ai/docs/providers).
|
|
163
|
+
|
|
164
|
+
---
|
|
165
|
+
|
|
166
|
+
## Documentation
|
|
167
|
+
|
|
168
|
+
| Guide | Description |
|
|
169
|
+
|-------|-------------|
|
|
170
|
+
| [**Agentic Workflows Guide**](src/clear_eval/agentic/README.md) | Multi-agent evaluation — trace preprocessing, step-by-step and trajectory analysis, configuration reference |
|
|
171
|
+
| [**Agentic Dashboard Guide**](docs/agentic/dashboard.md) | Dashboard features — workflow view, node analysis, trajectory explorer, path and temporal analysis |
|
|
172
|
+
| [**LLM Analysis Guide**](docs/llm-analysis.md) | Full pipeline reference — input formats, CLI arguments, configuration, and external judges |
|
|
173
|
+
|
|
174
|
+
---
|
|
175
|
+
|
|
176
|
+
## Citation
|
|
177
|
+
|
|
178
|
+
If you use CLEAR in your research, please cite the relevant paper(s):
|
|
179
|
+
|
|
180
|
+
**LLM Analysis** ([AAAI 2026](https://ojs.aaai.org/index.php/AAAI/article/view/42398)):
|
|
181
|
+
```bibtex
|
|
182
|
+
@inproceedings{yehudai2026clear,
|
|
183
|
+
title={CLEAR: Error analysis via llm-as-a-judge made easy},
|
|
184
|
+
author={Yehudai, Asaf and Eden, Lilach and Perlitz, Yotam and Bar-Haim, Roy and Shmueli-Scheuer, Michal},
|
|
185
|
+
booktitle={Proceedings of the AAAI Conference on Artificial Intelligence},
|
|
186
|
+
volume={40},
|
|
187
|
+
number={48},
|
|
188
|
+
pages={41736--41738},
|
|
189
|
+
year={2026}
|
|
190
|
+
}
|
|
191
|
+
```
|
|
192
|
+
|
|
193
|
+
**Agentic Analysis** (ACL 2026, to appear — [preprint](https://arxiv.org/abs/2605.22608)):
|
|
194
|
+
```bibtex
|
|
195
|
+
@article{yehudai2026agentic,
|
|
196
|
+
title={Agentic CLEAR: Automating Multi-Level Evaluation of LLM Agents},
|
|
197
|
+
author={Yehudai, Asaf and Eden, Lilach and Shmueli-Scheuer, Michal},
|
|
198
|
+
journal={arXiv preprint arXiv:2605.22608},
|
|
199
|
+
year={2026}
|
|
200
|
+
}
|
|
201
|
+
```
|
|
202
|
+
|
|
203
|
+
---
|
|
204
|
+
|
|
205
|
+
## License
|
|
206
|
+
|
|
207
|
+
Apache 2.0 — see [LICENSE](LICENSE) for details.
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "clear_eval"
|
|
7
|
-
version = "2.0.
|
|
7
|
+
version = "2.0.2"
|
|
8
8
|
description = "A python API sdk facilitating Error Analysis via LLM-as-a-Judge"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
authors = [{ name = "Lilach Eden", email = "lilache@il.ibm.com"},
|
|
@@ -9,6 +9,18 @@ import sys
|
|
|
9
9
|
from pathlib import Path
|
|
10
10
|
|
|
11
11
|
|
|
12
|
+
def _is_port_available(host: str, port: int) -> bool:
|
|
13
|
+
"""Check if a port is available for binding."""
|
|
14
|
+
import socket
|
|
15
|
+
try:
|
|
16
|
+
with socket.socket(socket.AF_INET, socket.SOCK_STREAM) as s:
|
|
17
|
+
s.setsockopt(socket.SOL_SOCKET, socket.SO_REUSEADDR, 1)
|
|
18
|
+
s.bind((host, port))
|
|
19
|
+
return True
|
|
20
|
+
except OSError:
|
|
21
|
+
return False
|
|
22
|
+
|
|
23
|
+
|
|
12
24
|
def main():
|
|
13
25
|
parser = argparse.ArgumentParser(description="Launch the Agentic Workflow Dashboard")
|
|
14
26
|
parser.add_argument("--port", type=int, default=8080, help="Port to run the dashboard on (default: 8080)")
|
|
@@ -16,6 +28,12 @@ def main():
|
|
|
16
28
|
parser.add_argument("--no-open", action="store_true", help="Don't auto-open browser")
|
|
17
29
|
args = parser.parse_args()
|
|
18
30
|
|
|
31
|
+
if not _is_port_available(args.host, args.port):
|
|
32
|
+
print(f"Error: Port {args.port} is already in use.")
|
|
33
|
+
print(f"Either stop the process using port {args.port}, or use a different port:")
|
|
34
|
+
print(f" run-clear-agentic-dashboard --port {args.port + 1}")
|
|
35
|
+
sys.exit(1)
|
|
36
|
+
|
|
19
37
|
# Ensure the server shuts down cleanly on termination signals,
|
|
20
38
|
# so the port is released immediately instead of staying in TIME_WAIT.
|
|
21
39
|
def _handle_signal(signum, frame):
|
|
@@ -690,7 +690,6 @@ def main():
|
|
|
690
690
|
validate_required_config(config, ['data_dir', 'results_dir'], parser)
|
|
691
691
|
|
|
692
692
|
# Get run output directory
|
|
693
|
-
user_provided_run_name = config.get('run_name') is not None
|
|
694
693
|
output_dir, run_name = get_run_output_dir(
|
|
695
694
|
config['results_dir'],
|
|
696
695
|
config.get('run_name')
|
|
@@ -708,14 +707,10 @@ def main():
|
|
|
708
707
|
logger.error(f"Input directory does not exist: {traj_input_dir}")
|
|
709
708
|
sys.exit(1)
|
|
710
709
|
|
|
711
|
-
#
|
|
712
|
-
if output_dir.exists()
|
|
713
|
-
if config.get('overwrite',
|
|
714
|
-
logger.
|
|
715
|
-
f"Output directory already exists: {output_dir}\n"
|
|
716
|
-
f"Use --overwrite false to resume, or delete the existing directory."
|
|
717
|
-
)
|
|
718
|
-
sys.exit(1)
|
|
710
|
+
# Log if output dir already exists
|
|
711
|
+
if output_dir.exists():
|
|
712
|
+
if config.get('overwrite', False):
|
|
713
|
+
logger.info(f"Re-running existing run with overwrite: {run_name}")
|
|
719
714
|
else:
|
|
720
715
|
logger.info(f"Resuming existing run: {run_name}")
|
|
721
716
|
|
{clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/agentic/pipeline/run_clear_agentic_eval.py
RENAMED
|
@@ -148,17 +148,20 @@ def prepare_traces_data(
|
|
|
148
148
|
else:
|
|
149
149
|
traces_data_dir = output_paths['base'] / 'traces_data'
|
|
150
150
|
|
|
151
|
+
overwrite = config.get('overwrite', False)
|
|
152
|
+
|
|
151
153
|
if from_raw_traces:
|
|
152
|
-
# Process raw JSON traces
|
|
154
|
+
# Process raw JSON traces — always runs; skips individual CSVs that
|
|
155
|
+
# already exist when overwrite=false (handles interrupted runs)
|
|
153
156
|
logger.info(f"Preprocessing raw traces from: {data_dir}")
|
|
154
|
-
|
|
157
|
+
|
|
155
158
|
try:
|
|
156
159
|
process_traces_to_traj_data(
|
|
157
160
|
input_dir=str(data_dir),
|
|
158
161
|
output_dir=str(traces_data_dir),
|
|
159
162
|
agent_framework=config.get('agent_framework'),
|
|
160
163
|
observability_framework=config.get('observability_framework'),
|
|
161
|
-
overwrite=
|
|
164
|
+
overwrite=overwrite,
|
|
162
165
|
)
|
|
163
166
|
logger.info(f"✓ Traces preprocessed")
|
|
164
167
|
return traces_data_dir
|
|
@@ -167,8 +170,14 @@ def prepare_traces_data(
|
|
|
167
170
|
return None
|
|
168
171
|
else:
|
|
169
172
|
# Copy existing CSV files
|
|
170
|
-
|
|
171
|
-
|
|
173
|
+
# If not overwriting and traces_data already has CSVs, reuse as-is
|
|
174
|
+
if not overwrite and traces_data_dir.exists():
|
|
175
|
+
csv_count = len(list(traces_data_dir.glob('*.csv')))
|
|
176
|
+
if csv_count > 0:
|
|
177
|
+
logger.info(f"Reusing existing traces_data ({csv_count} CSVs)")
|
|
178
|
+
return traces_data_dir
|
|
179
|
+
|
|
180
|
+
logger.info(f"Copying trajectory CSVs from: {data_dir}")
|
|
172
181
|
try:
|
|
173
182
|
if traces_data_dir.exists():
|
|
174
183
|
shutil.rmtree(traces_data_dir)
|
|
@@ -325,20 +334,16 @@ def main():
|
|
|
325
334
|
# Extract parameters
|
|
326
335
|
data_dir = Path(config['data_dir'])
|
|
327
336
|
from_raw_traces = config.get('from_raw_traces')
|
|
328
|
-
overwrite = config.get('overwrite', True)
|
|
329
|
-
user_provided_run_name = config.get('run_name') is not None
|
|
330
337
|
|
|
331
338
|
# Validate input directory exists
|
|
332
339
|
if not data_dir.exists():
|
|
333
340
|
parser.error(f"Input directory does not exist: {data_dir}")
|
|
334
341
|
|
|
335
|
-
#
|
|
336
|
-
if output_dir.exists()
|
|
342
|
+
# Log if output dir already exists
|
|
343
|
+
if output_dir.exists():
|
|
344
|
+
overwrite = config.get('overwrite', False)
|
|
337
345
|
if overwrite:
|
|
338
|
-
|
|
339
|
-
f"Output directory already exists: {output_dir}\n"
|
|
340
|
-
f"Use --overwrite false to resume, or delete the existing directory."
|
|
341
|
-
)
|
|
346
|
+
logger.info(f"Re-running existing run with overwrite: {run_name}")
|
|
342
347
|
else:
|
|
343
348
|
logger.info(f"Resuming existing run: {run_name}")
|
|
344
349
|
|
{clear_eval-2.0.1 → clear_eval-2.0.2}/src/clear_eval/agentic/pipeline/run_clear_step_analysis.py
RENAMED
|
@@ -642,7 +642,6 @@ Argument Precedence (lowest to highest):
|
|
|
642
642
|
validate_required_config(config_dict, ['data_dir', 'results_dir'], parser)
|
|
643
643
|
|
|
644
644
|
# Get run output directory
|
|
645
|
-
user_provided_run_name = config_dict.get('run_name') is not None
|
|
646
645
|
results_dir, run_name = get_run_output_dir(
|
|
647
646
|
config_dict['results_dir'],
|
|
648
647
|
config_dict.get('run_name')
|
|
@@ -655,16 +654,13 @@ Argument Precedence (lowest to highest):
|
|
|
655
654
|
# Extract parameters
|
|
656
655
|
from_raw_traces = config_dict.get('from_raw_traces')
|
|
657
656
|
data_dir = config_dict['data_dir']
|
|
658
|
-
overwrite = config_dict.get('overwrite',
|
|
657
|
+
overwrite = config_dict.get('overwrite', False)
|
|
659
658
|
memory_only = config_dict.get('memory_only')
|
|
660
659
|
|
|
661
|
-
#
|
|
662
|
-
if results_dir.exists()
|
|
660
|
+
# Log if output dir already exists
|
|
661
|
+
if results_dir.exists():
|
|
663
662
|
if overwrite:
|
|
664
|
-
|
|
665
|
-
f"Output directory already exists: {results_dir}\n"
|
|
666
|
-
f"Use --overwrite false to resume, or delete the existing directory."
|
|
667
|
-
)
|
|
663
|
+
logger.info(f"Re-running existing run with overwrite: {run_name}")
|
|
668
664
|
else:
|
|
669
665
|
logger.info(f"Resuming existing run: {run_name}")
|
|
670
666
|
|
|
@@ -25,7 +25,7 @@ separate_tools: false
|
|
|
25
25
|
max_files: null # Maximum number of files to evaluate (null = all)
|
|
26
26
|
|
|
27
27
|
# Execution
|
|
28
|
-
overwrite:
|
|
28
|
+
overwrite: false # If true: re-run steps even if outputs exist. If false: skip completed steps.
|
|
29
29
|
max_workers: 10 # Number of parallel workers
|
|
30
30
|
run_name: null # Unique identifier for this run (used in output path)
|
|
31
31
|
|