eval-framework 0.11.6__tar.gz → 0.11.8__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {eval_framework-0.11.6 → eval_framework-0.11.8}/PKG-INFO +5 -5
- {eval_framework-0.11.6 → eval_framework-0.11.8}/pyproject.toml +6 -6
- {eval_framework-0.11.6 → eval_framework-0.11.8}/pyproject.toml.orig +6 -6
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/evaluation_generator.py +11 -5
- {eval_framework-0.11.6 → eval_framework-0.11.8}/LICENSE +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/README.md +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/__init__.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/base_config.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/benchmarks/__init__.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/benchmarks/arc_de.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/benchmarks/csqa_ellamind.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/benchmarks/gpqa_ellamind.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/benchmarks/hellaswag_ellamind.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/benchmarks/hle_ellamind.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/benchmarks/piqa_ellamind.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/benchmarks/simpleqa_ellamind.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/benchmarks/siqa_ellamind.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/benchmarks/winogrande_ellamind.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/choices.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/composed.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/context/__init__.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/context/determined.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/context/eval.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/context/local.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/contract.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/eval_kind.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/exceptions.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/external/drop_process_results.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/external/ifeval_impl/README.md +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/external/ifeval_impl/instructions.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/external/ifeval_impl/instructions_registry.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/external/ifeval_impl/instructions_util.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/external/ifeval_impl/utils.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/llm/__init__.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/llm/aleph_alpha.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/llm/base.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/llm/huggingface.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/llm/models.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/llm/openai.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/logger.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/main.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/__init__.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/aggregators/__init__.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/aggregators/aggregators.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/base.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/completion/__init__.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/completion/accuracy_completion.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/completion/code_assertion.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/completion/code_execution_pass_at_one.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/completion/concordance_index.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/completion/csv_format.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/completion/drop_completion.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/completion/exponential_similarity.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/completion/f1.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/completion/format_checker.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/completion/grid_difference.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/completion/ifeval.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/completion/json_format.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/completion/language_checker.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/completion/length_control.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/completion/math_minerva_completion.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/completion/math_reasoning_completion.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/completion/minerva_math_utils.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/completion/multipl_e_assertion.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/completion/placeholder_checker.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/completion/repetition.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/completion/rouge_1.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/completion/rouge_2.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/completion/rouge_geometric_mean.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/completion/rouge_l.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/completion/text_counter.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/efficiency/__init__.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/efficiency/bytes_per_sequence_position.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/efficiency/token_counters.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/llm/__init__.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/llm/base.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/llm/graders/chatbot_style_grader.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/llm/graders/coherence_grader.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/llm/graders/comparison_grader.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/llm/graders/conciseness_grader.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/llm/graders/contains_names_grader.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/llm/graders/format_correctness_grader.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/llm/graders/instruction_grader.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/llm/graders/language.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/llm/graders/long_context_grader.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/llm/graders/models.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/llm/graders/refusal_grader.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/llm/graders/sql_quality_grader.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/llm/graders/summary_world_knowledge_grader.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/llm/llm_judge_chatbot_style.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/llm/llm_judge_coherence.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/llm/llm_judge_completion_accuracy.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/llm/llm_judge_conciseness.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/llm/llm_judge_contains_names.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/llm/llm_judge_format_correctness.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/llm/llm_judge_instruction.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/llm/llm_judge_mtbench_pair.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/llm/llm_judge_mtbench_single.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/llm/llm_judge_refusal.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/llm/llm_judge_sql.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/llm/llm_judge_world_knowledge.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/llm/utils.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/loglikelihood/__init__.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/loglikelihood/accuracy_loglikelihood.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/loglikelihood/base.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/loglikelihood/bits_per_byte.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/loglikelihood/confidence_weighted_accuracy.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/loglikelihood/dcs.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/loglikelihood/probability_mass.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/loglikelihood/ternary.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/py.typed +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/response_generator.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/result_processors/__init__.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/result_processors/base.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/result_processors/hf_uploader.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/result_processors/result_processor.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/result_processors/wandb_uploader.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/run.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/run_direct.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/shared/errors.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/shared/types.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/subjects.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/suite.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/Dockerfile_codebench +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/__init__.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/base.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/benchmarks/__init__.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/benchmarks/arc.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/benchmarks/arc_ellamind.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/benchmarks/bigcodebench.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/benchmarks/copa.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/benchmarks/csqa.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/benchmarks/drop.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/benchmarks/global_mmlu.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/benchmarks/goldenswag.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/benchmarks/gpqa.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/benchmarks/gsm8k.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/benchmarks/gsm8k_ellamind.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/benchmarks/hellaswag.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/benchmarks/hendrycks_math_ellamind.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/benchmarks/humaneval.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/benchmarks/humaneval_ellamind.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/benchmarks/ifeval.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/benchmarks/math_reasoning.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/benchmarks/mbpp.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/benchmarks/mbpp_ellamind.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/benchmarks/medqa.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/benchmarks/mmlu.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/benchmarks/mmlu_pro.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/benchmarks/multipl_e.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/benchmarks/naturalqs_open.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/benchmarks/piqa.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/benchmarks/sciq.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/benchmarks/social_iqa.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/benchmarks/squad.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/benchmarks/winogrande.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/dataset_loading.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/dataset_revisions.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/eval_config.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/frozen-hf-dataset-revisions.json +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/hf-dataset-revisions.json +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/lazy.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/markdown_doc.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/registry.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/task_loader.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/task_names.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/task_style.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/utils.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/utils/constants.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/utils/file_ops.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/utils/helpers.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/utils/logging.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/utils/packaging.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/utils/tqdm_handler.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/template_formatting/README.md +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/template_formatting/__init__.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/template_formatting/formatter.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/template_formatting/mistral_formatter.py +0 -0
- {eval_framework-0.11.6 → eval_framework-0.11.8}/src/template_formatting/py.typed +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.3
|
|
2
2
|
Name: eval-framework
|
|
3
|
-
Version: 0.11.
|
|
3
|
+
Version: 0.11.8
|
|
4
4
|
Summary: Evaluation Framework
|
|
5
5
|
Author: Aleph Alpha Research
|
|
6
6
|
License: Apache License
|
|
@@ -229,10 +229,10 @@ Requires-Dist: psycopg2-binary>=2.9.12,<3
|
|
|
229
229
|
Requires-Dist: sympy>=1.14.0,<2
|
|
230
230
|
Requires-Dist: llm-sandbox[docker]==0.3.44
|
|
231
231
|
Requires-Dist: jsonlines>=4,<5
|
|
232
|
-
Requires-Dist: lxml>=6.1.
|
|
232
|
+
Requires-Dist: lxml>=6.1.3,<7
|
|
233
233
|
Requires-Dist: python-iso639>=2026.7.23
|
|
234
234
|
Requires-Dist: wandb>=0.29.0,<1
|
|
235
|
-
Requires-Dist: boto3>=1.43.
|
|
235
|
+
Requires-Dist: boto3>=1.43.87,<2
|
|
236
236
|
Requires-Dist: numpy>=2.5.2
|
|
237
237
|
Requires-Dist: antlr4-python3-runtime==4.11.0
|
|
238
238
|
Requires-Dist: scipy>=1.18.1,<2
|
|
@@ -241,13 +241,13 @@ Requires-Dist: eval-framework[determined,api,openai,transformers,accelerate,opti
|
|
|
241
241
|
Requires-Dist: aleph-alpha-client>=11.5.1 ; extra == 'api'
|
|
242
242
|
Requires-Dist: determined>=0.38.1,<0.39 ; extra == 'determined'
|
|
243
243
|
Requires-Dist: tensorboard==2.21.0 ; extra == 'determined'
|
|
244
|
-
Requires-Dist: openai>=3.
|
|
244
|
+
Requires-Dist: openai>=3.7.0,<4 ; extra == 'openai'
|
|
245
245
|
Requires-Dist: tiktoken>=0.14.0,<1 ; extra == 'openai'
|
|
246
246
|
Requires-Dist: transformers>=4.45.2,<5 ; extra == 'openai'
|
|
247
247
|
Requires-Dist: transformers>=4.45.2,<5 ; extra == 'optional'
|
|
248
248
|
Requires-Dist: jinja2>=3.1.6,<4 ; extra == 'optional'
|
|
249
249
|
Requires-Dist: transformers>=4.45.2,<5 ; extra == 'transformers'
|
|
250
|
-
Requires-Dist: torch>=2.
|
|
250
|
+
Requires-Dist: torch>=2.14.0,<3 ; extra == 'transformers'
|
|
251
251
|
Requires-Dist: accelerate>=1.14.0,<2 ; extra == 'transformers'
|
|
252
252
|
Requires-Python: >=3.12, <3.15
|
|
253
253
|
Project-URL: repository, https://github.com/Aleph-Alpha-Research/eval-framework
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "eval-framework"
|
|
3
|
-
version = "0.11.
|
|
3
|
+
version = "0.11.8"
|
|
4
4
|
description = "Evaluation Framework"
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
requires-python = ">=3.12,<3.15"
|
|
@@ -32,10 +32,10 @@ dependencies = [
|
|
|
32
32
|
"sympy>=1.14.0,<2",
|
|
33
33
|
"llm-sandbox[docker]==0.3.44",
|
|
34
34
|
"jsonlines>=4,<5",
|
|
35
|
-
"lxml>=6.1.
|
|
35
|
+
"lxml>=6.1.3,<7",
|
|
36
36
|
"python-iso639>=2026.7.23",
|
|
37
37
|
"wandb>=0.29.0,<1",
|
|
38
|
-
"boto3>=1.43.
|
|
38
|
+
"boto3>=1.43.87,<2",
|
|
39
39
|
"numpy>=2.5.2",
|
|
40
40
|
"antlr4-python3-runtime==4.11.0",
|
|
41
41
|
"scipy>=1.18.1,<2",
|
|
@@ -54,13 +54,13 @@ determined = [
|
|
|
54
54
|
]
|
|
55
55
|
api = ["aleph-alpha-client>=11.5.1"]
|
|
56
56
|
openai = [
|
|
57
|
-
"openai>=3.
|
|
57
|
+
"openai>=3.7.0,<4",
|
|
58
58
|
"tiktoken>=0.14.0,<1",
|
|
59
59
|
"transformers>=4.45.2,<5",
|
|
60
60
|
]
|
|
61
61
|
transformers = [
|
|
62
62
|
"transformers>=4.45.2,<5",
|
|
63
|
-
"torch>=2.
|
|
63
|
+
"torch>=2.14.0,<3",
|
|
64
64
|
"accelerate>=1.14.0,<2",
|
|
65
65
|
]
|
|
66
66
|
accelerate = ["accelerate"]
|
|
@@ -96,7 +96,7 @@ flash-attn = [
|
|
|
96
96
|
]
|
|
97
97
|
|
|
98
98
|
[build-system]
|
|
99
|
-
requires = ["uv_build>=0.12.
|
|
99
|
+
requires = ["uv_build>=0.12.9,<0.12.10"]
|
|
100
100
|
build-backend = "uv_build"
|
|
101
101
|
|
|
102
102
|
[tool.uv]
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "eval-framework"
|
|
3
|
-
version = "0.11.
|
|
3
|
+
version = "0.11.8"
|
|
4
4
|
description = "Evaluation Framework"
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
license = { file = "LICENSE" }
|
|
@@ -36,10 +36,10 @@ dependencies = [
|
|
|
36
36
|
"sympy>=1.14.0,<2",
|
|
37
37
|
"llm-sandbox[docker]==0.3.44",
|
|
38
38
|
"jsonlines>=4,<5",
|
|
39
|
-
"lxml>=6.1.
|
|
39
|
+
"lxml>=6.1.3,<7",
|
|
40
40
|
"python-iso639>=2026.7.23",
|
|
41
41
|
"wandb>=0.29.0,<1",
|
|
42
|
-
"boto3>=1.43.
|
|
42
|
+
"boto3>=1.43.87,<2",
|
|
43
43
|
"numpy>=2.5.2",
|
|
44
44
|
# is a dependency of sympy, but not explicitly listed in the requirements.txt
|
|
45
45
|
# https://github.com/sympy/sympy/blob/0204fa34e8f6f6f8ccb4de01209be9a2345c9d6e/doc/src/contributing/dependencies.md?plain=1#L125
|
|
@@ -55,13 +55,13 @@ determined = [
|
|
|
55
55
|
]
|
|
56
56
|
api = ["aleph-alpha-client>=11.5.1"]
|
|
57
57
|
openai = [
|
|
58
|
-
"openai>=3.
|
|
58
|
+
"openai>=3.7.0,<4",
|
|
59
59
|
"tiktoken>=0.14.0,<1",
|
|
60
60
|
"transformers>=4.45.2,<5",
|
|
61
61
|
]
|
|
62
62
|
transformers = [
|
|
63
63
|
"transformers>=4.45.2,<5",
|
|
64
|
-
"torch>=2.
|
|
64
|
+
"torch>=2.14.0,<3",
|
|
65
65
|
"accelerate>=1.14.0,<2",
|
|
66
66
|
]
|
|
67
67
|
accelerate = ["accelerate"]
|
|
@@ -102,7 +102,7 @@ flash-attn = [
|
|
|
102
102
|
]
|
|
103
103
|
|
|
104
104
|
[build-system]
|
|
105
|
-
requires = ["uv_build>=0.12.
|
|
105
|
+
requires = ["uv_build>=0.12.9,<0.12.10"]
|
|
106
106
|
build-backend = "uv_build"
|
|
107
107
|
|
|
108
108
|
[tool.uv.build-backend]
|
|
@@ -136,9 +136,12 @@ class EvaluationGenerator:
|
|
|
136
136
|
error_free_ratio = float(len(data_subset_error_free) / total_count)
|
|
137
137
|
aggregated_results[f"ErrorFreeRatio {metric}"] = error_free_ratio
|
|
138
138
|
|
|
139
|
+
# Default average is weighted by key and subject (macro average).
|
|
139
140
|
# aggregate by key and subject first to have equal weights for all key / subject combinations
|
|
140
141
|
key_subject_mean = data_subset_error_free.groupby(["key", "subject"]).mean()
|
|
141
142
|
aggregated_results[f"Average {metric}"] = float(key_subject_mean[["value"]].mean()["value"])
|
|
143
|
+
# Additionally, report the plain mean over all samples (micro average or per-sample average).
|
|
144
|
+
aggregated_results[f"Average {metric} (micro)"] = float(data_subset_error_free["value"].mean())
|
|
142
145
|
|
|
143
146
|
if error_free_ratio < 1.0:
|
|
144
147
|
# Treat error samples (with value=None) as 0 for the "including errors" average
|
|
@@ -274,12 +277,15 @@ class EvaluationGenerator:
|
|
|
274
277
|
raise ValueError(f"Metric {metric_name} not found in metrics list")
|
|
275
278
|
|
|
276
279
|
for aggregator in current_metric.AGGREGATORS:
|
|
280
|
+
# Compute the aggregator per problem (collapsing the repeats of each problem into one score).
|
|
281
|
+
per_problem = aggregator(metric_group, ["subject", "problem"])
|
|
282
|
+
# Macro average: mean per key/subject group, then mean over groups, giving equal weight to every group.
|
|
277
283
|
aggregated_results[f"{aggregator.name} {current_metric_class}.{metric_name}"] = (
|
|
278
|
-
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
.item()
|
|
284
|
+
per_problem.groupby(["key", "subject"]).agg({"value": "mean"})["value"].mean().item()
|
|
285
|
+
)
|
|
286
|
+
# Micro average: plain mean over all problems.
|
|
287
|
+
aggregated_results[f"{aggregator.name} {current_metric_class}.{metric_name} (micro)"] = (
|
|
288
|
+
per_problem["value"].mean().item()
|
|
283
289
|
)
|
|
284
290
|
|
|
285
291
|
# Loop to additionally compute per-subject/per-key breakdown metric scores, e.g. for only subject="algebra"
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/benchmarks/csqa_ellamind.py
RENAMED
|
File without changes
|
{eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/benchmarks/gpqa_ellamind.py
RENAMED
|
File without changes
|
{eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/benchmarks/hellaswag_ellamind.py
RENAMED
|
File without changes
|
{eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/benchmarks/hle_ellamind.py
RENAMED
|
File without changes
|
{eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/benchmarks/piqa_ellamind.py
RENAMED
|
File without changes
|
{eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/benchmarks/simpleqa_ellamind.py
RENAMED
|
File without changes
|
{eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/benchmarks/siqa_ellamind.py
RENAMED
|
File without changes
|
{eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/benchmarks/winogrande_ellamind.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/external/drop_process_results.py
RENAMED
|
File without changes
|
{eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/external/ifeval_impl/README.md
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/external/ifeval_impl/utils.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/aggregators/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/completion/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/completion/csv_format.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/completion/ifeval.py
RENAMED
|
File without changes
|
{eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/completion/json_format.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/completion/repetition.py
RENAMED
|
File without changes
|
{eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/completion/rouge_1.py
RENAMED
|
File without changes
|
{eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/completion/rouge_2.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/completion/rouge_l.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/efficiency/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/llm/graders/language.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/llm/graders/models.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/llm/llm_judge_refusal.py
RENAMED
|
File without changes
|
{eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/llm/llm_judge_sql.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/loglikelihood/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/loglikelihood/base.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/loglikelihood/dcs.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/metrics/loglikelihood/ternary.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/result_processors/__init__.py
RENAMED
|
File without changes
|
{eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/result_processors/base.py
RENAMED
|
File without changes
|
{eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/result_processors/hf_uploader.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/Dockerfile_codebench
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/benchmarks/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/benchmarks/arc_ellamind.py
RENAMED
|
File without changes
|
{eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/benchmarks/bigcodebench.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/benchmarks/global_mmlu.py
RENAMED
|
File without changes
|
{eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/benchmarks/goldenswag.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/benchmarks/gsm8k.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/benchmarks/hellaswag.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/benchmarks/humaneval.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/benchmarks/ifeval.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/benchmarks/mbpp_ellamind.py
RENAMED
|
File without changes
|
{eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/benchmarks/medqa.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/benchmarks/mmlu_pro.py
RENAMED
|
File without changes
|
{eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/benchmarks/multipl_e.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/benchmarks/social_iqa.py
RENAMED
|
File without changes
|
{eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/benchmarks/squad.py
RENAMED
|
File without changes
|
{eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/benchmarks/winogrande.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/dataset_revisions.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.11.6 → eval_framework-0.11.8}/src/eval_framework/tasks/hf-dataset-revisions.json
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.11.6 → eval_framework-0.11.8}/src/template_formatting/mistral_formatter.py
RENAMED
|
File without changes
|
|
File without changes
|