eval-framework 0.8.11__tar.gz → 0.9.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {eval_framework-0.8.11 → eval_framework-0.9.0}/PKG-INFO +3 -3
- {eval_framework-0.8.11 → eval_framework-0.9.0}/pyproject.toml +3 -3
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/response_generator.py +1 -1
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/base.py +35 -2
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/gpqa.py +5 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/markdown_doc.py +1 -1
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/registry.py +28 -23
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/task_names.py +1 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/LICENSE +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/README.md +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/__init__.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/base_config.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/context/__init__.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/context/determined.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/context/eval.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/context/local.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/evaluation_generator.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/exceptions.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/external/drop_process_results.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/external/ifeval_impl/README.md +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/external/ifeval_impl/instructions.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/external/ifeval_impl/instructions_registry.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/external/ifeval_impl/instructions_util.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/external/ifeval_impl/utils.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/llm/__init__.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/llm/aleph_alpha.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/llm/base.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/llm/huggingface.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/llm/models.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/llm/openai.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/logger.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/main.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/__init__.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/aggregators/__init__.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/aggregators/aggregators.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/base.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/__init__.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/accuracy_completion.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/code_assertion.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/code_execution_pass_at_one.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/concordance_index.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/csv_format.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/drop_completion.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/exponential_similarity.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/f1.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/format_checker.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/grid_difference.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/ifeval.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/json_format.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/language_checker.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/length_control.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/math_minerva_completion.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/math_reasoning_completion.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/minerva_math_utils.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/multipl_e_assertion.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/placeholder_checker.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/repetition.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/rouge_1.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/rouge_2.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/rouge_geometric_mean.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/rouge_l.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/text_counter.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/efficiency/__init__.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/efficiency/bytes_per_sequence_position.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/__init__.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/base.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/graders/chatbot_style_grader.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/graders/coherence_grader.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/graders/comparison_grader.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/graders/conciseness_grader.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/graders/contains_names_grader.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/graders/format_correctness_grader.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/graders/instruction_grader.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/graders/language.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/graders/long_context_grader.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/graders/models.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/graders/refusal_grader.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/graders/sql_quality_grader.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/graders/summary_world_knowledge_grader.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/llm_judge_chatbot_style.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/llm_judge_coherence.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/llm_judge_completion_accuracy.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/llm_judge_conciseness.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/llm_judge_contains_names.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/llm_judge_format_correctness.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/llm_judge_instruction.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/llm_judge_mtbench_pair.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/llm_judge_mtbench_single.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/llm_judge_refusal.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/llm_judge_sql.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/llm_judge_world_knowledge.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/utils.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/loglikelihood/__init__.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/loglikelihood/accuracy_loglikelihood.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/loglikelihood/base.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/loglikelihood/bits_per_byte.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/loglikelihood/confidence_weighted_accuracy.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/loglikelihood/dcs.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/loglikelihood/probability_mass.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/loglikelihood/ternary.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/py.typed +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/result_processors/__init__.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/result_processors/base.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/result_processors/hf_uploader.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/result_processors/result_processor.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/result_processors/wandb_uploader.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/run.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/run_direct.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/shared/types.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/suite.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/Dockerfile_codebench +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/__init__.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/__init__.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/arc.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/arc_de.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/bigcodebench.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/copa.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/csqa.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/drop.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/global_mmlu.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/goldenswag.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/gsm8k.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/hellaswag.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/humaneval.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/ifeval.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/math_reasoning.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/mbpp.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/medqa.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/mmlu.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/mmlu_pro.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/multipl_e.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/naturalqs_open.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/piqa.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/sciq.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/social_iqa.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/squad.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/winogrande.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/dataset_revisions.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/eval_config.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/frozen-hf-dataset-revisions.json +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/hf-dataset-revisions.json +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/perturbation.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/task_loader.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/task_style.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/utils.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/utils/constants.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/utils/file_ops.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/utils/helpers.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/utils/logging.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/utils/packaging.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/utils/tqdm_handler.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/template_formatting/README.md +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/template_formatting/__init__.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/template_formatting/formatter.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/template_formatting/mistral_formatter.py +0 -0
- {eval_framework-0.8.11 → eval_framework-0.9.0}/src/template_formatting/py.typed +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.3
|
|
2
2
|
Name: eval-framework
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.9.0
|
|
4
4
|
Summary: Evaluation Framework
|
|
5
5
|
Author: Aleph Alpha Research
|
|
6
6
|
License: Apache License
|
|
@@ -226,7 +226,7 @@ Requires-Dist: jsonschema>=4.26.0,<5
|
|
|
226
226
|
Requires-Dist: mysql-connector-python>=26.7.0,<27
|
|
227
227
|
Requires-Dist: psycopg2-binary>=2.9.12,<3
|
|
228
228
|
Requires-Dist: sympy>=1.14.0,<2
|
|
229
|
-
Requires-Dist: llm-sandbox[docker]==0.3.
|
|
229
|
+
Requires-Dist: llm-sandbox[docker]==0.3.44
|
|
230
230
|
Requires-Dist: jsonlines>=4,<5
|
|
231
231
|
Requires-Dist: lxml>=6.1.1,<7
|
|
232
232
|
Requires-Dist: python-iso639>=2026.7.23
|
|
@@ -240,7 +240,7 @@ Requires-Dist: eval-framework[determined,api,openai,transformers,accelerate,opti
|
|
|
240
240
|
Requires-Dist: aleph-alpha-client>=11.5.1 ; extra == 'api'
|
|
241
241
|
Requires-Dist: determined>=0.38.1,<0.39 ; extra == 'determined'
|
|
242
242
|
Requires-Dist: tensorboard==2.21.0 ; extra == 'determined'
|
|
243
|
-
Requires-Dist: openai>=2.
|
|
243
|
+
Requires-Dist: openai>=2.53.0,<3 ; extra == 'openai'
|
|
244
244
|
Requires-Dist: tiktoken>=0.13.0,<1 ; extra == 'openai'
|
|
245
245
|
Requires-Dist: transformers>=4.45.2,<5 ; extra == 'openai'
|
|
246
246
|
Requires-Dist: transformers>=4.45.2,<5 ; extra == 'optional'
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "eval-framework"
|
|
3
|
-
version = "0.
|
|
3
|
+
version = "0.9.0"
|
|
4
4
|
description = "Evaluation Framework"
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
license = { file = "LICENSE" }
|
|
@@ -33,7 +33,7 @@ dependencies = [
|
|
|
33
33
|
"mysql-connector-python>=26.7.0,<27", # required for sql-related tasks
|
|
34
34
|
"psycopg2-binary>=2.9.12,<3", # required for sql-related tasks
|
|
35
35
|
"sympy>=1.14.0,<2",
|
|
36
|
-
"llm-sandbox[docker]==0.3.
|
|
36
|
+
"llm-sandbox[docker]==0.3.44",
|
|
37
37
|
"jsonlines>=4,<5",
|
|
38
38
|
"lxml>=6.1.1,<7",
|
|
39
39
|
"python-iso639>=2026.7.23",
|
|
@@ -54,7 +54,7 @@ determined = [
|
|
|
54
54
|
]
|
|
55
55
|
api = ["aleph-alpha-client>=11.5.1"]
|
|
56
56
|
openai = [
|
|
57
|
-
"openai>=2.
|
|
57
|
+
"openai>=2.53.0,<3",
|
|
58
58
|
"tiktoken>=0.13.0,<1",
|
|
59
59
|
"transformers>=4.45.2,<5",
|
|
60
60
|
]
|
|
@@ -179,7 +179,7 @@ class ResponseGenerator:
|
|
|
179
179
|
:param should_preempt_callable: function to check if preempt is called
|
|
180
180
|
:return: list of responses, preempted
|
|
181
181
|
"""
|
|
182
|
-
logger.info(f"{RED}[ Running task {self.task.
|
|
182
|
+
logger.info(f"{RED}[ Running task {self.task.display_name()} against model ------------ ]{RESET}")
|
|
183
183
|
self.start_time, monotonic_start = time.time(), time.monotonic()
|
|
184
184
|
run_fn = self._generative_output_type_selector()
|
|
185
185
|
self._verify_loaded_metadata_compatibility()
|
|
@@ -2,7 +2,7 @@ import logging
|
|
|
2
2
|
import os
|
|
3
3
|
import random
|
|
4
4
|
import traceback
|
|
5
|
-
from abc import ABC
|
|
5
|
+
from abc import ABC, abstractmethod
|
|
6
6
|
from collections.abc import Iterable, Sequence
|
|
7
7
|
from enum import Enum
|
|
8
8
|
from pathlib import Path
|
|
@@ -86,7 +86,37 @@ SubjectType = TypeVar("SubjectType")
|
|
|
86
86
|
logger = logging.getLogger(__name__)
|
|
87
87
|
|
|
88
88
|
|
|
89
|
-
class
|
|
89
|
+
class Task(ABC):
|
|
90
|
+
"""The contract a caller relies on to run an evaluation"""
|
|
91
|
+
|
|
92
|
+
@abstractmethod
|
|
93
|
+
def iterate_samples(self, num_samples: int | None = None) -> Iterable[Sample]: ...
|
|
94
|
+
|
|
95
|
+
@abstractmethod
|
|
96
|
+
def generate_completions(
|
|
97
|
+
self,
|
|
98
|
+
llm: "BaseLLM",
|
|
99
|
+
samples: list[Sample],
|
|
100
|
+
stop_sequences: list[str] | None = None,
|
|
101
|
+
max_tokens: int | None = None,
|
|
102
|
+
fail_on_error: bool = True,
|
|
103
|
+
) -> list[Completion]:
|
|
104
|
+
"""Run ``llm`` over ``samples`` and return their completions."""
|
|
105
|
+
|
|
106
|
+
@abstractmethod
|
|
107
|
+
def get_metadata(self) -> dict[str, str | list[str]]:
|
|
108
|
+
"""Descriptive metadata about the eval for result reporting."""
|
|
109
|
+
|
|
110
|
+
@abstractmethod
|
|
111
|
+
def get_response_type(self) -> ResponseType: ...
|
|
112
|
+
|
|
113
|
+
@abstractmethod
|
|
114
|
+
def display_name(self) -> str:
|
|
115
|
+
"""Human-readable display name. Is allowed to have special characters and whitespaces."""
|
|
116
|
+
...
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
class BaseTask[SubjectType](Task):
|
|
90
120
|
NAME: str
|
|
91
121
|
DATASET_PATH: str
|
|
92
122
|
SAMPLE_SPLIT: str
|
|
@@ -497,3 +527,6 @@ class BaseTask[SubjectType](ABC):
|
|
|
497
527
|
def METRICS(cls) -> list[type["BaseMetric"]]:
|
|
498
528
|
"""For backwards compatibility."""
|
|
499
529
|
return cls.get_metrics()
|
|
530
|
+
|
|
531
|
+
def display_name(self) -> str:
|
|
532
|
+
return self.NAME
|
|
@@ -47,7 +47,7 @@ def markdown_doc(
|
|
|
47
47
|
buf.write(f"- Module: `{module}`\n\n")
|
|
48
48
|
|
|
49
49
|
if http_path:
|
|
50
|
-
buf.write(f"- Link to dataset: [{http_path}]({http_path})\n
|
|
50
|
+
buf.write(f"- Link to dataset: [{http_path}]({http_path})\n")
|
|
51
51
|
else:
|
|
52
52
|
assert example_messages is not None, "a task without a dataset link must supply an example sample"
|
|
53
53
|
for split, size in (split_sizes or {}).items():
|
|
@@ -18,7 +18,6 @@ __all__ = [
|
|
|
18
18
|
"EvalFactory",
|
|
19
19
|
"Registry",
|
|
20
20
|
"with_registry",
|
|
21
|
-
"get_task",
|
|
22
21
|
"is_registered",
|
|
23
22
|
"validate_task_name",
|
|
24
23
|
"registered_task_names",
|
|
@@ -36,8 +35,8 @@ class EvalFactory(ABC):
|
|
|
36
35
|
"""
|
|
37
36
|
|
|
38
37
|
@abstractmethod
|
|
39
|
-
def
|
|
40
|
-
"
|
|
38
|
+
def id(self) -> str:
|
|
39
|
+
"Canonical key used to register this benchmark"
|
|
41
40
|
|
|
42
41
|
@property
|
|
43
42
|
@abstractmethod
|
|
@@ -75,13 +74,10 @@ class EvalFactory(ABC):
|
|
|
75
74
|
user_prompt_suffix: str | None = None,
|
|
76
75
|
) -> BaseTask: ...
|
|
77
76
|
|
|
77
|
+
@abstractmethod
|
|
78
78
|
def markdown_doc(self, formatters: Sequence[BaseFormatter]) -> str:
|
|
79
79
|
"""Render the eval's documentation as markdown."""
|
|
80
|
-
|
|
81
|
-
task = self.create(num_fewshot=1, custom_subjects=None, custom_hf_revision=None)
|
|
82
|
-
except (TypeError, ValueError, AssertionError):
|
|
83
|
-
task = self.create(num_fewshot=0, custom_subjects=None, custom_hf_revision=None)
|
|
84
|
-
return task.markdown_doc(formatters)
|
|
80
|
+
...
|
|
85
81
|
|
|
86
82
|
|
|
87
83
|
class _Lazy(EvalFactory):
|
|
@@ -104,6 +100,9 @@ class _Lazy(EvalFactory):
|
|
|
104
100
|
def source_module(self) -> str:
|
|
105
101
|
return self._module
|
|
106
102
|
|
|
103
|
+
def id(self) -> str:
|
|
104
|
+
return self._class_name
|
|
105
|
+
|
|
107
106
|
def task_class(self) -> type[BaseTask]:
|
|
108
107
|
if self._loaded is None:
|
|
109
108
|
module = importlib.import_module(self._module)
|
|
@@ -152,6 +151,13 @@ class _Lazy(EvalFactory):
|
|
|
152
151
|
"""The eval's human-readable display name (the task's ``NAME``)."""
|
|
153
152
|
return self.task_class().NAME
|
|
154
153
|
|
|
154
|
+
def markdown_doc(self, formatters: Sequence[BaseFormatter]) -> str:
|
|
155
|
+
try:
|
|
156
|
+
task = self.create(num_fewshot=1, custom_subjects=None, custom_hf_revision=None)
|
|
157
|
+
except (TypeError, ValueError, AssertionError):
|
|
158
|
+
task = self.create(num_fewshot=0, custom_subjects=None, custom_hf_revision=None)
|
|
159
|
+
return task.markdown_doc(formatters)
|
|
160
|
+
|
|
155
161
|
|
|
156
162
|
class _Eager(EvalFactory):
|
|
157
163
|
"""Wraps an already-imported task class."""
|
|
@@ -163,8 +169,8 @@ class _Eager(EvalFactory):
|
|
|
163
169
|
def source_module(self) -> str:
|
|
164
170
|
return self._task.__module__
|
|
165
171
|
|
|
166
|
-
def
|
|
167
|
-
return self._task
|
|
172
|
+
def id(self) -> str:
|
|
173
|
+
return self._task.__name__
|
|
168
174
|
|
|
169
175
|
def create(
|
|
170
176
|
self,
|
|
@@ -173,7 +179,7 @@ class _Eager(EvalFactory):
|
|
|
173
179
|
custom_hf_revision: str | None,
|
|
174
180
|
user_prompt_suffix: str | None = None,
|
|
175
181
|
) -> BaseTask:
|
|
176
|
-
return self.
|
|
182
|
+
return self._task.with_overwrite(
|
|
177
183
|
num_fewshot=num_fewshot,
|
|
178
184
|
custom_subjects=custom_subjects,
|
|
179
185
|
custom_hf_revision=custom_hf_revision,
|
|
@@ -188,7 +194,7 @@ class _Eager(EvalFactory):
|
|
|
188
194
|
custom_hf_revision: str | None,
|
|
189
195
|
user_prompt_suffix: str | None = None,
|
|
190
196
|
) -> BaseTask:
|
|
191
|
-
perturbation_task_class = create_perturbation_class(self.
|
|
197
|
+
perturbation_task_class = create_perturbation_class(self._task, perturbation_config)
|
|
192
198
|
return perturbation_task_class.with_overwrite(
|
|
193
199
|
num_fewshot=num_fewshot,
|
|
194
200
|
custom_subjects=custom_subjects,
|
|
@@ -198,15 +204,22 @@ class _Eager(EvalFactory):
|
|
|
198
204
|
|
|
199
205
|
def response_type(self) -> ResponseType:
|
|
200
206
|
"""The eval's response type"""
|
|
201
|
-
return self.
|
|
207
|
+
return self._task.get_response_type()
|
|
202
208
|
|
|
203
209
|
def metrics(self) -> list[type["BaseMetric"]]:
|
|
204
210
|
"""The eval's metrics"""
|
|
205
|
-
return self.
|
|
211
|
+
return self._task.get_metrics()
|
|
206
212
|
|
|
207
213
|
def display_name(self) -> str:
|
|
208
214
|
"""The eval's human-readable display name (the task's ``NAME``)."""
|
|
209
|
-
return self.
|
|
215
|
+
return self._task.NAME
|
|
216
|
+
|
|
217
|
+
def markdown_doc(self, formatters: Sequence[BaseFormatter]) -> str:
|
|
218
|
+
try:
|
|
219
|
+
task = self.create(num_fewshot=1, custom_subjects=None, custom_hf_revision=None)
|
|
220
|
+
except (TypeError, ValueError, AssertionError):
|
|
221
|
+
task = self.create(num_fewshot=0, custom_subjects=None, custom_hf_revision=None)
|
|
222
|
+
return task.markdown_doc(formatters)
|
|
210
223
|
|
|
211
224
|
|
|
212
225
|
class Registry:
|
|
@@ -317,14 +330,6 @@ def validate_task_name(name: str) -> str:
|
|
|
317
330
|
return name
|
|
318
331
|
|
|
319
332
|
|
|
320
|
-
def get_task(name: str, /) -> type[BaseTask]:
|
|
321
|
-
"""Return a registered task for a given name.
|
|
322
|
-
|
|
323
|
-
Note: This method will import any lazily registered task.
|
|
324
|
-
"""
|
|
325
|
-
return _REGISTRY[name].task_class()
|
|
326
|
-
|
|
327
|
-
|
|
328
333
|
def register_task(task: type[BaseTask]) -> str:
|
|
329
334
|
"""The class name is used as the task name."""
|
|
330
335
|
return registry().register(task)
|
|
@@ -24,6 +24,7 @@ def register_all_tasks() -> None:
|
|
|
24
24
|
register_lazy_task("eval_framework.tasks.benchmarks.goldenswag.GOLDENSWAG")
|
|
25
25
|
register_lazy_task("eval_framework.tasks.benchmarks.goldenswag.GOLDENSWAG_IDK")
|
|
26
26
|
register_lazy_task("eval_framework.tasks.benchmarks.gpqa.GPQA_OLMES")
|
|
27
|
+
register_lazy_task("eval_framework.tasks.benchmarks.gpqa.GPQA_DIAMOND_COT")
|
|
27
28
|
register_lazy_task("eval_framework.tasks.benchmarks.gsm8k.GSM8K_OLMES")
|
|
28
29
|
register_lazy_task("eval_framework.tasks.benchmarks.gsm8k.GSM8KBPB")
|
|
29
30
|
register_lazy_task("eval_framework.tasks.benchmarks.math_reasoning.MATHMinervaBPB")
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/external/drop_process_results.py
RENAMED
|
File without changes
|
{eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/external/ifeval_impl/README.md
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/external/ifeval_impl/utils.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/aggregators/__init__.py
RENAMED
|
File without changes
|
{eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/aggregators/aggregators.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/csv_format.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/ifeval.py
RENAMED
|
File without changes
|
{eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/json_format.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/repetition.py
RENAMED
|
File without changes
|
{eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/rouge_1.py
RENAMED
|
File without changes
|
{eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/rouge_2.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/rouge_l.py
RENAMED
|
File without changes
|
{eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/text_counter.py
RENAMED
|
File without changes
|
{eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/efficiency/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/graders/language.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/graders/models.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/llm_judge_coherence.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/llm_judge_refusal.py
RENAMED
|
File without changes
|
{eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/llm_judge_sql.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/loglikelihood/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/loglikelihood/base.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/loglikelihood/dcs.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/metrics/loglikelihood/ternary.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/result_processors/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/result_processors/hf_uploader.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/Dockerfile_codebench
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/arc_de.py
RENAMED
|
File without changes
|
{eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/bigcodebench.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/global_mmlu.py
RENAMED
|
File without changes
|
{eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/goldenswag.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/hellaswag.py
RENAMED
|
File without changes
|
{eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/humaneval.py
RENAMED
|
File without changes
|
{eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/ifeval.py
RENAMED
|
File without changes
|
{eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/math_reasoning.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/mmlu_pro.py
RENAMED
|
File without changes
|
{eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/multipl_e.py
RENAMED
|
File without changes
|
{eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/naturalqs_open.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/social_iqa.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/winogrande.py
RENAMED
|
File without changes
|
{eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/dataset_revisions.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.11 → eval_framework-0.9.0}/src/eval_framework/tasks/hf-dataset-revisions.json
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|