eval-framework 0.8.7__tar.gz → 0.9.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {eval_framework-0.8.7 → eval_framework-0.9.0}/PKG-INFO +5 -5
- {eval_framework-0.8.7 → eval_framework-0.9.0}/pyproject.toml +6 -6
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/response_generator.py +1 -1
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/suite.py +2 -2
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/tasks/base.py +35 -2
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/gpqa.py +5 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/tasks/hf-dataset-revisions.json +0 -1
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/tasks/markdown_doc.py +1 -1
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/tasks/registry.py +50 -50
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/tasks/task_loader.py +7 -7
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/tasks/task_names.py +1 -8
- eval_framework-0.8.7/src/eval_framework/tasks/benchmarks/triviaqa.py +0 -78
- {eval_framework-0.8.7 → eval_framework-0.9.0}/LICENSE +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/README.md +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/__init__.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/base_config.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/context/__init__.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/context/determined.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/context/eval.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/context/local.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/evaluation_generator.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/exceptions.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/external/drop_process_results.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/external/ifeval_impl/README.md +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/external/ifeval_impl/instructions.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/external/ifeval_impl/instructions_registry.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/external/ifeval_impl/instructions_util.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/external/ifeval_impl/utils.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/llm/__init__.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/llm/aleph_alpha.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/llm/base.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/llm/huggingface.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/llm/models.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/llm/openai.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/logger.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/main.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/__init__.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/aggregators/__init__.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/aggregators/aggregators.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/base.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/__init__.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/accuracy_completion.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/code_assertion.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/code_execution_pass_at_one.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/concordance_index.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/csv_format.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/drop_completion.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/exponential_similarity.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/f1.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/format_checker.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/grid_difference.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/ifeval.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/json_format.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/language_checker.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/length_control.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/math_minerva_completion.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/math_reasoning_completion.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/minerva_math_utils.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/multipl_e_assertion.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/placeholder_checker.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/repetition.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/rouge_1.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/rouge_2.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/rouge_geometric_mean.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/rouge_l.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/text_counter.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/efficiency/__init__.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/efficiency/bytes_per_sequence_position.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/__init__.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/base.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/graders/chatbot_style_grader.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/graders/coherence_grader.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/graders/comparison_grader.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/graders/conciseness_grader.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/graders/contains_names_grader.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/graders/format_correctness_grader.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/graders/instruction_grader.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/graders/language.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/graders/long_context_grader.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/graders/models.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/graders/refusal_grader.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/graders/sql_quality_grader.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/graders/summary_world_knowledge_grader.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/llm_judge_chatbot_style.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/llm_judge_coherence.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/llm_judge_completion_accuracy.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/llm_judge_conciseness.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/llm_judge_contains_names.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/llm_judge_format_correctness.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/llm_judge_instruction.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/llm_judge_mtbench_pair.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/llm_judge_mtbench_single.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/llm_judge_refusal.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/llm_judge_sql.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/llm_judge_world_knowledge.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/utils.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/loglikelihood/__init__.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/loglikelihood/accuracy_loglikelihood.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/loglikelihood/base.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/loglikelihood/bits_per_byte.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/loglikelihood/confidence_weighted_accuracy.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/loglikelihood/dcs.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/loglikelihood/probability_mass.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/loglikelihood/ternary.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/py.typed +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/result_processors/__init__.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/result_processors/base.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/result_processors/hf_uploader.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/result_processors/result_processor.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/result_processors/wandb_uploader.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/run.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/run_direct.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/shared/types.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/tasks/Dockerfile_codebench +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/tasks/__init__.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/__init__.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/arc.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/arc_de.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/bigcodebench.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/copa.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/csqa.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/drop.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/global_mmlu.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/goldenswag.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/gsm8k.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/hellaswag.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/humaneval.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/ifeval.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/math_reasoning.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/mbpp.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/medqa.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/mmlu.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/mmlu_pro.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/multipl_e.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/naturalqs_open.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/piqa.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/sciq.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/social_iqa.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/squad.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/winogrande.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/tasks/dataset_revisions.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/tasks/eval_config.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/tasks/frozen-hf-dataset-revisions.json +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/tasks/perturbation.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/tasks/task_style.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/tasks/utils.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/utils/constants.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/utils/file_ops.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/utils/helpers.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/utils/logging.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/utils/packaging.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/utils/tqdm_handler.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/template_formatting/README.md +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/template_formatting/__init__.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/template_formatting/formatter.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/template_formatting/mistral_formatter.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.9.0}/src/template_formatting/py.typed +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.3
|
|
2
2
|
Name: eval-framework
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.9.0
|
|
4
4
|
Summary: Evaluation Framework
|
|
5
5
|
Author: Aleph Alpha Research
|
|
6
6
|
License: Apache License
|
|
@@ -216,17 +216,17 @@ Requires-Dist: xmltodict>=1.0.4,<1.1
|
|
|
216
216
|
Requires-Dist: pydantic>=2.13.4,<3
|
|
217
217
|
Requires-Dist: datasets>=5.0.1,<6
|
|
218
218
|
Requires-Dist: pycountry>=26.2.16,<27
|
|
219
|
-
Requires-Dist: nltk>=3.10.
|
|
219
|
+
Requires-Dist: nltk>=3.10.1,<4
|
|
220
220
|
Requires-Dist: python-dotenv>=1.2.2,<2
|
|
221
221
|
Requires-Dist: lingua-language-detector>=2.2.0,<3
|
|
222
222
|
Requires-Dist: google-crc32c>=1.8.0,<2
|
|
223
223
|
Requires-Dist: langdetect>=1.0.9,<2
|
|
224
224
|
Requires-Dist: spacy>=3.8.14,<4
|
|
225
225
|
Requires-Dist: jsonschema>=4.26.0,<5
|
|
226
|
-
Requires-Dist: mysql-connector-python>=
|
|
226
|
+
Requires-Dist: mysql-connector-python>=26.7.0,<27
|
|
227
227
|
Requires-Dist: psycopg2-binary>=2.9.12,<3
|
|
228
228
|
Requires-Dist: sympy>=1.14.0,<2
|
|
229
|
-
Requires-Dist: llm-sandbox[docker]==0.3.
|
|
229
|
+
Requires-Dist: llm-sandbox[docker]==0.3.44
|
|
230
230
|
Requires-Dist: jsonlines>=4,<5
|
|
231
231
|
Requires-Dist: lxml>=6.1.1,<7
|
|
232
232
|
Requires-Dist: python-iso639>=2026.7.23
|
|
@@ -240,7 +240,7 @@ Requires-Dist: eval-framework[determined,api,openai,transformers,accelerate,opti
|
|
|
240
240
|
Requires-Dist: aleph-alpha-client>=11.5.1 ; extra == 'api'
|
|
241
241
|
Requires-Dist: determined>=0.38.1,<0.39 ; extra == 'determined'
|
|
242
242
|
Requires-Dist: tensorboard==2.21.0 ; extra == 'determined'
|
|
243
|
-
Requires-Dist: openai>=2.
|
|
243
|
+
Requires-Dist: openai>=2.53.0,<3 ; extra == 'openai'
|
|
244
244
|
Requires-Dist: tiktoken>=0.13.0,<1 ; extra == 'openai'
|
|
245
245
|
Requires-Dist: transformers>=4.45.2,<5 ; extra == 'openai'
|
|
246
246
|
Requires-Dist: transformers>=4.45.2,<5 ; extra == 'optional'
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "eval-framework"
|
|
3
|
-
version = "0.
|
|
3
|
+
version = "0.9.0"
|
|
4
4
|
description = "Evaluation Framework"
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
license = { file = "LICENSE" }
|
|
@@ -23,17 +23,17 @@ dependencies = [
|
|
|
23
23
|
"pydantic>=2.13.4,<3",
|
|
24
24
|
"datasets>=5.0.1,<6",
|
|
25
25
|
"pycountry>=26.2.16,<27",
|
|
26
|
-
"nltk>=3.10.
|
|
26
|
+
"nltk>=3.10.1,<4",
|
|
27
27
|
"python-dotenv>=1.2.2,<2",
|
|
28
28
|
"lingua-language-detector>=2.2.0,<3",
|
|
29
29
|
"google-crc32c>=1.8.0,<2",
|
|
30
30
|
"langdetect>=1.0.9,<2", # required by the original ifeval implementation
|
|
31
31
|
"spacy>=3.8.14,<4",
|
|
32
32
|
"jsonschema>=4.26.0,<5",
|
|
33
|
-
"mysql-connector-python>=
|
|
33
|
+
"mysql-connector-python>=26.7.0,<27", # required for sql-related tasks
|
|
34
34
|
"psycopg2-binary>=2.9.12,<3", # required for sql-related tasks
|
|
35
35
|
"sympy>=1.14.0,<2",
|
|
36
|
-
"llm-sandbox[docker]==0.3.
|
|
36
|
+
"llm-sandbox[docker]==0.3.44",
|
|
37
37
|
"jsonlines>=4,<5",
|
|
38
38
|
"lxml>=6.1.1,<7",
|
|
39
39
|
"python-iso639>=2026.7.23",
|
|
@@ -54,7 +54,7 @@ determined = [
|
|
|
54
54
|
]
|
|
55
55
|
api = ["aleph-alpha-client>=11.5.1"]
|
|
56
56
|
openai = [
|
|
57
|
-
"openai>=2.
|
|
57
|
+
"openai>=2.53.0,<3",
|
|
58
58
|
"tiktoken>=0.13.0,<1",
|
|
59
59
|
"transformers>=4.45.2,<5",
|
|
60
60
|
]
|
|
@@ -92,7 +92,7 @@ dev = [
|
|
|
92
92
|
"types-python-dateutil>=2.9.0.20260716,<3",
|
|
93
93
|
"types-requests>=2.33.0.20260712,<3",
|
|
94
94
|
"plotly>=6.9.0,<7",
|
|
95
|
-
"ruff>=0.16.
|
|
95
|
+
"ruff>=0.16.1",
|
|
96
96
|
"pip-licenses>=5.5.5",
|
|
97
97
|
]
|
|
98
98
|
flash-attn = [
|
|
@@ -179,7 +179,7 @@ class ResponseGenerator:
|
|
|
179
179
|
:param should_preempt_callable: function to check if preempt is called
|
|
180
180
|
:return: list of responses, preempted
|
|
181
181
|
"""
|
|
182
|
-
logger.info(f"{RED}[ Running task {self.task.
|
|
182
|
+
logger.info(f"{RED}[ Running task {self.task.display_name()} against model ------------ ]{RESET}")
|
|
183
183
|
self.start_time, monotonic_start = time.time(), time.monotonic()
|
|
184
184
|
run_fn = self._generative_output_type_selector()
|
|
185
185
|
self._verify_loaded_metadata_compatibility()
|
|
@@ -18,7 +18,7 @@ from eval_framework.context.local import _load_model
|
|
|
18
18
|
from eval_framework.result_processors.result_processor import generate_output_dir
|
|
19
19
|
from eval_framework.run import _run_single_task
|
|
20
20
|
from eval_framework.tasks.eval_config import EvalConfig
|
|
21
|
-
from eval_framework.tasks.registry import
|
|
21
|
+
from eval_framework.tasks.registry import registry
|
|
22
22
|
|
|
23
23
|
logger = logging.getLogger(__name__)
|
|
24
24
|
|
|
@@ -108,7 +108,7 @@ class TaskSuite(BaseModel):
|
|
|
108
108
|
if isinstance(self.tasks, str):
|
|
109
109
|
if self.name is None:
|
|
110
110
|
self.name = self.tasks
|
|
111
|
-
if not
|
|
111
|
+
if self.tasks not in registry():
|
|
112
112
|
raise ValueError(f"Task '{self.tasks}' is not registered.")
|
|
113
113
|
elif not self.tasks:
|
|
114
114
|
raise ValueError(f"TaskSuite '{self.name}': 'tasks' must not be empty.")
|
|
@@ -2,7 +2,7 @@ import logging
|
|
|
2
2
|
import os
|
|
3
3
|
import random
|
|
4
4
|
import traceback
|
|
5
|
-
from abc import ABC
|
|
5
|
+
from abc import ABC, abstractmethod
|
|
6
6
|
from collections.abc import Iterable, Sequence
|
|
7
7
|
from enum import Enum
|
|
8
8
|
from pathlib import Path
|
|
@@ -86,7 +86,37 @@ SubjectType = TypeVar("SubjectType")
|
|
|
86
86
|
logger = logging.getLogger(__name__)
|
|
87
87
|
|
|
88
88
|
|
|
89
|
-
class
|
|
89
|
+
class Task(ABC):
|
|
90
|
+
"""The contract a caller relies on to run an evaluation"""
|
|
91
|
+
|
|
92
|
+
@abstractmethod
|
|
93
|
+
def iterate_samples(self, num_samples: int | None = None) -> Iterable[Sample]: ...
|
|
94
|
+
|
|
95
|
+
@abstractmethod
|
|
96
|
+
def generate_completions(
|
|
97
|
+
self,
|
|
98
|
+
llm: "BaseLLM",
|
|
99
|
+
samples: list[Sample],
|
|
100
|
+
stop_sequences: list[str] | None = None,
|
|
101
|
+
max_tokens: int | None = None,
|
|
102
|
+
fail_on_error: bool = True,
|
|
103
|
+
) -> list[Completion]:
|
|
104
|
+
"""Run ``llm`` over ``samples`` and return their completions."""
|
|
105
|
+
|
|
106
|
+
@abstractmethod
|
|
107
|
+
def get_metadata(self) -> dict[str, str | list[str]]:
|
|
108
|
+
"""Descriptive metadata about the eval for result reporting."""
|
|
109
|
+
|
|
110
|
+
@abstractmethod
|
|
111
|
+
def get_response_type(self) -> ResponseType: ...
|
|
112
|
+
|
|
113
|
+
@abstractmethod
|
|
114
|
+
def display_name(self) -> str:
|
|
115
|
+
"""Human-readable display name. Is allowed to have special characters and whitespaces."""
|
|
116
|
+
...
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
class BaseTask[SubjectType](Task):
|
|
90
120
|
NAME: str
|
|
91
121
|
DATASET_PATH: str
|
|
92
122
|
SAMPLE_SPLIT: str
|
|
@@ -497,3 +527,6 @@ class BaseTask[SubjectType](ABC):
|
|
|
497
527
|
def METRICS(cls) -> list[type["BaseMetric"]]:
|
|
498
528
|
"""For backwards compatibility."""
|
|
499
529
|
return cls.get_metrics()
|
|
530
|
+
|
|
531
|
+
def display_name(self) -> str:
|
|
532
|
+
return self.NAME
|
{eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/tasks/hf-dataset-revisions.json
RENAMED
|
@@ -23,7 +23,6 @@
|
|
|
23
23
|
"google-research-datasets/nq_open": "5dd9790a83002ad084ddeb7c420dc716852c6f28",
|
|
24
24
|
"google/IFEval": "966cd89545d6b6acfd7638bc708b98261ca58e84",
|
|
25
25
|
"jzhang86/de_ifeval": "4f52d847003b3c83cc282e9d296853a24b34b19a",
|
|
26
|
-
"mandarjoshi/trivia_qa": "0f7faf33a3908546c6fd5b73a660e0f8ff173c2f",
|
|
27
26
|
"math-ai/aime25": "563bb8404243c5f09de6ec262f2db674fe5bce9b",
|
|
28
27
|
"math-ai/aime26": "79037aebdb6580008fb960d17cb21fd3099083e3",
|
|
29
28
|
"nuprl/MultiPL-E": "28441b6024e71d4a1c1c0f6bf171c935cd5a43f2",
|
|
@@ -47,7 +47,7 @@ def markdown_doc(
|
|
|
47
47
|
buf.write(f"- Module: `{module}`\n\n")
|
|
48
48
|
|
|
49
49
|
if http_path:
|
|
50
|
-
buf.write(f"- Link to dataset: [{http_path}]({http_path})\n
|
|
50
|
+
buf.write(f"- Link to dataset: [{http_path}]({http_path})\n")
|
|
51
51
|
else:
|
|
52
52
|
assert example_messages is not None, "a task without a dataset link must supply an example sample"
|
|
53
53
|
for split, size in (split_sizes or {}).items():
|
|
@@ -18,7 +18,6 @@ __all__ = [
|
|
|
18
18
|
"EvalFactory",
|
|
19
19
|
"Registry",
|
|
20
20
|
"with_registry",
|
|
21
|
-
"get_task",
|
|
22
21
|
"is_registered",
|
|
23
22
|
"validate_task_name",
|
|
24
23
|
"registered_task_names",
|
|
@@ -36,8 +35,8 @@ class EvalFactory(ABC):
|
|
|
36
35
|
"""
|
|
37
36
|
|
|
38
37
|
@abstractmethod
|
|
39
|
-
def
|
|
40
|
-
"
|
|
38
|
+
def id(self) -> str:
|
|
39
|
+
"Canonical key used to register this benchmark"
|
|
41
40
|
|
|
42
41
|
@property
|
|
43
42
|
@abstractmethod
|
|
@@ -75,13 +74,10 @@ class EvalFactory(ABC):
|
|
|
75
74
|
user_prompt_suffix: str | None = None,
|
|
76
75
|
) -> BaseTask: ...
|
|
77
76
|
|
|
77
|
+
@abstractmethod
|
|
78
78
|
def markdown_doc(self, formatters: Sequence[BaseFormatter]) -> str:
|
|
79
79
|
"""Render the eval's documentation as markdown."""
|
|
80
|
-
|
|
81
|
-
task = self.create(num_fewshot=1, custom_subjects=None, custom_hf_revision=None)
|
|
82
|
-
except (TypeError, ValueError, AssertionError):
|
|
83
|
-
task = self.create(num_fewshot=0, custom_subjects=None, custom_hf_revision=None)
|
|
84
|
-
return task.markdown_doc(formatters)
|
|
80
|
+
...
|
|
85
81
|
|
|
86
82
|
|
|
87
83
|
class _Lazy(EvalFactory):
|
|
@@ -104,6 +100,9 @@ class _Lazy(EvalFactory):
|
|
|
104
100
|
def source_module(self) -> str:
|
|
105
101
|
return self._module
|
|
106
102
|
|
|
103
|
+
def id(self) -> str:
|
|
104
|
+
return self._class_name
|
|
105
|
+
|
|
107
106
|
def task_class(self) -> type[BaseTask]:
|
|
108
107
|
if self._loaded is None:
|
|
109
108
|
module = importlib.import_module(self._module)
|
|
@@ -152,6 +151,13 @@ class _Lazy(EvalFactory):
|
|
|
152
151
|
"""The eval's human-readable display name (the task's ``NAME``)."""
|
|
153
152
|
return self.task_class().NAME
|
|
154
153
|
|
|
154
|
+
def markdown_doc(self, formatters: Sequence[BaseFormatter]) -> str:
|
|
155
|
+
try:
|
|
156
|
+
task = self.create(num_fewshot=1, custom_subjects=None, custom_hf_revision=None)
|
|
157
|
+
except (TypeError, ValueError, AssertionError):
|
|
158
|
+
task = self.create(num_fewshot=0, custom_subjects=None, custom_hf_revision=None)
|
|
159
|
+
return task.markdown_doc(formatters)
|
|
160
|
+
|
|
155
161
|
|
|
156
162
|
class _Eager(EvalFactory):
|
|
157
163
|
"""Wraps an already-imported task class."""
|
|
@@ -163,8 +169,8 @@ class _Eager(EvalFactory):
|
|
|
163
169
|
def source_module(self) -> str:
|
|
164
170
|
return self._task.__module__
|
|
165
171
|
|
|
166
|
-
def
|
|
167
|
-
return self._task
|
|
172
|
+
def id(self) -> str:
|
|
173
|
+
return self._task.__name__
|
|
168
174
|
|
|
169
175
|
def create(
|
|
170
176
|
self,
|
|
@@ -173,7 +179,7 @@ class _Eager(EvalFactory):
|
|
|
173
179
|
custom_hf_revision: str | None,
|
|
174
180
|
user_prompt_suffix: str | None = None,
|
|
175
181
|
) -> BaseTask:
|
|
176
|
-
return self.
|
|
182
|
+
return self._task.with_overwrite(
|
|
177
183
|
num_fewshot=num_fewshot,
|
|
178
184
|
custom_subjects=custom_subjects,
|
|
179
185
|
custom_hf_revision=custom_hf_revision,
|
|
@@ -188,7 +194,7 @@ class _Eager(EvalFactory):
|
|
|
188
194
|
custom_hf_revision: str | None,
|
|
189
195
|
user_prompt_suffix: str | None = None,
|
|
190
196
|
) -> BaseTask:
|
|
191
|
-
perturbation_task_class = create_perturbation_class(self.
|
|
197
|
+
perturbation_task_class = create_perturbation_class(self._task, perturbation_config)
|
|
192
198
|
return perturbation_task_class.with_overwrite(
|
|
193
199
|
num_fewshot=num_fewshot,
|
|
194
200
|
custom_subjects=custom_subjects,
|
|
@@ -198,15 +204,22 @@ class _Eager(EvalFactory):
|
|
|
198
204
|
|
|
199
205
|
def response_type(self) -> ResponseType:
|
|
200
206
|
"""The eval's response type"""
|
|
201
|
-
return self.
|
|
207
|
+
return self._task.get_response_type()
|
|
202
208
|
|
|
203
209
|
def metrics(self) -> list[type["BaseMetric"]]:
|
|
204
210
|
"""The eval's metrics"""
|
|
205
|
-
return self.
|
|
211
|
+
return self._task.get_metrics()
|
|
206
212
|
|
|
207
213
|
def display_name(self) -> str:
|
|
208
214
|
"""The eval's human-readable display name (the task's ``NAME``)."""
|
|
209
|
-
return self.
|
|
215
|
+
return self._task.NAME
|
|
216
|
+
|
|
217
|
+
def markdown_doc(self, formatters: Sequence[BaseFormatter]) -> str:
|
|
218
|
+
try:
|
|
219
|
+
task = self.create(num_fewshot=1, custom_subjects=None, custom_hf_revision=None)
|
|
220
|
+
except (TypeError, ValueError, AssertionError):
|
|
221
|
+
task = self.create(num_fewshot=0, custom_subjects=None, custom_hf_revision=None)
|
|
222
|
+
return task.markdown_doc(formatters)
|
|
210
223
|
|
|
211
224
|
|
|
212
225
|
class Registry:
|
|
@@ -255,10 +268,6 @@ class Registry:
|
|
|
255
268
|
|
|
256
269
|
return factory
|
|
257
270
|
|
|
258
|
-
def add(self, task: type[BaseTask]) -> None:
|
|
259
|
-
task_key = self._task_key(task.NAME)
|
|
260
|
-
self._registry[task_key] = (task.NAME, _Eager(task))
|
|
261
|
-
|
|
262
271
|
def __setitem__(self, name: str, factory: EvalFactory) -> None:
|
|
263
272
|
task_key = self._task_key(name)
|
|
264
273
|
if task_key in self._registry:
|
|
@@ -266,6 +275,24 @@ class Registry:
|
|
|
266
275
|
|
|
267
276
|
self._registry[task_key] = (name, factory)
|
|
268
277
|
|
|
278
|
+
def register(self, task: type[BaseTask]) -> str:
|
|
279
|
+
"""The class name is used as the task name."""
|
|
280
|
+
if not issubclass(task, BaseTask):
|
|
281
|
+
raise ValueError(f"Can only register subclasses of BaseTask, got {task}")
|
|
282
|
+
name = task.__name__
|
|
283
|
+
self[name] = _Eager(task)
|
|
284
|
+
return name
|
|
285
|
+
|
|
286
|
+
def register_lazy(self, class_path: str, /) -> None:
|
|
287
|
+
"""Register a task by its dotted class path, without importing its module."""
|
|
288
|
+
if "." not in class_path:
|
|
289
|
+
raise ValueError(
|
|
290
|
+
f"Invalid class path `{class_path}`. This needs to be a global path like "
|
|
291
|
+
"`eval_framework.tasks.benchmarks.mmlu.MMLU`): "
|
|
292
|
+
)
|
|
293
|
+
base_module, class_name = class_path.rsplit(".", maxsplit=1)
|
|
294
|
+
self[class_name] = _Lazy(class_name=class_name, module=base_module)
|
|
295
|
+
|
|
269
296
|
|
|
270
297
|
_REGISTRY = Registry()
|
|
271
298
|
|
|
@@ -298,43 +325,16 @@ def is_registered(name: str, /) -> bool:
|
|
|
298
325
|
|
|
299
326
|
def validate_task_name(name: str) -> str:
|
|
300
327
|
"""Pydantic-style validator for task names."""
|
|
301
|
-
if not
|
|
328
|
+
if name not in registry():
|
|
302
329
|
raise ValueError(f"Task not registered: {name}")
|
|
303
330
|
return name
|
|
304
331
|
|
|
305
332
|
|
|
306
|
-
def get_task(name: str, /) -> type[BaseTask]:
|
|
307
|
-
"""Return a registered task for a given name.
|
|
308
|
-
|
|
309
|
-
Note: This method will import any lazily registered task.
|
|
310
|
-
"""
|
|
311
|
-
return _REGISTRY[name].task_class()
|
|
312
|
-
|
|
313
|
-
|
|
314
333
|
def register_task(task: type[BaseTask]) -> str:
|
|
315
334
|
"""The class name is used as the task name."""
|
|
316
|
-
|
|
317
|
-
raise ValueError(f"Can only register subclasses of BaseTask, got {task}")
|
|
318
|
-
name = task.__name__
|
|
319
|
-
_REGISTRY[name] = _Eager(task)
|
|
320
|
-
return name
|
|
335
|
+
return registry().register(task)
|
|
321
336
|
|
|
322
337
|
|
|
323
338
|
def register_lazy_task(class_path: str, /) -> None:
|
|
324
|
-
"""Register a task without importing
|
|
325
|
-
|
|
326
|
-
Lazily register a task without importing the module.
|
|
327
|
-
|
|
328
|
-
Args:
|
|
329
|
-
class_path: The full path to the task class. For example,
|
|
330
|
-
`eval_framework.tasks.benchmarks.mmlu.MMLU`.
|
|
331
|
-
extras: Any extra dependencies of `eval_framework` that need to be installed for this task.
|
|
332
|
-
"""
|
|
333
|
-
if "." not in class_path:
|
|
334
|
-
raise ValueError(
|
|
335
|
-
f"Invalid class path `{class_path}`. This needs to be a global path like "
|
|
336
|
-
"`eval_framework.tasks.benchmarks.mmlu.MMLU`): "
|
|
337
|
-
)
|
|
338
|
-
|
|
339
|
-
base_module, class_name = class_path.rsplit(".", maxsplit=1)
|
|
340
|
-
_REGISTRY[class_name] = _Lazy(class_name=class_name, module=base_module)
|
|
339
|
+
"""Register a task by its dotted class path, without importing its module."""
|
|
340
|
+
registry().register_lazy(class_path)
|
|
@@ -8,7 +8,8 @@ from types import ModuleType
|
|
|
8
8
|
from typing import Any
|
|
9
9
|
|
|
10
10
|
from eval_framework.tasks.base import BaseTask
|
|
11
|
-
from eval_framework.tasks.registry import
|
|
11
|
+
from eval_framework.tasks.registry import Registry
|
|
12
|
+
from eval_framework.tasks.registry import registry as global_registry
|
|
12
13
|
|
|
13
14
|
logger = logging.getLogger(__name__)
|
|
14
15
|
|
|
@@ -46,14 +47,13 @@ def import_file(f: str | os.PathLike, /) -> Any:
|
|
|
46
47
|
return user_module
|
|
47
48
|
|
|
48
49
|
|
|
49
|
-
def load_extra_tasks(module_paths: Sequence[str | os.PathLike]) -> None:
|
|
50
|
+
def load_extra_tasks(module_paths: Sequence[str | os.PathLike], registry: Registry | None = None) -> None:
|
|
50
51
|
"""Dynamically load and register user-defined tasks from a list of files or directories.
|
|
51
52
|
|
|
52
|
-
Each .py file found
|
|
53
|
-
in the TaskName enum for use by name.
|
|
54
|
-
Provides clear error messages for missing/invalid files or import errors.
|
|
53
|
+
Each .py file found is imported, and any BaseTask subclass is registered
|
|
55
54
|
"""
|
|
56
55
|
assert not (isinstance(module_paths, str)), "module_paths must be a sequence of strings / os.PathLike objects"
|
|
56
|
+
registry = registry if registry is not None else global_registry()
|
|
57
57
|
for file_path in find_all_python_files(*module_paths):
|
|
58
58
|
user_module = import_file(file_path)
|
|
59
59
|
|
|
@@ -63,7 +63,7 @@ def load_extra_tasks(module_paths: Sequence[str | os.PathLike]) -> None:
|
|
|
63
63
|
if not hasattr(obj, "NAME"):
|
|
64
64
|
logger.info(f"[User Task Loader] Skipping {obj.__module__} - no NAME attribute present.")
|
|
65
65
|
else:
|
|
66
|
-
if
|
|
66
|
+
if obj.NAME in registry:
|
|
67
67
|
# two classes with the same NAME attribute
|
|
68
68
|
logger.info(obj.__module__)
|
|
69
69
|
|
|
@@ -77,5 +77,5 @@ def load_extra_tasks(module_paths: Sequence[str | os.PathLike]) -> None:
|
|
|
77
77
|
else:
|
|
78
78
|
# if there is no duplicate name conflict then register the new task
|
|
79
79
|
class_obj = getattr(user_module, name)
|
|
80
|
-
|
|
80
|
+
registry.register(class_obj)
|
|
81
81
|
logger.info(f"[User Task Loader] Registered task: {class_obj.NAME}")
|
|
@@ -24,6 +24,7 @@ def register_all_tasks() -> None:
|
|
|
24
24
|
register_lazy_task("eval_framework.tasks.benchmarks.goldenswag.GOLDENSWAG")
|
|
25
25
|
register_lazy_task("eval_framework.tasks.benchmarks.goldenswag.GOLDENSWAG_IDK")
|
|
26
26
|
register_lazy_task("eval_framework.tasks.benchmarks.gpqa.GPQA_OLMES")
|
|
27
|
+
register_lazy_task("eval_framework.tasks.benchmarks.gpqa.GPQA_DIAMOND_COT")
|
|
27
28
|
register_lazy_task("eval_framework.tasks.benchmarks.gsm8k.GSM8K_OLMES")
|
|
28
29
|
register_lazy_task("eval_framework.tasks.benchmarks.gsm8k.GSM8KBPB")
|
|
29
30
|
register_lazy_task("eval_framework.tasks.benchmarks.math_reasoning.MATHMinervaBPB")
|
|
@@ -71,8 +72,6 @@ def register_all_tasks() -> None:
|
|
|
71
72
|
register_lazy_task("eval_framework.tasks.benchmarks.squad.SQuAD_OLMES")
|
|
72
73
|
register_lazy_task("eval_framework.tasks.benchmarks.squad.SQuAD2_MA")
|
|
73
74
|
register_lazy_task("eval_framework.tasks.benchmarks.squad.SQuAD2_MA_NO_SYSPROMPT")
|
|
74
|
-
register_lazy_task("eval_framework.tasks.benchmarks.triviaqa.TRIVIAQA")
|
|
75
|
-
register_lazy_task("eval_framework.tasks.benchmarks.triviaqa.TriviaQA_MA")
|
|
76
75
|
register_lazy_task("eval_framework.tasks.benchmarks.winogrande.WINOGRANDECloze")
|
|
77
76
|
register_lazy_task("eval_framework.tasks.benchmarks.csqa.CommonsenseQAMC_OLMES")
|
|
78
77
|
register_lazy_task("eval_framework.tasks.benchmarks.drop.DropCompletion_OLMES")
|
|
@@ -81,9 +80,3 @@ def register_all_tasks() -> None:
|
|
|
81
80
|
register_lazy_task("eval_framework.tasks.benchmarks.naturalqs_open.NaturalQsOpenMC_OLMES")
|
|
82
81
|
register_lazy_task("eval_framework.tasks.benchmarks.social_iqa.SocialIQAMC_OLMES")
|
|
83
82
|
register_lazy_task("eval_framework.tasks.benchmarks.medqa.MedQAMC_OLMES")
|
|
84
|
-
try:
|
|
85
|
-
# Importing the companion registers the additional tasks from the module.
|
|
86
|
-
# This is mostly for convenience for internal use-cases
|
|
87
|
-
import eval_framework_companion # noqa
|
|
88
|
-
except ImportError:
|
|
89
|
-
pass
|
|
@@ -1,78 +0,0 @@
|
|
|
1
|
-
import random
|
|
2
|
-
from typing import Any
|
|
3
|
-
|
|
4
|
-
from eval_framework.metrics.completion.accuracy_completion import AccuracyCompletion
|
|
5
|
-
from eval_framework.metrics.completion.f1 import F1, F1SquadNormalized
|
|
6
|
-
from eval_framework.tasks.base import BaseTask, Language, ResponseType, Sample
|
|
7
|
-
from eval_framework.tasks.dataset_revisions import HF_REVISIONS_LOCKFILE
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
class TRIVIAQA(BaseTask[str]):
|
|
11
|
-
"""Trivia QA dataset: https://huggingface.co/datasets/mandarjoshi/trivia_qa"""
|
|
12
|
-
|
|
13
|
-
REVISION_LOCKFILE = HF_REVISIONS_LOCKFILE
|
|
14
|
-
|
|
15
|
-
NAME = "TriviaQA"
|
|
16
|
-
DATASET_PATH = "mandarjoshi/trivia_qa"
|
|
17
|
-
SAMPLE_SPLIT = "validation"
|
|
18
|
-
FEWSHOT_SPLIT = "train"
|
|
19
|
-
RESPONSE_TYPE = ResponseType.COMPLETION
|
|
20
|
-
METRICS = [AccuracyCompletion, F1]
|
|
21
|
-
SUBJECTS = ["rc.wikipedia.nocontext"]
|
|
22
|
-
PERTURBATION_UNMODIFIABLE_WORDS = ["Question", "Answer"]
|
|
23
|
-
LANGUAGE = Language.ENG
|
|
24
|
-
|
|
25
|
-
def __init__(self, num_fewshot: int = 0) -> None:
|
|
26
|
-
super().__init__(num_fewshot)
|
|
27
|
-
self.stop_sequences = ["\n"]
|
|
28
|
-
self.max_tokens = 400 # the max length of the ground truth is 282 characters while the average is ~16
|
|
29
|
-
self.rnd_choice_shuffle = random.Random()
|
|
30
|
-
|
|
31
|
-
def _get_instruction_text(self, item: dict[str, Any]) -> str:
|
|
32
|
-
prompt = f"Question: {item['question'].strip()}\nAnswer:"
|
|
33
|
-
return prompt
|
|
34
|
-
|
|
35
|
-
def _get_fewshot_target_text(self, item: dict[str, Any]) -> str:
|
|
36
|
-
target = self._get_ground_truth(item)[0]
|
|
37
|
-
assert target is not None
|
|
38
|
-
assert isinstance(target, str)
|
|
39
|
-
return f" {target}"
|
|
40
|
-
|
|
41
|
-
def _get_ground_truth(self, item: dict[str, Any]) -> list[str]:
|
|
42
|
-
return item["answer"]["aliases"]
|
|
43
|
-
|
|
44
|
-
def post_process_generated_completion(self, completion_text: str, sample: Sample | None = None) -> str:
|
|
45
|
-
return completion_text.strip().rstrip(".")
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
class TriviaQA_MA(TRIVIAQA):
|
|
49
|
-
"""TriviaQA with the exact system prompt used in MA training"""
|
|
50
|
-
|
|
51
|
-
REVISION_LOCKFILE = HF_REVISIONS_LOCKFILE
|
|
52
|
-
|
|
53
|
-
NAME = "TriviaQA_MA"
|
|
54
|
-
SUBJECTS = ["rc.wikipedia"]
|
|
55
|
-
UNANSWERABLE_STR = "unanswerable"
|
|
56
|
-
|
|
57
|
-
METRICS = [AccuracyCompletion, F1, F1SquadNormalized]
|
|
58
|
-
PERTURBATION_UNMODIFIABLE_WORDS = ["Question", "Answer", "Context", "unanswerable"]
|
|
59
|
-
|
|
60
|
-
def __init__(self, num_fewshot: int = 0) -> None:
|
|
61
|
-
super().__init__(num_fewshot)
|
|
62
|
-
self.stop_sequences = []
|
|
63
|
-
self.max_tokens = 27_000
|
|
64
|
-
|
|
65
|
-
def _get_context_text(self, item: dict[str, Any]) -> str:
|
|
66
|
-
return "\n\n".join(item["entity_pages"]["wiki_context"])
|
|
67
|
-
|
|
68
|
-
def _get_system_prompt_text(self, item: dict[str, Any]) -> str | None:
|
|
69
|
-
return (
|
|
70
|
-
"You are a helpful assistant and will answer the user's questions carefully, "
|
|
71
|
-
"logically, accurately and well-reasoned.\n"
|
|
72
|
-
"Use the given context to answer the question faithfully. Answer only if the "
|
|
73
|
-
f"answer is present in the given context, otherwise respond with '{self.UNANSWERABLE_STR}' "
|
|
74
|
-
"if the answer is not present in the context."
|
|
75
|
-
)
|
|
76
|
-
|
|
77
|
-
def _get_instruction_text(self, item: dict[str, Any]) -> str:
|
|
78
|
-
return f"Context:\n{self._get_context_text(item)}\n\nQuestion:\n{item['question'].strip()}\n"
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/external/drop_process_results.py
RENAMED
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/external/ifeval_impl/README.md
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/external/ifeval_impl/utils.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/aggregators/__init__.py
RENAMED
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/aggregators/aggregators.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/csv_format.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/ifeval.py
RENAMED
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/json_format.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/repetition.py
RENAMED
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/rouge_1.py
RENAMED
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/rouge_2.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/rouge_l.py
RENAMED
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/completion/text_counter.py
RENAMED
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/efficiency/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/graders/language.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/graders/models.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/llm_judge_coherence.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/llm_judge_refusal.py
RENAMED
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/llm/llm_judge_sql.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/loglikelihood/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/loglikelihood/base.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/loglikelihood/dcs.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/metrics/loglikelihood/ternary.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/result_processors/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/result_processors/hf_uploader.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/result_processors/wandb_uploader.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/bigcodebench.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/global_mmlu.py
RENAMED
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/goldenswag.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/hellaswag.py
RENAMED
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/humaneval.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/math_reasoning.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/mmlu_pro.py
RENAMED
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/multipl_e.py
RENAMED
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/naturalqs_open.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/social_iqa.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.9.0}/src/eval_framework/tasks/benchmarks/winogrande.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|