eval-framework 0.8.7__tar.gz → 0.8.11__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {eval_framework-0.8.7 → eval_framework-0.8.11}/PKG-INFO +4 -4
- {eval_framework-0.8.7 → eval_framework-0.8.11}/pyproject.toml +5 -5
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/suite.py +2 -2
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/hf-dataset-revisions.json +0 -1
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/registry.py +22 -27
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/task_loader.py +7 -7
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/task_names.py +0 -8
- eval_framework-0.8.7/src/eval_framework/tasks/benchmarks/triviaqa.py +0 -78
- {eval_framework-0.8.7 → eval_framework-0.8.11}/LICENSE +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/README.md +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/__init__.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/base_config.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/context/__init__.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/context/determined.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/context/eval.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/context/local.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/evaluation_generator.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/exceptions.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/external/drop_process_results.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/external/ifeval_impl/README.md +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/external/ifeval_impl/instructions.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/external/ifeval_impl/instructions_registry.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/external/ifeval_impl/instructions_util.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/external/ifeval_impl/utils.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/llm/__init__.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/llm/aleph_alpha.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/llm/base.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/llm/huggingface.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/llm/models.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/llm/openai.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/logger.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/main.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/__init__.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/aggregators/__init__.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/aggregators/aggregators.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/base.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/completion/__init__.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/completion/accuracy_completion.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/completion/code_assertion.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/completion/code_execution_pass_at_one.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/completion/concordance_index.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/completion/csv_format.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/completion/drop_completion.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/completion/exponential_similarity.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/completion/f1.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/completion/format_checker.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/completion/grid_difference.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/completion/ifeval.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/completion/json_format.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/completion/language_checker.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/completion/length_control.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/completion/math_minerva_completion.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/completion/math_reasoning_completion.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/completion/minerva_math_utils.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/completion/multipl_e_assertion.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/completion/placeholder_checker.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/completion/repetition.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/completion/rouge_1.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/completion/rouge_2.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/completion/rouge_geometric_mean.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/completion/rouge_l.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/completion/text_counter.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/efficiency/__init__.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/efficiency/bytes_per_sequence_position.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/llm/__init__.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/llm/base.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/llm/graders/chatbot_style_grader.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/llm/graders/coherence_grader.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/llm/graders/comparison_grader.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/llm/graders/conciseness_grader.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/llm/graders/contains_names_grader.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/llm/graders/format_correctness_grader.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/llm/graders/instruction_grader.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/llm/graders/language.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/llm/graders/long_context_grader.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/llm/graders/models.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/llm/graders/refusal_grader.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/llm/graders/sql_quality_grader.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/llm/graders/summary_world_knowledge_grader.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/llm/llm_judge_chatbot_style.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/llm/llm_judge_coherence.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/llm/llm_judge_completion_accuracy.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/llm/llm_judge_conciseness.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/llm/llm_judge_contains_names.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/llm/llm_judge_format_correctness.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/llm/llm_judge_instruction.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/llm/llm_judge_mtbench_pair.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/llm/llm_judge_mtbench_single.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/llm/llm_judge_refusal.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/llm/llm_judge_sql.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/llm/llm_judge_world_knowledge.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/llm/utils.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/loglikelihood/__init__.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/loglikelihood/accuracy_loglikelihood.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/loglikelihood/base.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/loglikelihood/bits_per_byte.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/loglikelihood/confidence_weighted_accuracy.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/loglikelihood/dcs.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/loglikelihood/probability_mass.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/loglikelihood/ternary.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/py.typed +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/response_generator.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/result_processors/__init__.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/result_processors/base.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/result_processors/hf_uploader.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/result_processors/result_processor.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/result_processors/wandb_uploader.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/run.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/run_direct.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/shared/types.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/Dockerfile_codebench +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/__init__.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/base.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/benchmarks/__init__.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/benchmarks/arc.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/benchmarks/arc_de.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/benchmarks/bigcodebench.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/benchmarks/copa.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/benchmarks/csqa.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/benchmarks/drop.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/benchmarks/global_mmlu.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/benchmarks/goldenswag.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/benchmarks/gpqa.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/benchmarks/gsm8k.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/benchmarks/hellaswag.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/benchmarks/humaneval.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/benchmarks/ifeval.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/benchmarks/math_reasoning.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/benchmarks/mbpp.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/benchmarks/medqa.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/benchmarks/mmlu.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/benchmarks/mmlu_pro.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/benchmarks/multipl_e.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/benchmarks/naturalqs_open.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/benchmarks/piqa.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/benchmarks/sciq.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/benchmarks/social_iqa.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/benchmarks/squad.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/benchmarks/winogrande.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/dataset_revisions.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/eval_config.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/frozen-hf-dataset-revisions.json +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/markdown_doc.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/perturbation.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/task_style.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/utils.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/utils/constants.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/utils/file_ops.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/utils/helpers.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/utils/logging.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/utils/packaging.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/utils/tqdm_handler.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/template_formatting/README.md +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/template_formatting/__init__.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/template_formatting/formatter.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/template_formatting/mistral_formatter.py +0 -0
- {eval_framework-0.8.7 → eval_framework-0.8.11}/src/template_formatting/py.typed +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.3
|
|
2
2
|
Name: eval-framework
|
|
3
|
-
Version: 0.8.
|
|
3
|
+
Version: 0.8.11
|
|
4
4
|
Summary: Evaluation Framework
|
|
5
5
|
Author: Aleph Alpha Research
|
|
6
6
|
License: Apache License
|
|
@@ -216,14 +216,14 @@ Requires-Dist: xmltodict>=1.0.4,<1.1
|
|
|
216
216
|
Requires-Dist: pydantic>=2.13.4,<3
|
|
217
217
|
Requires-Dist: datasets>=5.0.1,<6
|
|
218
218
|
Requires-Dist: pycountry>=26.2.16,<27
|
|
219
|
-
Requires-Dist: nltk>=3.10.
|
|
219
|
+
Requires-Dist: nltk>=3.10.1,<4
|
|
220
220
|
Requires-Dist: python-dotenv>=1.2.2,<2
|
|
221
221
|
Requires-Dist: lingua-language-detector>=2.2.0,<3
|
|
222
222
|
Requires-Dist: google-crc32c>=1.8.0,<2
|
|
223
223
|
Requires-Dist: langdetect>=1.0.9,<2
|
|
224
224
|
Requires-Dist: spacy>=3.8.14,<4
|
|
225
225
|
Requires-Dist: jsonschema>=4.26.0,<5
|
|
226
|
-
Requires-Dist: mysql-connector-python>=
|
|
226
|
+
Requires-Dist: mysql-connector-python>=26.7.0,<27
|
|
227
227
|
Requires-Dist: psycopg2-binary>=2.9.12,<3
|
|
228
228
|
Requires-Dist: sympy>=1.14.0,<2
|
|
229
229
|
Requires-Dist: llm-sandbox[docker]==0.3.39
|
|
@@ -240,7 +240,7 @@ Requires-Dist: eval-framework[determined,api,openai,transformers,accelerate,opti
|
|
|
240
240
|
Requires-Dist: aleph-alpha-client>=11.5.1 ; extra == 'api'
|
|
241
241
|
Requires-Dist: determined>=0.38.1,<0.39 ; extra == 'determined'
|
|
242
242
|
Requires-Dist: tensorboard==2.21.0 ; extra == 'determined'
|
|
243
|
-
Requires-Dist: openai>=2.
|
|
243
|
+
Requires-Dist: openai>=2.52.0,<3 ; extra == 'openai'
|
|
244
244
|
Requires-Dist: tiktoken>=0.13.0,<1 ; extra == 'openai'
|
|
245
245
|
Requires-Dist: transformers>=4.45.2,<5 ; extra == 'openai'
|
|
246
246
|
Requires-Dist: transformers>=4.45.2,<5 ; extra == 'optional'
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "eval-framework"
|
|
3
|
-
version = "0.8.
|
|
3
|
+
version = "0.8.11"
|
|
4
4
|
description = "Evaluation Framework"
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
license = { file = "LICENSE" }
|
|
@@ -23,14 +23,14 @@ dependencies = [
|
|
|
23
23
|
"pydantic>=2.13.4,<3",
|
|
24
24
|
"datasets>=5.0.1,<6",
|
|
25
25
|
"pycountry>=26.2.16,<27",
|
|
26
|
-
"nltk>=3.10.
|
|
26
|
+
"nltk>=3.10.1,<4",
|
|
27
27
|
"python-dotenv>=1.2.2,<2",
|
|
28
28
|
"lingua-language-detector>=2.2.0,<3",
|
|
29
29
|
"google-crc32c>=1.8.0,<2",
|
|
30
30
|
"langdetect>=1.0.9,<2", # required by the original ifeval implementation
|
|
31
31
|
"spacy>=3.8.14,<4",
|
|
32
32
|
"jsonschema>=4.26.0,<5",
|
|
33
|
-
"mysql-connector-python>=
|
|
33
|
+
"mysql-connector-python>=26.7.0,<27", # required for sql-related tasks
|
|
34
34
|
"psycopg2-binary>=2.9.12,<3", # required for sql-related tasks
|
|
35
35
|
"sympy>=1.14.0,<2",
|
|
36
36
|
"llm-sandbox[docker]==0.3.39",
|
|
@@ -54,7 +54,7 @@ determined = [
|
|
|
54
54
|
]
|
|
55
55
|
api = ["aleph-alpha-client>=11.5.1"]
|
|
56
56
|
openai = [
|
|
57
|
-
"openai>=2.
|
|
57
|
+
"openai>=2.52.0,<3",
|
|
58
58
|
"tiktoken>=0.13.0,<1",
|
|
59
59
|
"transformers>=4.45.2,<5",
|
|
60
60
|
]
|
|
@@ -92,7 +92,7 @@ dev = [
|
|
|
92
92
|
"types-python-dateutil>=2.9.0.20260716,<3",
|
|
93
93
|
"types-requests>=2.33.0.20260712,<3",
|
|
94
94
|
"plotly>=6.9.0,<7",
|
|
95
|
-
"ruff>=0.16.
|
|
95
|
+
"ruff>=0.16.1",
|
|
96
96
|
"pip-licenses>=5.5.5",
|
|
97
97
|
]
|
|
98
98
|
flash-attn = [
|
|
@@ -18,7 +18,7 @@ from eval_framework.context.local import _load_model
|
|
|
18
18
|
from eval_framework.result_processors.result_processor import generate_output_dir
|
|
19
19
|
from eval_framework.run import _run_single_task
|
|
20
20
|
from eval_framework.tasks.eval_config import EvalConfig
|
|
21
|
-
from eval_framework.tasks.registry import
|
|
21
|
+
from eval_framework.tasks.registry import registry
|
|
22
22
|
|
|
23
23
|
logger = logging.getLogger(__name__)
|
|
24
24
|
|
|
@@ -108,7 +108,7 @@ class TaskSuite(BaseModel):
|
|
|
108
108
|
if isinstance(self.tasks, str):
|
|
109
109
|
if self.name is None:
|
|
110
110
|
self.name = self.tasks
|
|
111
|
-
if not
|
|
111
|
+
if self.tasks not in registry():
|
|
112
112
|
raise ValueError(f"Task '{self.tasks}' is not registered.")
|
|
113
113
|
elif not self.tasks:
|
|
114
114
|
raise ValueError(f"TaskSuite '{self.name}': 'tasks' must not be empty.")
|
{eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/hf-dataset-revisions.json
RENAMED
|
@@ -23,7 +23,6 @@
|
|
|
23
23
|
"google-research-datasets/nq_open": "5dd9790a83002ad084ddeb7c420dc716852c6f28",
|
|
24
24
|
"google/IFEval": "966cd89545d6b6acfd7638bc708b98261ca58e84",
|
|
25
25
|
"jzhang86/de_ifeval": "4f52d847003b3c83cc282e9d296853a24b34b19a",
|
|
26
|
-
"mandarjoshi/trivia_qa": "0f7faf33a3908546c6fd5b73a660e0f8ff173c2f",
|
|
27
26
|
"math-ai/aime25": "563bb8404243c5f09de6ec262f2db674fe5bce9b",
|
|
28
27
|
"math-ai/aime26": "79037aebdb6580008fb960d17cb21fd3099083e3",
|
|
29
28
|
"nuprl/MultiPL-E": "28441b6024e71d4a1c1c0f6bf171c935cd5a43f2",
|
|
@@ -255,10 +255,6 @@ class Registry:
|
|
|
255
255
|
|
|
256
256
|
return factory
|
|
257
257
|
|
|
258
|
-
def add(self, task: type[BaseTask]) -> None:
|
|
259
|
-
task_key = self._task_key(task.NAME)
|
|
260
|
-
self._registry[task_key] = (task.NAME, _Eager(task))
|
|
261
|
-
|
|
262
258
|
def __setitem__(self, name: str, factory: EvalFactory) -> None:
|
|
263
259
|
task_key = self._task_key(name)
|
|
264
260
|
if task_key in self._registry:
|
|
@@ -266,6 +262,24 @@ class Registry:
|
|
|
266
262
|
|
|
267
263
|
self._registry[task_key] = (name, factory)
|
|
268
264
|
|
|
265
|
+
def register(self, task: type[BaseTask]) -> str:
|
|
266
|
+
"""The class name is used as the task name."""
|
|
267
|
+
if not issubclass(task, BaseTask):
|
|
268
|
+
raise ValueError(f"Can only register subclasses of BaseTask, got {task}")
|
|
269
|
+
name = task.__name__
|
|
270
|
+
self[name] = _Eager(task)
|
|
271
|
+
return name
|
|
272
|
+
|
|
273
|
+
def register_lazy(self, class_path: str, /) -> None:
|
|
274
|
+
"""Register a task by its dotted class path, without importing its module."""
|
|
275
|
+
if "." not in class_path:
|
|
276
|
+
raise ValueError(
|
|
277
|
+
f"Invalid class path `{class_path}`. This needs to be a global path like "
|
|
278
|
+
"`eval_framework.tasks.benchmarks.mmlu.MMLU`): "
|
|
279
|
+
)
|
|
280
|
+
base_module, class_name = class_path.rsplit(".", maxsplit=1)
|
|
281
|
+
self[class_name] = _Lazy(class_name=class_name, module=base_module)
|
|
282
|
+
|
|
269
283
|
|
|
270
284
|
_REGISTRY = Registry()
|
|
271
285
|
|
|
@@ -298,7 +312,7 @@ def is_registered(name: str, /) -> bool:
|
|
|
298
312
|
|
|
299
313
|
def validate_task_name(name: str) -> str:
|
|
300
314
|
"""Pydantic-style validator for task names."""
|
|
301
|
-
if not
|
|
315
|
+
if name not in registry():
|
|
302
316
|
raise ValueError(f"Task not registered: {name}")
|
|
303
317
|
return name
|
|
304
318
|
|
|
@@ -313,28 +327,9 @@ def get_task(name: str, /) -> type[BaseTask]:
|
|
|
313
327
|
|
|
314
328
|
def register_task(task: type[BaseTask]) -> str:
|
|
315
329
|
"""The class name is used as the task name."""
|
|
316
|
-
|
|
317
|
-
raise ValueError(f"Can only register subclasses of BaseTask, got {task}")
|
|
318
|
-
name = task.__name__
|
|
319
|
-
_REGISTRY[name] = _Eager(task)
|
|
320
|
-
return name
|
|
330
|
+
return registry().register(task)
|
|
321
331
|
|
|
322
332
|
|
|
323
333
|
def register_lazy_task(class_path: str, /) -> None:
|
|
324
|
-
"""Register a task without importing
|
|
325
|
-
|
|
326
|
-
Lazily register a task without importing the module.
|
|
327
|
-
|
|
328
|
-
Args:
|
|
329
|
-
class_path: The full path to the task class. For example,
|
|
330
|
-
`eval_framework.tasks.benchmarks.mmlu.MMLU`.
|
|
331
|
-
extras: Any extra dependencies of `eval_framework` that need to be installed for this task.
|
|
332
|
-
"""
|
|
333
|
-
if "." not in class_path:
|
|
334
|
-
raise ValueError(
|
|
335
|
-
f"Invalid class path `{class_path}`. This needs to be a global path like "
|
|
336
|
-
"`eval_framework.tasks.benchmarks.mmlu.MMLU`): "
|
|
337
|
-
)
|
|
338
|
-
|
|
339
|
-
base_module, class_name = class_path.rsplit(".", maxsplit=1)
|
|
340
|
-
_REGISTRY[class_name] = _Lazy(class_name=class_name, module=base_module)
|
|
334
|
+
"""Register a task by its dotted class path, without importing its module."""
|
|
335
|
+
registry().register_lazy(class_path)
|
|
@@ -8,7 +8,8 @@ from types import ModuleType
|
|
|
8
8
|
from typing import Any
|
|
9
9
|
|
|
10
10
|
from eval_framework.tasks.base import BaseTask
|
|
11
|
-
from eval_framework.tasks.registry import
|
|
11
|
+
from eval_framework.tasks.registry import Registry
|
|
12
|
+
from eval_framework.tasks.registry import registry as global_registry
|
|
12
13
|
|
|
13
14
|
logger = logging.getLogger(__name__)
|
|
14
15
|
|
|
@@ -46,14 +47,13 @@ def import_file(f: str | os.PathLike, /) -> Any:
|
|
|
46
47
|
return user_module
|
|
47
48
|
|
|
48
49
|
|
|
49
|
-
def load_extra_tasks(module_paths: Sequence[str | os.PathLike]) -> None:
|
|
50
|
+
def load_extra_tasks(module_paths: Sequence[str | os.PathLike], registry: Registry | None = None) -> None:
|
|
50
51
|
"""Dynamically load and register user-defined tasks from a list of files or directories.
|
|
51
52
|
|
|
52
|
-
Each .py file found
|
|
53
|
-
in the TaskName enum for use by name.
|
|
54
|
-
Provides clear error messages for missing/invalid files or import errors.
|
|
53
|
+
Each .py file found is imported, and any BaseTask subclass is registered
|
|
55
54
|
"""
|
|
56
55
|
assert not (isinstance(module_paths, str)), "module_paths must be a sequence of strings / os.PathLike objects"
|
|
56
|
+
registry = registry if registry is not None else global_registry()
|
|
57
57
|
for file_path in find_all_python_files(*module_paths):
|
|
58
58
|
user_module = import_file(file_path)
|
|
59
59
|
|
|
@@ -63,7 +63,7 @@ def load_extra_tasks(module_paths: Sequence[str | os.PathLike]) -> None:
|
|
|
63
63
|
if not hasattr(obj, "NAME"):
|
|
64
64
|
logger.info(f"[User Task Loader] Skipping {obj.__module__} - no NAME attribute present.")
|
|
65
65
|
else:
|
|
66
|
-
if
|
|
66
|
+
if obj.NAME in registry:
|
|
67
67
|
# two classes with the same NAME attribute
|
|
68
68
|
logger.info(obj.__module__)
|
|
69
69
|
|
|
@@ -77,5 +77,5 @@ def load_extra_tasks(module_paths: Sequence[str | os.PathLike]) -> None:
|
|
|
77
77
|
else:
|
|
78
78
|
# if there is no duplicate name conflict then register the new task
|
|
79
79
|
class_obj = getattr(user_module, name)
|
|
80
|
-
|
|
80
|
+
registry.register(class_obj)
|
|
81
81
|
logger.info(f"[User Task Loader] Registered task: {class_obj.NAME}")
|
|
@@ -71,8 +71,6 @@ def register_all_tasks() -> None:
|
|
|
71
71
|
register_lazy_task("eval_framework.tasks.benchmarks.squad.SQuAD_OLMES")
|
|
72
72
|
register_lazy_task("eval_framework.tasks.benchmarks.squad.SQuAD2_MA")
|
|
73
73
|
register_lazy_task("eval_framework.tasks.benchmarks.squad.SQuAD2_MA_NO_SYSPROMPT")
|
|
74
|
-
register_lazy_task("eval_framework.tasks.benchmarks.triviaqa.TRIVIAQA")
|
|
75
|
-
register_lazy_task("eval_framework.tasks.benchmarks.triviaqa.TriviaQA_MA")
|
|
76
74
|
register_lazy_task("eval_framework.tasks.benchmarks.winogrande.WINOGRANDECloze")
|
|
77
75
|
register_lazy_task("eval_framework.tasks.benchmarks.csqa.CommonsenseQAMC_OLMES")
|
|
78
76
|
register_lazy_task("eval_framework.tasks.benchmarks.drop.DropCompletion_OLMES")
|
|
@@ -81,9 +79,3 @@ def register_all_tasks() -> None:
|
|
|
81
79
|
register_lazy_task("eval_framework.tasks.benchmarks.naturalqs_open.NaturalQsOpenMC_OLMES")
|
|
82
80
|
register_lazy_task("eval_framework.tasks.benchmarks.social_iqa.SocialIQAMC_OLMES")
|
|
83
81
|
register_lazy_task("eval_framework.tasks.benchmarks.medqa.MedQAMC_OLMES")
|
|
84
|
-
try:
|
|
85
|
-
# Importing the companion registers the additional tasks from the module.
|
|
86
|
-
# This is mostly for convenience for internal use-cases
|
|
87
|
-
import eval_framework_companion # noqa
|
|
88
|
-
except ImportError:
|
|
89
|
-
pass
|
|
@@ -1,78 +0,0 @@
|
|
|
1
|
-
import random
|
|
2
|
-
from typing import Any
|
|
3
|
-
|
|
4
|
-
from eval_framework.metrics.completion.accuracy_completion import AccuracyCompletion
|
|
5
|
-
from eval_framework.metrics.completion.f1 import F1, F1SquadNormalized
|
|
6
|
-
from eval_framework.tasks.base import BaseTask, Language, ResponseType, Sample
|
|
7
|
-
from eval_framework.tasks.dataset_revisions import HF_REVISIONS_LOCKFILE
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
class TRIVIAQA(BaseTask[str]):
|
|
11
|
-
"""Trivia QA dataset: https://huggingface.co/datasets/mandarjoshi/trivia_qa"""
|
|
12
|
-
|
|
13
|
-
REVISION_LOCKFILE = HF_REVISIONS_LOCKFILE
|
|
14
|
-
|
|
15
|
-
NAME = "TriviaQA"
|
|
16
|
-
DATASET_PATH = "mandarjoshi/trivia_qa"
|
|
17
|
-
SAMPLE_SPLIT = "validation"
|
|
18
|
-
FEWSHOT_SPLIT = "train"
|
|
19
|
-
RESPONSE_TYPE = ResponseType.COMPLETION
|
|
20
|
-
METRICS = [AccuracyCompletion, F1]
|
|
21
|
-
SUBJECTS = ["rc.wikipedia.nocontext"]
|
|
22
|
-
PERTURBATION_UNMODIFIABLE_WORDS = ["Question", "Answer"]
|
|
23
|
-
LANGUAGE = Language.ENG
|
|
24
|
-
|
|
25
|
-
def __init__(self, num_fewshot: int = 0) -> None:
|
|
26
|
-
super().__init__(num_fewshot)
|
|
27
|
-
self.stop_sequences = ["\n"]
|
|
28
|
-
self.max_tokens = 400 # the max length of the ground truth is 282 characters while the average is ~16
|
|
29
|
-
self.rnd_choice_shuffle = random.Random()
|
|
30
|
-
|
|
31
|
-
def _get_instruction_text(self, item: dict[str, Any]) -> str:
|
|
32
|
-
prompt = f"Question: {item['question'].strip()}\nAnswer:"
|
|
33
|
-
return prompt
|
|
34
|
-
|
|
35
|
-
def _get_fewshot_target_text(self, item: dict[str, Any]) -> str:
|
|
36
|
-
target = self._get_ground_truth(item)[0]
|
|
37
|
-
assert target is not None
|
|
38
|
-
assert isinstance(target, str)
|
|
39
|
-
return f" {target}"
|
|
40
|
-
|
|
41
|
-
def _get_ground_truth(self, item: dict[str, Any]) -> list[str]:
|
|
42
|
-
return item["answer"]["aliases"]
|
|
43
|
-
|
|
44
|
-
def post_process_generated_completion(self, completion_text: str, sample: Sample | None = None) -> str:
|
|
45
|
-
return completion_text.strip().rstrip(".")
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
class TriviaQA_MA(TRIVIAQA):
|
|
49
|
-
"""TriviaQA with the exact system prompt used in MA training"""
|
|
50
|
-
|
|
51
|
-
REVISION_LOCKFILE = HF_REVISIONS_LOCKFILE
|
|
52
|
-
|
|
53
|
-
NAME = "TriviaQA_MA"
|
|
54
|
-
SUBJECTS = ["rc.wikipedia"]
|
|
55
|
-
UNANSWERABLE_STR = "unanswerable"
|
|
56
|
-
|
|
57
|
-
METRICS = [AccuracyCompletion, F1, F1SquadNormalized]
|
|
58
|
-
PERTURBATION_UNMODIFIABLE_WORDS = ["Question", "Answer", "Context", "unanswerable"]
|
|
59
|
-
|
|
60
|
-
def __init__(self, num_fewshot: int = 0) -> None:
|
|
61
|
-
super().__init__(num_fewshot)
|
|
62
|
-
self.stop_sequences = []
|
|
63
|
-
self.max_tokens = 27_000
|
|
64
|
-
|
|
65
|
-
def _get_context_text(self, item: dict[str, Any]) -> str:
|
|
66
|
-
return "\n\n".join(item["entity_pages"]["wiki_context"])
|
|
67
|
-
|
|
68
|
-
def _get_system_prompt_text(self, item: dict[str, Any]) -> str | None:
|
|
69
|
-
return (
|
|
70
|
-
"You are a helpful assistant and will answer the user's questions carefully, "
|
|
71
|
-
"logically, accurately and well-reasoned.\n"
|
|
72
|
-
"Use the given context to answer the question faithfully. Answer only if the "
|
|
73
|
-
f"answer is present in the given context, otherwise respond with '{self.UNANSWERABLE_STR}' "
|
|
74
|
-
"if the answer is not present in the context."
|
|
75
|
-
)
|
|
76
|
-
|
|
77
|
-
def _get_instruction_text(self, item: dict[str, Any]) -> str:
|
|
78
|
-
return f"Context:\n{self._get_context_text(item)}\n\nQuestion:\n{item['question'].strip()}\n"
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/external/drop_process_results.py
RENAMED
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/external/ifeval_impl/README.md
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/external/ifeval_impl/utils.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/aggregators/__init__.py
RENAMED
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/aggregators/aggregators.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/completion/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/completion/csv_format.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/completion/ifeval.py
RENAMED
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/completion/json_format.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/completion/repetition.py
RENAMED
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/completion/rouge_1.py
RENAMED
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/completion/rouge_2.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/completion/rouge_l.py
RENAMED
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/completion/text_counter.py
RENAMED
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/efficiency/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/llm/graders/language.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/llm/graders/models.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/llm/llm_judge_coherence.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/llm/llm_judge_refusal.py
RENAMED
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/llm/llm_judge_sql.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/loglikelihood/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/loglikelihood/base.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/loglikelihood/dcs.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/metrics/loglikelihood/ternary.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/result_processors/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/result_processors/hf_uploader.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/Dockerfile_codebench
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/benchmarks/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/benchmarks/arc_de.py
RENAMED
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/benchmarks/bigcodebench.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/benchmarks/global_mmlu.py
RENAMED
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/benchmarks/goldenswag.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/benchmarks/hellaswag.py
RENAMED
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/benchmarks/humaneval.py
RENAMED
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/benchmarks/ifeval.py
RENAMED
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/benchmarks/math_reasoning.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/benchmarks/mmlu_pro.py
RENAMED
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/benchmarks/multipl_e.py
RENAMED
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/benchmarks/naturalqs_open.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/benchmarks/social_iqa.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/benchmarks/winogrande.py
RENAMED
|
File without changes
|
{eval_framework-0.8.7 → eval_framework-0.8.11}/src/eval_framework/tasks/dataset_revisions.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|