eval-framework 0.13.0__tar.gz → 0.13.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {eval_framework-0.13.0 → eval_framework-0.13.1}/PKG-INFO +1 -1
- {eval_framework-0.13.0 → eval_framework-0.13.1}/pyproject.toml +1 -1
- {eval_framework-0.13.0 → eval_framework-0.13.1}/pyproject.toml.orig +1 -1
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/completion/multipl_e_assertion.py +11 -2
- {eval_framework-0.13.0 → eval_framework-0.13.1}/LICENSE +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/README.md +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/__init__.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/base_config.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/benchmarks/__init__.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/benchmarks/arc_de.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/benchmarks/copa.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/benchmarks/csqa.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/benchmarks/csqa_ellamind.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/benchmarks/goldenswag.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/benchmarks/gpqa_ellamind.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/benchmarks/hellaswag.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/benchmarks/hellaswag_ellamind.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/benchmarks/hle_ellamind.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/benchmarks/medqa.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/benchmarks/mmlu.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/benchmarks/piqa.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/benchmarks/piqa_ellamind.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/benchmarks/sciq.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/benchmarks/simpleqa_ellamind.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/benchmarks/siqa_ellamind.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/benchmarks/winogrande_ellamind.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/choices.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/composed.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/context/__init__.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/context/determined.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/context/eval.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/context/local.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/contract.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/eval_kind.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/evaluation_generator.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/exceptions.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/external/drop_process_results.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/external/ifeval_impl/README.md +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/external/ifeval_impl/instructions.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/external/ifeval_impl/instructions_registry.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/external/ifeval_impl/instructions_util.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/external/ifeval_impl/utils.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/fewshot.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/llm/__init__.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/llm/aleph_alpha.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/llm/base.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/llm/huggingface.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/llm/models.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/llm/openai.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/logger.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/main.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/__init__.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/aggregators/__init__.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/aggregators/aggregators.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/base.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/completion/__init__.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/completion/accuracy_completion.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/completion/code_assertion.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/completion/code_execution_pass_at_one.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/completion/concordance_index.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/completion/csv_format.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/completion/drop_completion.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/completion/exponential_similarity.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/completion/f1.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/completion/format_checker.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/completion/grid_difference.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/completion/ifeval.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/completion/json_format.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/completion/language_checker.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/completion/length_control.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/completion/math_minerva_completion.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/completion/math_reasoning_completion.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/completion/minerva_math_utils.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/completion/placeholder_checker.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/completion/repetition.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/completion/rouge_1.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/completion/rouge_2.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/completion/rouge_geometric_mean.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/completion/rouge_l.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/completion/text_counter.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/efficiency/__init__.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/efficiency/bytes_per_sequence_position.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/efficiency/finish_reason.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/efficiency/token_counters.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/llm/__init__.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/llm/base.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/llm/graders/chatbot_style_grader.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/llm/graders/coherence_grader.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/llm/graders/comparison_grader.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/llm/graders/conciseness_grader.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/llm/graders/contains_names_grader.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/llm/graders/format_correctness_grader.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/llm/graders/instruction_grader.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/llm/graders/language.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/llm/graders/long_context_grader.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/llm/graders/models.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/llm/graders/refusal_grader.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/llm/graders/sql_quality_grader.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/llm/graders/summary_world_knowledge_grader.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/llm/llm_judge_chatbot_style.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/llm/llm_judge_coherence.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/llm/llm_judge_completion_accuracy.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/llm/llm_judge_conciseness.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/llm/llm_judge_contains_names.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/llm/llm_judge_format_correctness.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/llm/llm_judge_instruction.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/llm/llm_judge_mtbench_pair.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/llm/llm_judge_mtbench_single.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/llm/llm_judge_refusal.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/llm/llm_judge_sql.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/llm/llm_judge_world_knowledge.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/llm/utils.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/loglikelihood/__init__.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/loglikelihood/accuracy_loglikelihood.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/loglikelihood/base.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/loglikelihood/bits_per_byte.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/loglikelihood/confidence_weighted_accuracy.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/loglikelihood/dcs.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/loglikelihood/probability_mass.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/loglikelihood/ternary.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/py.typed +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/response_generator.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/result_processors/__init__.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/result_processors/base.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/result_processors/hf_uploader.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/result_processors/result_processor.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/result_processors/wandb_uploader.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/run.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/run_direct.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/shared/errors.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/shared/types.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/subjects.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/suite.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/Dockerfile_codebench +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/__init__.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/base.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/benchmarks/__init__.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/benchmarks/arc.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/benchmarks/arc_ellamind.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/benchmarks/bigcodebench.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/benchmarks/drop.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/benchmarks/global_mmlu.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/benchmarks/gpqa.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/benchmarks/gsm8k.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/benchmarks/gsm8k_ellamind.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/benchmarks/hendrycks_math_ellamind.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/benchmarks/humaneval.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/benchmarks/humaneval_ellamind.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/benchmarks/humaneval_plus.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/benchmarks/ifeval.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/benchmarks/math_reasoning.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/benchmarks/mbpp.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/benchmarks/mbpp_ellamind.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/benchmarks/mmlu_pro.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/benchmarks/multipl_e.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/benchmarks/naturalqs_open.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/benchmarks/social_iqa.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/benchmarks/squad.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/benchmarks/winogrande.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/dataset_loading.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/dataset_revisions.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/eval_config.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/frozen-hf-dataset-revisions.json +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/hf-dataset-revisions.json +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/lazy.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/markdown_doc.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/registry.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/task_loader.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/task_names.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/task_style.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/utils.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/utils/constants.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/utils/file_ops.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/utils/helpers.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/utils/logging.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/utils/packaging.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/utils/tqdm_handler.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/template_formatting/README.md +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/template_formatting/__init__.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/template_formatting/formatter.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/template_formatting/mistral_formatter.py +0 -0
- {eval_framework-0.13.0 → eval_framework-0.13.1}/src/template_formatting/py.typed +0 -0
|
@@ -21,6 +21,9 @@ _SANDBOX_LANG_MAP: dict[str, str] = {
|
|
|
21
21
|
"js": SupportedLanguage.JAVASCRIPT,
|
|
22
22
|
}
|
|
23
23
|
|
|
24
|
+
# Memory cap applied to every sandbox container
|
|
25
|
+
_RUNTIME_CONFIGS: dict[str, str] = {"mem_limit": "1g"}
|
|
26
|
+
|
|
24
27
|
|
|
25
28
|
class _CustomLangConfig(NamedTuple):
|
|
26
29
|
image: str
|
|
@@ -152,7 +155,7 @@ class MultiPLECodeAssertion(BaseMetric[Completion]):
|
|
|
152
155
|
def _execute_via_sandbox_run(full_code: str, sandbox_lang: str, timeout: int) -> tuple[bool, str]:
|
|
153
156
|
"""Use llm-sandbox's native session.run() for cpp, java, js."""
|
|
154
157
|
image = getattr(DefaultImage, sandbox_lang.upper())
|
|
155
|
-
pool = get_or_create_pool(image=image, lang=sandbox_lang)
|
|
158
|
+
pool = get_or_create_pool(image=image, lang=sandbox_lang, runtime_configs=_RUNTIME_CONFIGS)
|
|
156
159
|
with SandboxSession(pool=pool, lang=sandbox_lang) as session:
|
|
157
160
|
result: Any = session.run(full_code, timeout=timeout)
|
|
158
161
|
return result.success(), result.stdout + result.stderr
|
|
@@ -185,7 +188,13 @@ class MultiPLECodeAssertion(BaseMetric[Completion]):
|
|
|
185
188
|
code_file = f"{container_dir}/{code_filename}"
|
|
186
189
|
output = ""
|
|
187
190
|
|
|
188
|
-
pool = get_or_create_pool(
|
|
191
|
+
pool = get_or_create_pool(
|
|
192
|
+
image=cfg.image,
|
|
193
|
+
lang=SupportedLanguage.PYTHON,
|
|
194
|
+
min_pool_size=1,
|
|
195
|
+
max_pool_size=1,
|
|
196
|
+
runtime_configs=_RUNTIME_CONFIGS,
|
|
197
|
+
)
|
|
189
198
|
|
|
190
199
|
with tempfile.TemporaryDirectory() as tmp_dir:
|
|
191
200
|
tmp_path = os.path.join(tmp_dir, code_filename)
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/benchmarks/csqa_ellamind.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/benchmarks/gpqa_ellamind.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/benchmarks/hellaswag_ellamind.py
RENAMED
|
File without changes
|
{eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/benchmarks/hle_ellamind.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/benchmarks/piqa_ellamind.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/benchmarks/simpleqa_ellamind.py
RENAMED
|
File without changes
|
{eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/benchmarks/siqa_ellamind.py
RENAMED
|
File without changes
|
{eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/benchmarks/winogrande_ellamind.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/external/drop_process_results.py
RENAMED
|
File without changes
|
{eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/external/ifeval_impl/README.md
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/external/ifeval_impl/utils.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/aggregators/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/completion/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/completion/csv_format.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/completion/ifeval.py
RENAMED
|
File without changes
|
{eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/completion/json_format.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/completion/repetition.py
RENAMED
|
File without changes
|
{eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/completion/rouge_1.py
RENAMED
|
File without changes
|
{eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/completion/rouge_2.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/completion/rouge_l.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/efficiency/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/llm/graders/language.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/llm/graders/models.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/llm/llm_judge_refusal.py
RENAMED
|
File without changes
|
{eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/llm/llm_judge_sql.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/loglikelihood/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/loglikelihood/base.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/loglikelihood/dcs.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/metrics/loglikelihood/ternary.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/result_processors/__init__.py
RENAMED
|
File without changes
|
{eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/result_processors/base.py
RENAMED
|
File without changes
|
{eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/result_processors/hf_uploader.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/Dockerfile_codebench
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/benchmarks/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/benchmarks/arc_ellamind.py
RENAMED
|
File without changes
|
{eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/benchmarks/bigcodebench.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/benchmarks/global_mmlu.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/benchmarks/gsm8k.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/benchmarks/humaneval.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/benchmarks/ifeval.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/benchmarks/mbpp_ellamind.py
RENAMED
|
File without changes
|
{eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/benchmarks/mmlu_pro.py
RENAMED
|
File without changes
|
{eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/benchmarks/multipl_e.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/benchmarks/social_iqa.py
RENAMED
|
File without changes
|
{eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/benchmarks/squad.py
RENAMED
|
File without changes
|
{eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/benchmarks/winogrande.py
RENAMED
|
File without changes
|
|
File without changes
|
{eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/dataset_revisions.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.13.0 → eval_framework-0.13.1}/src/eval_framework/tasks/hf-dataset-revisions.json
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{eval_framework-0.13.0 → eval_framework-0.13.1}/src/template_formatting/mistral_formatter.py
RENAMED
|
File without changes
|
|
File without changes
|