PyPI - evalscope - Versions diffs - 1.0.0__py3-none-any.whl → 1.2.0__py3-none-any.whl - Mend

evalscope 1.0.0py3-none-any.whl → 1.2.0py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Files changed (324) hide show

evalscope/api/benchmark/__init__.py +9 -1
evalscope/api/benchmark/adapters/__init__.py +4 -0
evalscope/api/benchmark/adapters/agent_adapter.py +8 -0
evalscope/api/benchmark/adapters/default_data_adapter.py +75 -4
evalscope/api/benchmark/adapters/image_edit_adapter.py +82 -0
evalscope/api/benchmark/adapters/multi_choice_adapter.py +5 -2
evalscope/api/benchmark/adapters/ner_adapter.py +212 -0
evalscope/api/benchmark/adapters/text2image_adapter.py +12 -10
evalscope/api/benchmark/adapters/vision_language_adapter.py +8 -0
evalscope/api/benchmark/benchmark.py +85 -2
evalscope/api/benchmark/meta.py +10 -1
evalscope/api/dataset/dataset.py +27 -6
evalscope/api/dataset/loader.py +8 -3
evalscope/api/evaluator/cache.py +31 -4
evalscope/api/evaluator/evaluator.py +5 -0
evalscope/api/evaluator/state.py +17 -1
evalscope/api/messages/__init__.py +1 -0
evalscope/api/messages/chat_message.py +52 -2
evalscope/api/metric/__init__.py +1 -1
evalscope/api/metric/metric.py +6 -1
evalscope/api/metric/scorer.py +15 -7
evalscope/api/mixin/__init__.py +1 -1
evalscope/api/mixin/llm_judge_mixin.py +2 -0
evalscope/api/mixin/sandbox_mixin.py +182 -0
evalscope/api/model/generate_config.py +10 -6
evalscope/api/model/model.py +5 -2
evalscope/api/tool/tool_info.py +1 -1
evalscope/app/app.py +3 -0
evalscope/app/ui/multi_model.py +6 -1
evalscope/app/ui/single_model.py +11 -5
evalscope/app/utils/data_utils.py +8 -7
evalscope/app/utils/env_utils.py +12 -0
evalscope/app/utils/text_utils.py +14 -12
evalscope/app/utils/visualization.py +2 -2
evalscope/arguments.py +8 -4
evalscope/backend/opencompass/backend_manager.py +0 -2
evalscope/backend/rag_eval/utils/embedding.py +9 -1
evalscope/benchmarks/aa_lcr/aa_lcr_adapter.py +205 -0
evalscope/benchmarks/ai2d/ai2d_adapter.py +54 -0
evalscope/benchmarks/aime/aime24_adapter.py +5 -0
evalscope/benchmarks/aime/aime25_adapter.py +136 -1
evalscope/benchmarks/aime/grader.py +307 -0
evalscope/benchmarks/aime/math_normalize.py +189 -0
evalscope/benchmarks/amc/amc_adapter.py +51 -0
evalscope/benchmarks/arena_hard/arena_hard_adapter.py +1 -0
evalscope/benchmarks/bbh/bbh_adapter.py +43 -17
evalscope/benchmarks/bfcl/{bfcl_adapter.py → v3/bfcl_v3_adapter.py} +131 -19
evalscope/benchmarks/bfcl/{generation.py → v3/generation.py} +9 -9
evalscope/benchmarks/bfcl/v3/utils.py +23 -0
evalscope/benchmarks/bfcl/v4/__init__.py +0 -0
evalscope/benchmarks/bfcl/v4/bfcl_v4_adapter.py +229 -0
evalscope/benchmarks/bfcl/v4/utils.py +410 -0
evalscope/benchmarks/biomix_qa/__init__.py +0 -0
evalscope/benchmarks/biomix_qa/biomix_qa_adapter.py +36 -0
evalscope/benchmarks/blink/__init__.py +0 -0
evalscope/benchmarks/blink/blink_adapter.py +61 -0
evalscope/benchmarks/ceval/ceval_adapter.py +1 -2
evalscope/benchmarks/chartqa/__init__.py +0 -0
evalscope/benchmarks/chartqa/chartqa_adapter.py +80 -0
evalscope/benchmarks/chartqa/utils.py +38 -0
evalscope/benchmarks/coin_flip/__init__.py +0 -0
evalscope/benchmarks/coin_flip/coin_flip_adapter.py +128 -0
evalscope/benchmarks/commonsense_qa/__init__.py +0 -0
evalscope/benchmarks/commonsense_qa/commonsense_qa_adapter.py +32 -0
evalscope/benchmarks/competition_math/competition_math_adapter.py +5 -0
evalscope/benchmarks/data_collection/data_collection_adapter.py +24 -19
evalscope/benchmarks/docvqa/__init__.py +0 -0
evalscope/benchmarks/docvqa/docvqa_adapter.py +67 -0
evalscope/benchmarks/drivelology/__init__.py +0 -0
evalscope/benchmarks/drivelology/drivelology_binary_adapter.py +170 -0
evalscope/benchmarks/drivelology/drivelology_multilabel_adapter.py +254 -0
evalscope/benchmarks/drivelology/drivelology_selection_adapter.py +49 -0
evalscope/benchmarks/drivelology/drivelology_writing_adapter.py +218 -0
evalscope/benchmarks/drop/drop_adapter.py +15 -44
evalscope/benchmarks/drop/utils.py +97 -0
evalscope/benchmarks/frames/frames_adapter.py +2 -1
evalscope/benchmarks/general_arena/general_arena_adapter.py +7 -2
evalscope/benchmarks/general_arena/utils.py +2 -1
evalscope/benchmarks/general_mcq/general_mcq_adapter.py +1 -1
evalscope/benchmarks/general_qa/general_qa_adapter.py +1 -1
evalscope/benchmarks/gsm8k/gsm8k_adapter.py +25 -9
evalscope/benchmarks/hallusion_bench/__init__.py +0 -0
evalscope/benchmarks/hallusion_bench/hallusion_bench_adapter.py +159 -0
evalscope/benchmarks/halu_eval/__init__.py +0 -0
evalscope/benchmarks/halu_eval/halu_eval_adapter.py +128 -0
evalscope/benchmarks/halu_eval/halu_eval_instructions.py +84 -0
evalscope/benchmarks/healthbench/__init__.py +0 -0
evalscope/benchmarks/healthbench/healthbench_adapter.py +282 -0
evalscope/benchmarks/healthbench/utils.py +102 -0
evalscope/benchmarks/hle/hle_adapter.py +3 -2
evalscope/benchmarks/humaneval/humaneval_adapter.py +24 -52
evalscope/benchmarks/humaneval/utils.py +235 -0
evalscope/benchmarks/ifeval/instructions_util.py +2 -3
evalscope/benchmarks/image_edit/__init__.py +0 -0
evalscope/benchmarks/image_edit/gedit/__init__.py +0 -0
evalscope/benchmarks/image_edit/gedit/gedit_adapter.py +138 -0
evalscope/benchmarks/image_edit/gedit/utils.py +372 -0
evalscope/benchmarks/image_edit/gedit/vie_prompts.py +406 -0
evalscope/benchmarks/infovqa/__init__.py +0 -0
evalscope/benchmarks/infovqa/infovqa_adapter.py +66 -0
evalscope/benchmarks/live_code_bench/evaluate_utils.py +13 -6
evalscope/benchmarks/live_code_bench/live_code_bench_adapter.py +66 -54
evalscope/benchmarks/live_code_bench/sandbox_evaluate_utils.py +220 -0
evalscope/benchmarks/logi_qa/__int__.py +0 -0
evalscope/benchmarks/logi_qa/logi_qa_adapter.py +41 -0
evalscope/benchmarks/math_500/math_500_adapter.py +5 -1
evalscope/benchmarks/math_qa/__init__.py +0 -0
evalscope/benchmarks/math_qa/math_qa_adapter.py +35 -0
evalscope/benchmarks/math_verse/__init__.py +0 -0
evalscope/benchmarks/math_verse/math_verse_adapter.py +105 -0
evalscope/benchmarks/math_vision/__init__.py +0 -0
evalscope/benchmarks/math_vision/math_vision_adapter.py +116 -0
evalscope/benchmarks/math_vista/__init__.py +0 -0
evalscope/benchmarks/math_vista/math_vista_adapter.py +114 -0
evalscope/benchmarks/med_mcqa/__init__.py +0 -0
evalscope/benchmarks/med_mcqa/med_mcqa_adapter.py +32 -0
evalscope/benchmarks/minerva_math/__init__.py +0 -0
evalscope/benchmarks/minerva_math/minerva_math_adapter.py +53 -0
evalscope/benchmarks/mm_bench/__init__.py +0 -0
evalscope/benchmarks/mm_bench/mm_bench_adapter.py +99 -0
evalscope/benchmarks/mm_star/__init__.py +0 -0
evalscope/benchmarks/mm_star/mm_star_adapter.py +73 -0
evalscope/benchmarks/mmlu_pro/mmlu_pro_adapter.py +1 -1
evalscope/benchmarks/mmmu/__init__.py +0 -0
evalscope/benchmarks/mmmu/mmmu_adapter.py +159 -0
evalscope/benchmarks/mmmu_pro/__init__.py +0 -0
evalscope/benchmarks/mmmu_pro/mmmu_pro_adapter.py +124 -0
evalscope/benchmarks/mri_mcqa/__init__.py +0 -0
evalscope/benchmarks/mri_mcqa/mri_mcqa_adapter.py +34 -0
evalscope/benchmarks/multi_if/__init__.py +0 -0
evalscope/benchmarks/multi_if/ifeval.py +3354 -0
evalscope/benchmarks/multi_if/metrics.py +120 -0
evalscope/benchmarks/multi_if/multi_if_adapter.py +161 -0
evalscope/benchmarks/music_trivia/__init__.py +0 -0
evalscope/benchmarks/music_trivia/music_trivia_adapter.py +36 -0
evalscope/benchmarks/needle_haystack/needle_haystack_adapter.py +7 -6
evalscope/benchmarks/ner/__init__.py +0 -0
evalscope/benchmarks/ner/broad_twitter_corpus_adapter.py +52 -0
evalscope/benchmarks/ner/conll2003_adapter.py +48 -0
evalscope/benchmarks/ner/copious_adapter.py +85 -0
evalscope/benchmarks/ner/cross_ner_adapter.py +120 -0
evalscope/benchmarks/ner/cross_ner_entities/__init__.py +0 -0
evalscope/benchmarks/ner/cross_ner_entities/ai.py +54 -0
evalscope/benchmarks/ner/cross_ner_entities/literature.py +36 -0
evalscope/benchmarks/ner/cross_ner_entities/music.py +39 -0
evalscope/benchmarks/ner/cross_ner_entities/politics.py +37 -0
evalscope/benchmarks/ner/cross_ner_entities/science.py +58 -0
evalscope/benchmarks/ner/genia_ner_adapter.py +66 -0
evalscope/benchmarks/ner/harvey_ner_adapter.py +58 -0
evalscope/benchmarks/ner/mit_movie_trivia_adapter.py +74 -0
evalscope/benchmarks/ner/mit_restaurant_adapter.py +66 -0
evalscope/benchmarks/ner/ontonotes5_adapter.py +87 -0
evalscope/benchmarks/ner/wnut2017_adapter.py +61 -0
evalscope/benchmarks/ocr_bench/__init__.py +0 -0
evalscope/benchmarks/ocr_bench/ocr_bench/__init__.py +0 -0
evalscope/benchmarks/ocr_bench/ocr_bench/ocr_bench_adapter.py +101 -0
evalscope/benchmarks/ocr_bench/ocr_bench_v2/IoUscore_metric.py +87 -0
evalscope/benchmarks/ocr_bench/ocr_bench_v2/TEDS_metric.py +963 -0
evalscope/benchmarks/ocr_bench/ocr_bench_v2/__init__.py +0 -0
evalscope/benchmarks/ocr_bench/ocr_bench_v2/ocr_bench_v2_adapter.py +161 -0
evalscope/benchmarks/ocr_bench/ocr_bench_v2/page_ocr_metric.py +50 -0
evalscope/benchmarks/ocr_bench/ocr_bench_v2/parallel.py +46 -0
evalscope/benchmarks/ocr_bench/ocr_bench_v2/spotting_eval/__init__.py +0 -0
evalscope/benchmarks/ocr_bench/ocr_bench_v2/spotting_eval/readme.txt +26 -0
evalscope/benchmarks/ocr_bench/ocr_bench_v2/spotting_eval/rrc_evaluation_funcs_1_1.py +537 -0
evalscope/benchmarks/ocr_bench/ocr_bench_v2/spotting_eval/script.py +481 -0
evalscope/benchmarks/ocr_bench/ocr_bench_v2/spotting_metric.py +179 -0
evalscope/benchmarks/ocr_bench/ocr_bench_v2/utils.py +433 -0
evalscope/benchmarks/ocr_bench/ocr_bench_v2/vqa_metric.py +254 -0
evalscope/benchmarks/olympiad_bench/__init__.py +0 -0
evalscope/benchmarks/olympiad_bench/olympiad_bench_adapter.py +163 -0
evalscope/benchmarks/olympiad_bench/utils.py +565 -0
evalscope/benchmarks/omni_bench/__init__.py +0 -0
evalscope/benchmarks/omni_bench/omni_bench_adapter.py +86 -0
evalscope/benchmarks/omnidoc_bench/__init__.py +0 -0
evalscope/benchmarks/omnidoc_bench/end2end_eval.py +349 -0
evalscope/benchmarks/omnidoc_bench/metrics.py +547 -0
evalscope/benchmarks/omnidoc_bench/omnidoc_bench_adapter.py +135 -0
evalscope/benchmarks/omnidoc_bench/utils.py +1937 -0
evalscope/benchmarks/piqa/__init__.py +0 -0
evalscope/benchmarks/piqa/piqa_adapter.py +32 -0
evalscope/benchmarks/poly_math/__init__.py +0 -0
evalscope/benchmarks/poly_math/poly_math_adapter.py +132 -0
evalscope/benchmarks/poly_math/utils/instruction.py +105 -0
evalscope/benchmarks/pope/__init__.py +0 -0
evalscope/benchmarks/pope/pope_adapter.py +112 -0
evalscope/benchmarks/process_bench/process_bench_adapter.py +1 -0
evalscope/benchmarks/pumed_qa/__init__.py +0 -0
evalscope/benchmarks/pumed_qa/pubmed_qa_adapter.py +175 -0
evalscope/benchmarks/qasc/__init__.py +0 -0
evalscope/benchmarks/qasc/qasc_adapter.py +35 -0
evalscope/benchmarks/real_world_qa/__init__.py +0 -0
evalscope/benchmarks/real_world_qa/real_world_qa_adapter.py +64 -0
evalscope/benchmarks/sciq/__init__.py +0 -0
evalscope/benchmarks/sciq/sciq_adapter.py +36 -0
evalscope/benchmarks/seed_bench_2_plus/__init__.py +0 -0
evalscope/benchmarks/seed_bench_2_plus/seed_bench_2_plus_adapter.py +72 -0
evalscope/benchmarks/simple_qa/simple_qa_adapter.py +1 -1
evalscope/benchmarks/simple_vqa/__init__.py +0 -0
evalscope/benchmarks/simple_vqa/simple_vqa_adapter.py +169 -0
evalscope/benchmarks/siqa/__init__.py +0 -0
evalscope/benchmarks/siqa/siqa_adapter.py +39 -0
evalscope/benchmarks/tau_bench/tau2_bench/__init__.py +0 -0
evalscope/benchmarks/tau_bench/tau2_bench/generation.py +158 -0
evalscope/benchmarks/tau_bench/tau2_bench/tau2_bench_adapter.py +146 -0
evalscope/benchmarks/tau_bench/tau_bench/__init__.py +0 -0
evalscope/benchmarks/tau_bench/{generation.py → tau_bench/generation.py} +1 -1
evalscope/benchmarks/tau_bench/{tau_bench_adapter.py → tau_bench/tau_bench_adapter.py} +29 -29
evalscope/benchmarks/text2image/__init__.py +0 -0
evalscope/benchmarks/{aigc/t2i → text2image}/evalmuse_adapter.py +3 -1
evalscope/benchmarks/{aigc/t2i → text2image}/genai_bench_adapter.py +2 -2
evalscope/benchmarks/{aigc/t2i → text2image}/general_t2i_adapter.py +1 -1
evalscope/benchmarks/{aigc/t2i → text2image}/hpdv2_adapter.py +7 -2
evalscope/benchmarks/{aigc/t2i → text2image}/tifa_adapter.py +1 -0
evalscope/benchmarks/tool_bench/tool_bench_adapter.py +3 -3
evalscope/benchmarks/truthful_qa/truthful_qa_adapter.py +1 -2
evalscope/benchmarks/visu_logic/__init__.py +0 -0
evalscope/benchmarks/visu_logic/visu_logic_adapter.py +75 -0
evalscope/benchmarks/wmt/__init__.py +0 -0
evalscope/benchmarks/wmt/wmt24_adapter.py +294 -0
evalscope/benchmarks/zerobench/__init__.py +0 -0
evalscope/benchmarks/zerobench/zerobench_adapter.py +64 -0
evalscope/cli/start_app.py +7 -1
evalscope/cli/start_perf.py +7 -1
evalscope/config.py +103 -18
evalscope/constants.py +18 -0
evalscope/evaluator/evaluator.py +138 -82
evalscope/metrics/bert_score/__init__.py +0 -0
evalscope/metrics/bert_score/scorer.py +338 -0
evalscope/metrics/bert_score/utils.py +697 -0
evalscope/metrics/llm_judge.py +19 -7
evalscope/metrics/math_parser.py +14 -0
evalscope/metrics/metric.py +317 -13
evalscope/metrics/metrics.py +37 -0
evalscope/metrics/t2v_metrics/models/vqascore_models/lavis/common/config.py +0 -0
evalscope/metrics/t2v_metrics/models/vqascore_models/lavis/common/dist_utils.py +0 -0
evalscope/metrics/t2v_metrics/models/vqascore_models/lavis/common/gradcam.py +0 -0
evalscope/metrics/t2v_metrics/models/vqascore_models/lavis/common/logger.py +0 -0
evalscope/metrics/t2v_metrics/models/vqascore_models/lavis/common/optims.py +0 -0
evalscope/metrics/t2v_metrics/models/vqascore_models/lavis/common/registry.py +0 -0
evalscope/metrics/t2v_metrics/models/vqascore_models/lavis/common/utils.py +0 -0
evalscope/metrics/t2v_metrics/models/vqascore_models/lavis/common/vqa_tools/__init__.py +0 -0
evalscope/metrics/t2v_metrics/models/vqascore_models/lavis/common/vqa_tools/vqa.py +0 -0
evalscope/metrics/t2v_metrics/models/vqascore_models/lavis/common/vqa_tools/vqa_eval.py +0 -0
evalscope/metrics/t2v_metrics/models/vqascore_models/lavis/models/blip2_models/Qformer.py +2 -6
evalscope/metrics/t2v_metrics/models/vqascore_models/lavis/models/blip_models/nlvr_encoder.py +2 -6
evalscope/metrics/t2v_metrics/models/vqascore_models/lavis/models/med.py +2 -6
evalscope/models/image_edit_model.py +125 -0
evalscope/models/model_apis.py +22 -0
evalscope/models/openai_compatible.py +21 -0
evalscope/models/text2image_model.py +2 -2
evalscope/models/utils/openai.py +16 -6
evalscope/perf/arguments.py +26 -4
evalscope/perf/benchmark.py +76 -89
evalscope/perf/http_client.py +31 -16
evalscope/perf/main.py +15 -2
evalscope/perf/plugin/api/base.py +9 -7
evalscope/perf/plugin/api/custom_api.py +13 -58
evalscope/perf/plugin/api/default_api.py +188 -79
evalscope/perf/plugin/api/openai_api.py +85 -20
evalscope/perf/plugin/datasets/base.py +21 -0
evalscope/perf/plugin/datasets/custom.py +2 -3
evalscope/perf/plugin/datasets/flickr8k.py +2 -2
evalscope/perf/plugin/datasets/kontext_bench.py +2 -2
evalscope/perf/plugin/datasets/line_by_line.py +2 -3
evalscope/perf/plugin/datasets/longalpaca.py +2 -3
evalscope/perf/plugin/datasets/openqa.py +2 -4
evalscope/perf/plugin/datasets/random_dataset.py +1 -3
evalscope/perf/plugin/datasets/random_vl_dataset.py +2 -2
evalscope/perf/utils/benchmark_util.py +43 -27
evalscope/perf/utils/db_util.py +14 -19
evalscope/perf/utils/local_server.py +3 -44
evalscope/perf/utils/log_utils.py +21 -6
evalscope/report/__init__.py +13 -3
evalscope/report/combinator.py +91 -20
evalscope/report/generator.py +8 -87
evalscope/report/report.py +8 -4
evalscope/run.py +13 -5
evalscope/third_party/toolbench_static/llm/swift_infer.py +0 -4
evalscope/utils/argument_utils.py +1 -1
evalscope/utils/chat_service.py +1 -1
evalscope/utils/function_utils.py +249 -12
evalscope/utils/import_utils.py +73 -1
evalscope/utils/io_utils.py +132 -7
evalscope/utils/json_schema.py +25 -2
evalscope/utils/logger.py +69 -18
evalscope/utils/model_utils.py +4 -3
evalscope/utils/multi_choices.py +39 -7
evalscope/utils/ner.py +377 -0
evalscope/version.py +2 -2
{evalscope-1.0.0.dist-info → evalscope-1.2.0.dist-info}/METADATA +252 -408
{evalscope-1.0.0.dist-info → evalscope-1.2.0.dist-info}/RECORD +290 -154
{evalscope-1.0.0.dist-info → evalscope-1.2.0.dist-info}/WHEEL +1 -1
{evalscope-1.0.0.dist-info → evalscope-1.2.0.dist-info}/top_level.txt +0 -1
evalscope/api/mixin/dataset_mixin.py +0 -105
evalscope/benchmarks/aigc/i2i/general_i2i_adapter.py +0 -44
tests/__init__.py +0 -1
tests/aigc/__init__.py +0 -1
tests/aigc/test_t2i.py +0 -142
tests/benchmark/__init__.py +0 -1
tests/benchmark/test_eval.py +0 -386
tests/cli/__init__.py +0 -1
tests/cli/test_all.py +0 -229
tests/cli/test_collection.py +0 -96
tests/cli/test_custom.py +0 -268
tests/perf/__init__.py +0 -1
tests/perf/test_perf.py +0 -176
tests/rag/test_clip_benchmark.py +0 -90
tests/rag/test_mteb.py +0 -213
tests/rag/test_ragas.py +0 -128
tests/swift/__init__.py +0 -1
tests/swift/test_run_swift_eval.py +0 -146
tests/swift/test_run_swift_vlm_eval.py +0 -128
tests/swift/test_run_swift_vlm_jugde_eval.py +0 -157
tests/test_run_all.py +0 -12
tests/utils.py +0 -13
tests/vlm/__init__.py +0 -1
tests/vlm/test_vlmeval.py +0 -102
/evalscope/benchmarks/{aigc → aa_lcr}/__init__.py +0 -0
/evalscope/benchmarks/{aigc/i2i → ai2d}/__init__.py +0 -0
/evalscope/benchmarks/{aigc/t2i → amc}/__init__.py +0 -0
{tests/rag → evalscope/benchmarks/bfcl/v3}/__init__.py +0 -0
{evalscope-1.0.0.dist-info → evalscope-1.2.0.dist-info}/entry_points.txt +0 -0
{evalscope-1.0.0.dist-info → evalscope-1.2.0.dist-info/licenses}/LICENSE +0 -0

evalscope/benchmarks/live_code_bench/live_code_bench_adapter.py CHANGED Viewed

@@ -1,3 +1,4 @@
+# flake8: noqa: E501
 from typing import Any, Dict
 from evalscope.api.benchmark import BenchmarkMeta, DefaultDataAdapter
@@ -7,7 +8,7 @@ from evalscope.api.messages.chat_message import ChatMessageUser
 from evalscope.api.metric import Score
 from evalscope.api.registry import register_benchmark
 from evalscope.constants import Tags
-from evalscope.utils.io_utils import convert_numpy_types
+from evalscope.utils.io_utils import convert_normal_types
 from evalscope.utils.logger import get_logger
 logger = get_logger()
@@ -19,17 +20,18 @@ logger = get_logger()
         pretty_name='Live-Code-Bench',
         tags=[Tags.CODING],
         description=
-        'Live Code Bench is a benchmark for evaluating code generation models on real-world coding tasks. It includes a variety of programming problems with test cases to assess the model\'s ability to generate correct and efficient code solutions.',  # noqa: E501
+        'Live Code Bench is a benchmark for evaluating code generation models on real-world coding tasks. It includes a variety of programming problems with test cases to assess the model\'s ability to generate correct and efficient code solutions. '
+        '**By default the code is executed in local environment. We recommend using sandbox execution to safely run and evaluate the generated code, please refer to the [documentation](https://evalscope.readthedocs.io/en/latest/user_guides/sandbox.html) for more details.**',
         dataset_id='AI-ModelScope/code_generation_lite',
         subset_list=['release_latest'],
-        metric_list=['Pass@1'],
+        aggregation='mean_and_pass_at_k',
         eval_split='test',
         prompt_template=
         '### Question:\n{question_content}\n\n{format_prompt} ### Answer: (use the provided format with backticks)\n\n',
+        review_timeout=6,
         extra_params={
             'start_date': None,
             'end_date': None,
-            'timeout': 6,
             'debug': False
         },
     )
@@ -42,7 +44,6 @@ class LiveCodeBenchAdapter(DefaultDataAdapter):
     def __init__(self, **kwargs):
         super().__init__(**kwargs)
-        self.timeout = self.extra_params.get('timeout', 6)
         self.debug = self.extra_params.get('debug', False)
         self.start_date = self.extra_params.get('start_date')
         self.end_date = self.extra_params.get('end_date')
@@ -81,58 +82,69 @@ class LiveCodeBenchAdapter(DefaultDataAdapter):
     def match_score(
         self, original_prediction: str, filtered_prediction: str, reference: str, task_state: TaskState
     ) -> Score:
-        from .evaluate_utils import codegen_metrics
         score = Score(
             extracted_prediction=filtered_prediction,
             prediction=original_prediction,
         )
-        references = [{'input_output': task_state.metadata['evaluation_sample']}]
-        predictions = [[filtered_prediction]]
-        try:
-            metrics, eval_results, final_metadata = codegen_metrics(
-                references,
-                predictions,
-                k_list=[1],
-                num_process_evaluate=1,
-                timeout=self.timeout,
-                debug=self.debug,
-            )
-            pass_rate = metrics['pass@1'] / 100  # convert to point scale
-            score.value = {'pass': float(pass_rate > 0)}
-            score.explanation = f"Pass@1: {metrics['pass@1']}%"
-            # Convert numpy types to native Python types for JSON serialization
-            serializable_eval_results = convert_numpy_types(eval_results)
-            serializable_final_metadata = convert_numpy_types(final_metadata)
-            score.metadata = {
-                'pass_rate': float(pass_rate),
-                'timeout': self.timeout,
-                'debug': self.debug,
-                'eval_results': serializable_eval_results,
-                'final_metadata': serializable_final_metadata
-            }
-        except Exception as e:
-            score.value = {'pass': False}
-            score.explanation = f'Evaluation failed: {str(e)}'
-            score.metadata = {'error': str(e)}
-        score.main_score_name = 'pass'
+        if not self.use_sandbox:
+            # Use original evaluation method
+            from .evaluate_utils import codegen_metrics
+            references = [{'input_output': task_state.metadata['evaluation_sample']}]
+            predictions = [[filtered_prediction]]
+            try:
+                metrics, eval_results, final_metadata = codegen_metrics(
+                    references,
+                    predictions,
+                    k_list=[1],
+                    num_process_evaluate=1,
+                    timeout=self.review_timeout,
+                    debug=self.debug,
+                )
+                pass_rate = metrics['pass@1'] / 100  # convert to point scale
+                score.value = {'acc': float(pass_rate > 0)}
+                score.explanation = f"Pass@1: {metrics['pass@1']}%"
+                # Convert numpy types to native Python types for JSON serialization
+                serializable_eval_results = convert_normal_types(eval_results)
+                serializable_final_metadata = convert_normal_types(final_metadata)
+                score.metadata = {
+                    'pass_rate': float(pass_rate),
+                    'timeout': self.review_timeout,
+                    'debug': self.debug,
+                    'eval_results': serializable_eval_results,
+                    'final_metadata': serializable_final_metadata
+                }
+            except Exception as e:
+                score.value = {'acc': False}
+                score.explanation = f'Evaluation failed: {str(e)}'
+                score.metadata = {'error': str(e)}
+        else:
+            # Use sandbox execution
+            try:
+                from .sandbox_evaluate_utils import evaluate_in_sandbox
+                evaluation_sample = task_state.metadata['evaluation_sample']
+                passed, detailed_results = evaluate_in_sandbox(
+                    self, filtered_prediction, evaluation_sample, timeout=self.review_timeout, debug=self.debug
+                )
+                score.value = {'acc': passed}
+                score.explanation = f"Sandbox execution: {'Passed' if passed else 'Failed'}"
+                score.metadata = {
+                    'timeout': self.review_timeout,
+                    'debug': self.debug,
+                    'execution_method': 'sandbox',
+                    'detailed_results': detailed_results
+                }
+            except Exception as e:
+                score.value = {'acc': False}
+                score.explanation = f'Sandbox evaluation failed: {str(e)}'
+                score.metadata = {'error': str(e), 'execution_method': 'sandbox'}
+        score.main_score_name = 'acc'
         return score
-    def aggregate_scores(self, sample_scores):
-        from evalscope.metrics.metric import PassAtK
-        # calculate pass@k here
-        agg_list = []
-        for metric in self.metric_list:
-            if metric.lower().startswith('pass@'):
-                k = int(metric.split('@')[1])
-                # Get the scores for this metric
-                agg = PassAtK(k)
-                agg_list.extend(agg(sample_scores))
-        return agg_list

evalscope/benchmarks/live_code_bench/sandbox_evaluate_utils.py ADDED Viewed

@@ -0,0 +1,220 @@
+import json
+from typing import TYPE_CHECKING, Dict, List, Tuple
+from evalscope.utils.logger import get_logger
+if TYPE_CHECKING:
+    from evalscope.api.mixin.sandbox_mixin import SandboxMixin
+logger = get_logger()
+def evaluate_in_sandbox(
+    adapter: 'SandboxMixin',
+    code: str,
+    evaluation_sample: str,
+    timeout: int = 6,
+    debug: bool = False
+) -> Tuple[bool, Dict]:
+    """
+    Evaluate code in sandbox environment for Live Code Bench.
+    Args:
+        adapter: The adapter instance with sandbox capabilities
+        code: The code to evaluate
+        evaluation_sample: JSON string containing input/output test cases
+        timeout: Timeout for execution
+        debug: Whether to enable debug logging
+    Returns:
+        Tuple[bool, Dict]: (overall_pass, detailed_results)
+    """
+    try:
+        # Parse the evaluation sample
+        test_data = json.loads(evaluation_sample)
+        inputs = test_data.get('inputs', [])
+        outputs = test_data.get('outputs', [])
+        fn_name = test_data.get('fn_name')
+        if debug:
+            logger.info(f'Evaluating code with {len(inputs)} test cases')
+            logger.info(f'Function name: {fn_name}')
+        # Determine if this is call-based or stdio-based
+        if fn_name:
+            # Call-based evaluation
+            return _evaluate_call_based_in_sandbox(adapter, code, inputs, outputs, fn_name, timeout, debug)
+        else:
+            # Standard input/output evaluation
+            return _evaluate_stdio_in_sandbox(adapter, code, inputs, outputs, timeout, debug)
+    except Exception as e:
+        if debug:
+            logger.error(f'Sandbox evaluation error: {str(e)}')
+        return False, {'error': str(e), 'total_tests': 0, 'passed_tests': 0}
+def _evaluate_call_based_in_sandbox(
+    adapter: 'SandboxMixin', code: str, inputs: list, outputs: list, fn_name: str, timeout: int, debug: bool
+) -> Tuple[bool, Dict]:
+    """Evaluate call-based problems in sandbox."""
+    try:
+        all_passed = True
+        passed_count = 0
+        failed_cases = []
+        for i, (test_input, expected_output) in enumerate(zip(inputs, outputs)):
+            # Prepare individual test code for each test case
+            test_code = f"""
+import json
+import sys
+# User's code
+{code}
+# Test execution for single test case
+try:
+    test_input = {repr(test_input)}
+    expected_output = {repr(expected_output)}
+    if 'class Solution' in '''{code}''':
+        # LeetCode style
+        solution = Solution()
+        method = getattr(solution, '{fn_name}')
+    else:
+        # Function is directly available
+        method = {fn_name}
+    # Parse input if it's JSON string
+    if isinstance(test_input, str):
+        try:
+            test_input = json.loads(test_input)
+        except:
+            pass  # Keep as string if not valid JSON
+    # Call the method
+    if isinstance(test_input, list):
+        result = method(*test_input)
+    else:
+        result = method(test_input)
+    # Parse expected output if it's JSON string
+    if isinstance(expected_output, str):
+        try:
+            expected_output = json.loads(expected_output)
+        except:
+            pass  # Keep as string if not valid JSON
+    # Convert tuple to list for comparison
+    if isinstance(result, tuple):
+        result = list(result)
+    if result == expected_output:
+        print("TEST_PASSED")
+    else:
+        print(f"TEST_FAILED: expected {{expected_output}}, got {{result}}")
+except Exception as e:
+    print(f"EXECUTION_ERROR: {{str(e)}}")
+    import traceback
+    traceback.print_exc()
+"""
+            # Execute in sandbox
+            result = adapter.execute_code_in_sandbox(code=test_code, timeout=timeout, language='python')
+            if debug:
+                logger.info(f'Test case {i} execution result: {result}')
+            # Check if execution was successful and test passed
+            if result.get('status') == 'success':
+                output = result.get('output', '')
+                if 'TEST_PASSED' in output:
+                    passed_count += 1
+                elif 'TEST_FAILED:' in output:
+                    # Extract failure details from output
+                    for line in output.split('\n'):
+                        if line.startswith('TEST_FAILED:'):
+                            failed_cases.append(f"Test {i}: {line.replace('TEST_FAILED: ', '')}")
+                            break
+                    all_passed = False
+                    break
+                elif 'EXECUTION_ERROR:' in output:
+                    # Extract error details
+                    for line in output.split('\n'):
+                        if line.startswith('EXECUTION_ERROR:'):
+                            failed_cases.append(f'Test {i}: {line}')
+                            break
+                    all_passed = False
+                    break
+                else:
+                    failed_cases.append(f'Test {i}: Unknown error in output. Result: {result}')
+                    all_passed = False
+                    break
+            else:
+                failed_cases.append(f'Test {i}: Sandbox execution failed - Result: {result}')
+                all_passed = False
+                break
+        detailed_results = {'total_tests': len(inputs), 'passed_tests': passed_count, 'failed_cases': failed_cases}
+        return all_passed, detailed_results
+    except Exception as e:
+        if debug:
+            logger.error(f'Call-based evaluation error: {str(e)}')
+        return False, {'error': str(e), 'total_tests': len(inputs), 'passed_tests': 0}
+def _evaluate_stdio_in_sandbox(
+    adapter: 'SandboxMixin', code: str, inputs: list, outputs: list, timeout: int, debug: bool
+) -> Tuple[bool, Dict]:
+    """Evaluate stdio-based problems in sandbox."""
+    try:
+        all_passed = True
+        passed_count = 0
+        failed_cases = []
+        for i, (test_input, expected_output) in enumerate(zip(inputs, outputs)):
+            test_code = f"""
+import sys
+from io import StringIO
+# Redirect stdin
+sys.stdin = StringIO('''{test_input}''')
+# User's code
+{code}
+"""
+            # Execute in sandbox
+            result = adapter.execute_code_in_sandbox(code=test_code, timeout=timeout, language='python')
+            if result.get('status') != 'success':
+                if debug:
+                    logger.error(f'Test case {i} execution failed: {result}')
+                failed_cases.append(f'Test {i}: Execution error - Result: {result}')
+                all_passed = False
+                break
+            # Compare output
+            actual_output = result.get('output', '').strip()
+            expected_output = expected_output.strip()
+            if actual_output == expected_output:
+                passed_count += 1
+            else:
+                if debug:
+                    logger.info(f"Test case {i} failed: expected '{expected_output}', got '{actual_output}'")
+                failed_cases.append(f"Test {i}: Expected '{expected_output}', got '{actual_output}'")
+                all_passed = False
+                break
+        detailed_results = {'total_tests': len(inputs), 'passed_tests': passed_count, 'failed_cases': failed_cases}
+        return all_passed, detailed_results
+    except Exception as e:
+        if debug:
+            logger.error(f'Stdio evaluation error: {str(e)}')
+        return False, {'error': str(e), 'total_tests': len(inputs), 'passed_tests': 0}

evalscope/benchmarks/logi_qa/__int__.py ADDED Viewed

File without changes

evalscope/benchmarks/logi_qa/logi_qa_adapter.py ADDED Viewed

@@ -0,0 +1,41 @@
+# flake8: noqa: E501
+from evalscope.api.benchmark import BenchmarkMeta, MultiChoiceAdapter
+from evalscope.api.dataset import Sample
+from evalscope.api.registry import register_benchmark
+from evalscope.constants import Tags
+DESCRIPTION = 'LogiQA is a dataset sourced from expert-written questions for testing human Logical reasoning.'
+PROMPT_TEMPLATE = r"""
+Answer the following multiple choice question. The entire content of your response should be of the following format: 'ANSWER: $LETTER' (without quotes) where LETTER is one of {letters}.
+{question}
+{choices}
+""".strip()
+@register_benchmark(
+    BenchmarkMeta(
+        name='logi_qa',
+        pretty_name='LogiQA',
+        tags=[Tags.REASONING, Tags.MULTIPLE_CHOICE],
+        description=DESCRIPTION.strip(),
+        dataset_id='extraordinarylab/logiqa',
+        metric_list=['acc'],
+        few_shot_num=0,
+        train_split='validation',
+        eval_split='test',
+        prompt_template=PROMPT_TEMPLATE,
+    )
+)
+class LogiQAAdapter(MultiChoiceAdapter):
+    def record_to_sample(self, record) -> Sample:
+        return Sample(
+            input=f"{record['context']}\n{record['question']}",
+            choices=record['choices'],
+            target=record['answer'],
+            metadata={},
+        )

evalscope/benchmarks/math_500/math_500_adapter.py CHANGED Viewed

@@ -4,7 +4,6 @@ from typing import Any, Dict
 from evalscope.api.benchmark import BenchmarkMeta, DefaultDataAdapter
 from evalscope.api.dataset import Sample
-from evalscope.api.evaluator import TaskState
 from evalscope.api.registry import register_benchmark
 from evalscope.constants import Tags
 from evalscope.utils.logger import get_logger
@@ -49,3 +48,8 @@ class Math500Adapter(DefaultDataAdapter):
                 'solution': record['solution'],
             },
         )
+    def extract_answer(self, prediction: str, task_state):
+        from evalscope.metrics.math_parser import extract_answer
+        return extract_answer(prediction)

evalscope/benchmarks/math_qa/__init__.py ADDED Viewed

File without changes

evalscope/benchmarks/math_qa/math_qa_adapter.py ADDED Viewed

@@ -0,0 +1,35 @@
+from evalscope.api.benchmark import BenchmarkMeta, MultiChoiceAdapter
+from evalscope.api.dataset import Sample
+from evalscope.api.registry import register_benchmark
+from evalscope.constants import Tags
+from evalscope.utils.multi_choices import MultipleChoiceTemplate
+DESCRIPTION = (
+    'MathQA dataset is gathered by using a new representation language to annotate over the '
+    'AQuA-RAT dataset with fully-specified operational programs.'
+)
+@register_benchmark(
+    BenchmarkMeta(
+        name='math_qa',
+        pretty_name='MathQA',
+        tags=[Tags.REASONING, Tags.MATH, Tags.MULTIPLE_CHOICE],
+        description=DESCRIPTION.strip(),
+        dataset_id='extraordinarylab/math-qa',
+        metric_list=['acc'],
+        few_shot_num=0,
+        train_split=None,
+        eval_split='test',
+        prompt_template=MultipleChoiceTemplate.SINGLE_ANSWER_COT,
+    )
+)
+class MathQAAdapter(MultiChoiceAdapter):
+    def record_to_sample(self, record) -> Sample:
+        return Sample(
+            input=record['question'],
+            choices=record['choices'],
+            target=record['answer'],
+            metadata={'reasoning': record['reasoning']},
+        )

evalscope/benchmarks/math_verse/__init__.py ADDED Viewed

File without changes

evalscope/benchmarks/math_verse/math_verse_adapter.py ADDED Viewed

@@ -0,0 +1,105 @@
+# flake8: noqa: E501
+from typing import Any, Dict
+from evalscope.api.benchmark import BenchmarkMeta, VisionLanguageAdapter
+from evalscope.api.dataset import Sample
+from evalscope.api.messages import ChatMessageUser, Content, ContentImage, ContentText
+from evalscope.api.registry import register_benchmark
+from evalscope.constants import Tags
+from evalscope.utils.io_utils import bytes_to_base64
+from evalscope.utils.logger import get_logger
+logger = get_logger()
+MULTI_CHOICE_TYPE = 'multi-choice'
+OPEN_TYPE = 'free-form'
+OPEN_PROMPT = '{question}\nPlease reason step by step, and put your final answer within \\boxed{{}}.'
+MULT_CHOICE_PROMPT = """
+Answer the following multiple choice question. The last line of your response should be of the following format: 'ANSWER: $LETTER' (without quotes) where LETTER is one of A, B, C, D. Think step by step before answering.
+{question}
+"""
+SUBSET_LIST = ['Text Dominant', 'Text Lite', 'Vision Intensive', 'Vision Dominant', 'Vision Only']
+@register_benchmark(
+    BenchmarkMeta(
+        name='math_verse',
+        pretty_name='MathVerse',
+        dataset_id='evalscope/MathVerse',
+        tags=[Tags.MATH, Tags.REASONING, Tags.MULTIPLE_CHOICE, Tags.MULTI_MODAL],
+        description=
+        'MathVerse, an all-around visual math benchmark designed for an equitable and in-depth evaluation of MLLMs. 2,612 high-quality, multi-subject math problems with diagrams from publicly available sources. Each problem is then transformed by human annotators into six distinct versions, each offering varying degrees of information content in multi-modality, contributing to 15K test samples in total. This approach allows MathVerse to comprehensively assess whether and how much MLLMs can truly understand the visual diagrams for mathematical reasoning.',
+        subset_list=SUBSET_LIST,
+        metric_list=[{
+            'acc': {
+                'numeric': True
+            }
+        }],
+        default_subset='testmini',
+        eval_split='testmini',
+        prompt_template=OPEN_PROMPT,
+    )
+)
+class MathVerseAdapter(VisionLanguageAdapter):
+    def __init__(self, **kwargs):
+        super().__init__(**kwargs)
+        self.reformat_subset = True
+        self._use_llm_judge = True
+    def record_to_sample(self, record: Dict[str, Any]) -> Sample:
+        """
+        Convert a dataset record to a Sample. Unifies handling for both multi-choice and free-form.
+        Builds the content list inline and appends image content if provided.
+        Args:
+            record: Raw dataset record.
+        Returns:
+            Sample: The standardized sample ready for evaluation.
+        """
+        question_type = record.get('question_type', OPEN_TYPE)
+        question: str = record.get('question', '')
+        content_list: list[Content] = []
+        # Choose prompt text based on type; keep a single unified flow for creating Sample
+        if question_type == MULTI_CHOICE_TYPE:
+            prompt_text = MULT_CHOICE_PROMPT.format(question=question).strip()
+        else:
+            prompt_text = OPEN_PROMPT.format(question=question).strip()
+        content_list.append(ContentText(text=prompt_text))
+        # Append image if exists
+        image = record.get('image')
+        if image and isinstance(image, dict):
+            image_bytes = image.get('bytes')
+            if image_bytes:
+                image_base64 = bytes_to_base64(image_bytes, format='png', add_header=True)
+                content_list.append(ContentImage(image=image_base64))
+        metadata: Dict[str, Any] = {
+            'sample_index': record.get('sample_index'),
+            'problem_index': record.get('problem_index'),
+            'problem_version': record.get('problem_version'),
+            'question_type': question_type,
+            'query_wo': record.get('query_wo'),
+            'query_cot': record.get('query_cot'),
+            'question_for_eval': record.get('question_for_eval'),
+        }
+        return Sample(
+            input=[ChatMessageUser(content=content_list)],
+            target=record['answer'],
+            subset_key=record['problem_version'],
+            metadata=metadata,
+        )
+    def extract_answer(self, prediction: str, task_state):
+        from evalscope.metrics.math_parser import extract_answer
+        return extract_answer(prediction)

evalscope/benchmarks/math_vision/__init__.py ADDED Viewed

File without changes

evalscope 1.0.0__py3-none-any.whl → 1.2.0__py3-none-any.whl

evalscope 1.0.0py3-none-any.whl → 1.2.0py3-none-any.whl