PyPI - evalscope - Versions diffs - 1.0.0__py3-none-any.whl → 1.2.0__py3-none-any.whl - Mend

evalscope 1.0.0py3-none-any.whl → 1.2.0py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Files changed (324) hide show

evalscope/api/benchmark/__init__.py +9 -1
evalscope/api/benchmark/adapters/__init__.py +4 -0
evalscope/api/benchmark/adapters/agent_adapter.py +8 -0
evalscope/api/benchmark/adapters/default_data_adapter.py +75 -4
evalscope/api/benchmark/adapters/image_edit_adapter.py +82 -0
evalscope/api/benchmark/adapters/multi_choice_adapter.py +5 -2
evalscope/api/benchmark/adapters/ner_adapter.py +212 -0
evalscope/api/benchmark/adapters/text2image_adapter.py +12 -10
evalscope/api/benchmark/adapters/vision_language_adapter.py +8 -0
evalscope/api/benchmark/benchmark.py +85 -2
evalscope/api/benchmark/meta.py +10 -1
evalscope/api/dataset/dataset.py +27 -6
evalscope/api/dataset/loader.py +8 -3
evalscope/api/evaluator/cache.py +31 -4
evalscope/api/evaluator/evaluator.py +5 -0
evalscope/api/evaluator/state.py +17 -1
evalscope/api/messages/__init__.py +1 -0
evalscope/api/messages/chat_message.py +52 -2
evalscope/api/metric/__init__.py +1 -1
evalscope/api/metric/metric.py +6 -1
evalscope/api/metric/scorer.py +15 -7
evalscope/api/mixin/__init__.py +1 -1
evalscope/api/mixin/llm_judge_mixin.py +2 -0
evalscope/api/mixin/sandbox_mixin.py +182 -0
evalscope/api/model/generate_config.py +10 -6
evalscope/api/model/model.py +5 -2
evalscope/api/tool/tool_info.py +1 -1
evalscope/app/app.py +3 -0
evalscope/app/ui/multi_model.py +6 -1
evalscope/app/ui/single_model.py +11 -5
evalscope/app/utils/data_utils.py +8 -7
evalscope/app/utils/env_utils.py +12 -0
evalscope/app/utils/text_utils.py +14 -12
evalscope/app/utils/visualization.py +2 -2
evalscope/arguments.py +8 -4
evalscope/backend/opencompass/backend_manager.py +0 -2
evalscope/backend/rag_eval/utils/embedding.py +9 -1
evalscope/benchmarks/aa_lcr/aa_lcr_adapter.py +205 -0
evalscope/benchmarks/ai2d/ai2d_adapter.py +54 -0
evalscope/benchmarks/aime/aime24_adapter.py +5 -0
evalscope/benchmarks/aime/aime25_adapter.py +136 -1
evalscope/benchmarks/aime/grader.py +307 -0
evalscope/benchmarks/aime/math_normalize.py +189 -0
evalscope/benchmarks/amc/amc_adapter.py +51 -0
evalscope/benchmarks/arena_hard/arena_hard_adapter.py +1 -0
evalscope/benchmarks/bbh/bbh_adapter.py +43 -17
evalscope/benchmarks/bfcl/{bfcl_adapter.py → v3/bfcl_v3_adapter.py} +131 -19
evalscope/benchmarks/bfcl/{generation.py → v3/generation.py} +9 -9
evalscope/benchmarks/bfcl/v3/utils.py +23 -0
evalscope/benchmarks/bfcl/v4/__init__.py +0 -0
evalscope/benchmarks/bfcl/v4/bfcl_v4_adapter.py +229 -0
evalscope/benchmarks/bfcl/v4/utils.py +410 -0
evalscope/benchmarks/biomix_qa/__init__.py +0 -0
evalscope/benchmarks/biomix_qa/biomix_qa_adapter.py +36 -0
evalscope/benchmarks/blink/__init__.py +0 -0
evalscope/benchmarks/blink/blink_adapter.py +61 -0
evalscope/benchmarks/ceval/ceval_adapter.py +1 -2
evalscope/benchmarks/chartqa/__init__.py +0 -0
evalscope/benchmarks/chartqa/chartqa_adapter.py +80 -0
evalscope/benchmarks/chartqa/utils.py +38 -0
evalscope/benchmarks/coin_flip/__init__.py +0 -0
evalscope/benchmarks/coin_flip/coin_flip_adapter.py +128 -0
evalscope/benchmarks/commonsense_qa/__init__.py +0 -0
evalscope/benchmarks/commonsense_qa/commonsense_qa_adapter.py +32 -0
evalscope/benchmarks/competition_math/competition_math_adapter.py +5 -0
evalscope/benchmarks/data_collection/data_collection_adapter.py +24 -19
evalscope/benchmarks/docvqa/__init__.py +0 -0
evalscope/benchmarks/docvqa/docvqa_adapter.py +67 -0
evalscope/benchmarks/drivelology/__init__.py +0 -0
evalscope/benchmarks/drivelology/drivelology_binary_adapter.py +170 -0
evalscope/benchmarks/drivelology/drivelology_multilabel_adapter.py +254 -0
evalscope/benchmarks/drivelology/drivelology_selection_adapter.py +49 -0
evalscope/benchmarks/drivelology/drivelology_writing_adapter.py +218 -0
evalscope/benchmarks/drop/drop_adapter.py +15 -44
evalscope/benchmarks/drop/utils.py +97 -0
evalscope/benchmarks/frames/frames_adapter.py +2 -1
evalscope/benchmarks/general_arena/general_arena_adapter.py +7 -2
evalscope/benchmarks/general_arena/utils.py +2 -1
evalscope/benchmarks/general_mcq/general_mcq_adapter.py +1 -1
evalscope/benchmarks/general_qa/general_qa_adapter.py +1 -1
evalscope/benchmarks/gsm8k/gsm8k_adapter.py +25 -9
evalscope/benchmarks/hallusion_bench/__init__.py +0 -0
evalscope/benchmarks/hallusion_bench/hallusion_bench_adapter.py +159 -0
evalscope/benchmarks/halu_eval/__init__.py +0 -0
evalscope/benchmarks/halu_eval/halu_eval_adapter.py +128 -0
evalscope/benchmarks/halu_eval/halu_eval_instructions.py +84 -0
evalscope/benchmarks/healthbench/__init__.py +0 -0
evalscope/benchmarks/healthbench/healthbench_adapter.py +282 -0
evalscope/benchmarks/healthbench/utils.py +102 -0
evalscope/benchmarks/hle/hle_adapter.py +3 -2
evalscope/benchmarks/humaneval/humaneval_adapter.py +24 -52
evalscope/benchmarks/humaneval/utils.py +235 -0
evalscope/benchmarks/ifeval/instructions_util.py +2 -3
evalscope/benchmarks/image_edit/__init__.py +0 -0
evalscope/benchmarks/image_edit/gedit/__init__.py +0 -0
evalscope/benchmarks/image_edit/gedit/gedit_adapter.py +138 -0
evalscope/benchmarks/image_edit/gedit/utils.py +372 -0
evalscope/benchmarks/image_edit/gedit/vie_prompts.py +406 -0
evalscope/benchmarks/infovqa/__init__.py +0 -0
evalscope/benchmarks/infovqa/infovqa_adapter.py +66 -0
evalscope/benchmarks/live_code_bench/evaluate_utils.py +13 -6
evalscope/benchmarks/live_code_bench/live_code_bench_adapter.py +66 -54
evalscope/benchmarks/live_code_bench/sandbox_evaluate_utils.py +220 -0
evalscope/benchmarks/logi_qa/__int__.py +0 -0
evalscope/benchmarks/logi_qa/logi_qa_adapter.py +41 -0
evalscope/benchmarks/math_500/math_500_adapter.py +5 -1
evalscope/benchmarks/math_qa/__init__.py +0 -0
evalscope/benchmarks/math_qa/math_qa_adapter.py +35 -0
evalscope/benchmarks/math_verse/__init__.py +0 -0
evalscope/benchmarks/math_verse/math_verse_adapter.py +105 -0
evalscope/benchmarks/math_vision/__init__.py +0 -0
evalscope/benchmarks/math_vision/math_vision_adapter.py +116 -0
evalscope/benchmarks/math_vista/__init__.py +0 -0
evalscope/benchmarks/math_vista/math_vista_adapter.py +114 -0
evalscope/benchmarks/med_mcqa/__init__.py +0 -0
evalscope/benchmarks/med_mcqa/med_mcqa_adapter.py +32 -0
evalscope/benchmarks/minerva_math/__init__.py +0 -0
evalscope/benchmarks/minerva_math/minerva_math_adapter.py +53 -0
evalscope/benchmarks/mm_bench/__init__.py +0 -0
evalscope/benchmarks/mm_bench/mm_bench_adapter.py +99 -0
evalscope/benchmarks/mm_star/__init__.py +0 -0
evalscope/benchmarks/mm_star/mm_star_adapter.py +73 -0
evalscope/benchmarks/mmlu_pro/mmlu_pro_adapter.py +1 -1
evalscope/benchmarks/mmmu/__init__.py +0 -0
evalscope/benchmarks/mmmu/mmmu_adapter.py +159 -0
evalscope/benchmarks/mmmu_pro/__init__.py +0 -0
evalscope/benchmarks/mmmu_pro/mmmu_pro_adapter.py +124 -0
evalscope/benchmarks/mri_mcqa/__init__.py +0 -0
evalscope/benchmarks/mri_mcqa/mri_mcqa_adapter.py +34 -0
evalscope/benchmarks/multi_if/__init__.py +0 -0
evalscope/benchmarks/multi_if/ifeval.py +3354 -0
evalscope/benchmarks/multi_if/metrics.py +120 -0
evalscope/benchmarks/multi_if/multi_if_adapter.py +161 -0
evalscope/benchmarks/music_trivia/__init__.py +0 -0
evalscope/benchmarks/music_trivia/music_trivia_adapter.py +36 -0
evalscope/benchmarks/needle_haystack/needle_haystack_adapter.py +7 -6
evalscope/benchmarks/ner/__init__.py +0 -0
evalscope/benchmarks/ner/broad_twitter_corpus_adapter.py +52 -0
evalscope/benchmarks/ner/conll2003_adapter.py +48 -0
evalscope/benchmarks/ner/copious_adapter.py +85 -0
evalscope/benchmarks/ner/cross_ner_adapter.py +120 -0
evalscope/benchmarks/ner/cross_ner_entities/__init__.py +0 -0
evalscope/benchmarks/ner/cross_ner_entities/ai.py +54 -0
evalscope/benchmarks/ner/cross_ner_entities/literature.py +36 -0
evalscope/benchmarks/ner/cross_ner_entities/music.py +39 -0
evalscope/benchmarks/ner/cross_ner_entities/politics.py +37 -0
evalscope/benchmarks/ner/cross_ner_entities/science.py +58 -0
evalscope/benchmarks/ner/genia_ner_adapter.py +66 -0
evalscope/benchmarks/ner/harvey_ner_adapter.py +58 -0
evalscope/benchmarks/ner/mit_movie_trivia_adapter.py +74 -0
evalscope/benchmarks/ner/mit_restaurant_adapter.py +66 -0
evalscope/benchmarks/ner/ontonotes5_adapter.py +87 -0
evalscope/benchmarks/ner/wnut2017_adapter.py +61 -0
evalscope/benchmarks/ocr_bench/__init__.py +0 -0
evalscope/benchmarks/ocr_bench/ocr_bench/__init__.py +0 -0
evalscope/benchmarks/ocr_bench/ocr_bench/ocr_bench_adapter.py +101 -0
evalscope/benchmarks/ocr_bench/ocr_bench_v2/IoUscore_metric.py +87 -0
evalscope/benchmarks/ocr_bench/ocr_bench_v2/TEDS_metric.py +963 -0
evalscope/benchmarks/ocr_bench/ocr_bench_v2/__init__.py +0 -0
evalscope/benchmarks/ocr_bench/ocr_bench_v2/ocr_bench_v2_adapter.py +161 -0
evalscope/benchmarks/ocr_bench/ocr_bench_v2/page_ocr_metric.py +50 -0
evalscope/benchmarks/ocr_bench/ocr_bench_v2/parallel.py +46 -0
evalscope/benchmarks/ocr_bench/ocr_bench_v2/spotting_eval/__init__.py +0 -0
evalscope/benchmarks/ocr_bench/ocr_bench_v2/spotting_eval/readme.txt +26 -0
evalscope/benchmarks/ocr_bench/ocr_bench_v2/spotting_eval/rrc_evaluation_funcs_1_1.py +537 -0
evalscope/benchmarks/ocr_bench/ocr_bench_v2/spotting_eval/script.py +481 -0
evalscope/benchmarks/ocr_bench/ocr_bench_v2/spotting_metric.py +179 -0
evalscope/benchmarks/ocr_bench/ocr_bench_v2/utils.py +433 -0
evalscope/benchmarks/ocr_bench/ocr_bench_v2/vqa_metric.py +254 -0
evalscope/benchmarks/olympiad_bench/__init__.py +0 -0
evalscope/benchmarks/olympiad_bench/olympiad_bench_adapter.py +163 -0
evalscope/benchmarks/olympiad_bench/utils.py +565 -0
evalscope/benchmarks/omni_bench/__init__.py +0 -0
evalscope/benchmarks/omni_bench/omni_bench_adapter.py +86 -0
evalscope/benchmarks/omnidoc_bench/__init__.py +0 -0
evalscope/benchmarks/omnidoc_bench/end2end_eval.py +349 -0
evalscope/benchmarks/omnidoc_bench/metrics.py +547 -0
evalscope/benchmarks/omnidoc_bench/omnidoc_bench_adapter.py +135 -0
evalscope/benchmarks/omnidoc_bench/utils.py +1937 -0
evalscope/benchmarks/piqa/__init__.py +0 -0
evalscope/benchmarks/piqa/piqa_adapter.py +32 -0
evalscope/benchmarks/poly_math/__init__.py +0 -0
evalscope/benchmarks/poly_math/poly_math_adapter.py +132 -0
evalscope/benchmarks/poly_math/utils/instruction.py +105 -0
evalscope/benchmarks/pope/__init__.py +0 -0
evalscope/benchmarks/pope/pope_adapter.py +112 -0
evalscope/benchmarks/process_bench/process_bench_adapter.py +1 -0
evalscope/benchmarks/pumed_qa/__init__.py +0 -0
evalscope/benchmarks/pumed_qa/pubmed_qa_adapter.py +175 -0
evalscope/benchmarks/qasc/__init__.py +0 -0
evalscope/benchmarks/qasc/qasc_adapter.py +35 -0
evalscope/benchmarks/real_world_qa/__init__.py +0 -0
evalscope/benchmarks/real_world_qa/real_world_qa_adapter.py +64 -0
evalscope/benchmarks/sciq/__init__.py +0 -0
evalscope/benchmarks/sciq/sciq_adapter.py +36 -0
evalscope/benchmarks/seed_bench_2_plus/__init__.py +0 -0
evalscope/benchmarks/seed_bench_2_plus/seed_bench_2_plus_adapter.py +72 -0
evalscope/benchmarks/simple_qa/simple_qa_adapter.py +1 -1
evalscope/benchmarks/simple_vqa/__init__.py +0 -0
evalscope/benchmarks/simple_vqa/simple_vqa_adapter.py +169 -0
evalscope/benchmarks/siqa/__init__.py +0 -0
evalscope/benchmarks/siqa/siqa_adapter.py +39 -0
evalscope/benchmarks/tau_bench/tau2_bench/__init__.py +0 -0
evalscope/benchmarks/tau_bench/tau2_bench/generation.py +158 -0
evalscope/benchmarks/tau_bench/tau2_bench/tau2_bench_adapter.py +146 -0
evalscope/benchmarks/tau_bench/tau_bench/__init__.py +0 -0
evalscope/benchmarks/tau_bench/{generation.py → tau_bench/generation.py} +1 -1
evalscope/benchmarks/tau_bench/{tau_bench_adapter.py → tau_bench/tau_bench_adapter.py} +29 -29
evalscope/benchmarks/text2image/__init__.py +0 -0
evalscope/benchmarks/{aigc/t2i → text2image}/evalmuse_adapter.py +3 -1
evalscope/benchmarks/{aigc/t2i → text2image}/genai_bench_adapter.py +2 -2
evalscope/benchmarks/{aigc/t2i → text2image}/general_t2i_adapter.py +1 -1
evalscope/benchmarks/{aigc/t2i → text2image}/hpdv2_adapter.py +7 -2
evalscope/benchmarks/{aigc/t2i → text2image}/tifa_adapter.py +1 -0
evalscope/benchmarks/tool_bench/tool_bench_adapter.py +3 -3
evalscope/benchmarks/truthful_qa/truthful_qa_adapter.py +1 -2
evalscope/benchmarks/visu_logic/__init__.py +0 -0
evalscope/benchmarks/visu_logic/visu_logic_adapter.py +75 -0
evalscope/benchmarks/wmt/__init__.py +0 -0
evalscope/benchmarks/wmt/wmt24_adapter.py +294 -0
evalscope/benchmarks/zerobench/__init__.py +0 -0
evalscope/benchmarks/zerobench/zerobench_adapter.py +64 -0
evalscope/cli/start_app.py +7 -1
evalscope/cli/start_perf.py +7 -1
evalscope/config.py +103 -18
evalscope/constants.py +18 -0
evalscope/evaluator/evaluator.py +138 -82
evalscope/metrics/bert_score/__init__.py +0 -0
evalscope/metrics/bert_score/scorer.py +338 -0
evalscope/metrics/bert_score/utils.py +697 -0
evalscope/metrics/llm_judge.py +19 -7
evalscope/metrics/math_parser.py +14 -0
evalscope/metrics/metric.py +317 -13
evalscope/metrics/metrics.py +37 -0
evalscope/metrics/t2v_metrics/models/vqascore_models/lavis/common/config.py +0 -0
evalscope/metrics/t2v_metrics/models/vqascore_models/lavis/common/dist_utils.py +0 -0
evalscope/metrics/t2v_metrics/models/vqascore_models/lavis/common/gradcam.py +0 -0
evalscope/metrics/t2v_metrics/models/vqascore_models/lavis/common/logger.py +0 -0
evalscope/metrics/t2v_metrics/models/vqascore_models/lavis/common/optims.py +0 -0
evalscope/metrics/t2v_metrics/models/vqascore_models/lavis/common/registry.py +0 -0
evalscope/metrics/t2v_metrics/models/vqascore_models/lavis/common/utils.py +0 -0
evalscope/metrics/t2v_metrics/models/vqascore_models/lavis/common/vqa_tools/__init__.py +0 -0
evalscope/metrics/t2v_metrics/models/vqascore_models/lavis/common/vqa_tools/vqa.py +0 -0
evalscope/metrics/t2v_metrics/models/vqascore_models/lavis/common/vqa_tools/vqa_eval.py +0 -0
evalscope/metrics/t2v_metrics/models/vqascore_models/lavis/models/blip2_models/Qformer.py +2 -6
evalscope/metrics/t2v_metrics/models/vqascore_models/lavis/models/blip_models/nlvr_encoder.py +2 -6
evalscope/metrics/t2v_metrics/models/vqascore_models/lavis/models/med.py +2 -6
evalscope/models/image_edit_model.py +125 -0
evalscope/models/model_apis.py +22 -0
evalscope/models/openai_compatible.py +21 -0
evalscope/models/text2image_model.py +2 -2
evalscope/models/utils/openai.py +16 -6
evalscope/perf/arguments.py +26 -4
evalscope/perf/benchmark.py +76 -89
evalscope/perf/http_client.py +31 -16
evalscope/perf/main.py +15 -2
evalscope/perf/plugin/api/base.py +9 -7
evalscope/perf/plugin/api/custom_api.py +13 -58
evalscope/perf/plugin/api/default_api.py +188 -79
evalscope/perf/plugin/api/openai_api.py +85 -20
evalscope/perf/plugin/datasets/base.py +21 -0
evalscope/perf/plugin/datasets/custom.py +2 -3
evalscope/perf/plugin/datasets/flickr8k.py +2 -2
evalscope/perf/plugin/datasets/kontext_bench.py +2 -2
evalscope/perf/plugin/datasets/line_by_line.py +2 -3
evalscope/perf/plugin/datasets/longalpaca.py +2 -3
evalscope/perf/plugin/datasets/openqa.py +2 -4
evalscope/perf/plugin/datasets/random_dataset.py +1 -3
evalscope/perf/plugin/datasets/random_vl_dataset.py +2 -2
evalscope/perf/utils/benchmark_util.py +43 -27
evalscope/perf/utils/db_util.py +14 -19
evalscope/perf/utils/local_server.py +3 -44
evalscope/perf/utils/log_utils.py +21 -6
evalscope/report/__init__.py +13 -3
evalscope/report/combinator.py +91 -20
evalscope/report/generator.py +8 -87
evalscope/report/report.py +8 -4
evalscope/run.py +13 -5
evalscope/third_party/toolbench_static/llm/swift_infer.py +0 -4
evalscope/utils/argument_utils.py +1 -1
evalscope/utils/chat_service.py +1 -1
evalscope/utils/function_utils.py +249 -12
evalscope/utils/import_utils.py +73 -1
evalscope/utils/io_utils.py +132 -7
evalscope/utils/json_schema.py +25 -2
evalscope/utils/logger.py +69 -18
evalscope/utils/model_utils.py +4 -3
evalscope/utils/multi_choices.py +39 -7
evalscope/utils/ner.py +377 -0
evalscope/version.py +2 -2
{evalscope-1.0.0.dist-info → evalscope-1.2.0.dist-info}/METADATA +252 -408
{evalscope-1.0.0.dist-info → evalscope-1.2.0.dist-info}/RECORD +290 -154
{evalscope-1.0.0.dist-info → evalscope-1.2.0.dist-info}/WHEEL +1 -1
{evalscope-1.0.0.dist-info → evalscope-1.2.0.dist-info}/top_level.txt +0 -1
evalscope/api/mixin/dataset_mixin.py +0 -105
evalscope/benchmarks/aigc/i2i/general_i2i_adapter.py +0 -44
tests/__init__.py +0 -1
tests/aigc/__init__.py +0 -1
tests/aigc/test_t2i.py +0 -142
tests/benchmark/__init__.py +0 -1
tests/benchmark/test_eval.py +0 -386
tests/cli/__init__.py +0 -1
tests/cli/test_all.py +0 -229
tests/cli/test_collection.py +0 -96
tests/cli/test_custom.py +0 -268
tests/perf/__init__.py +0 -1
tests/perf/test_perf.py +0 -176
tests/rag/test_clip_benchmark.py +0 -90
tests/rag/test_mteb.py +0 -213
tests/rag/test_ragas.py +0 -128
tests/swift/__init__.py +0 -1
tests/swift/test_run_swift_eval.py +0 -146
tests/swift/test_run_swift_vlm_eval.py +0 -128
tests/swift/test_run_swift_vlm_jugde_eval.py +0 -157
tests/test_run_all.py +0 -12
tests/utils.py +0 -13
tests/vlm/__init__.py +0 -1
tests/vlm/test_vlmeval.py +0 -102
/evalscope/benchmarks/{aigc → aa_lcr}/__init__.py +0 -0
/evalscope/benchmarks/{aigc/i2i → ai2d}/__init__.py +0 -0
/evalscope/benchmarks/{aigc/t2i → amc}/__init__.py +0 -0
{tests/rag → evalscope/benchmarks/bfcl/v3}/__init__.py +0 -0
{evalscope-1.0.0.dist-info → evalscope-1.2.0.dist-info}/entry_points.txt +0 -0
{evalscope-1.0.0.dist-info → evalscope-1.2.0.dist-info/licenses}/LICENSE +0 -0

evalscope/benchmarks/drop/utils.py CHANGED Viewed

@@ -1,5 +1,7 @@
+import numpy as np
 import re
 import string
+from typing import List
 _ARTICLES = re.compile(r'\b(a|an|the)\b', re.UNICODE)
@@ -57,3 +59,98 @@ def _normalize(answer):
     tokens = [token for token in tokens if token.strip()]
     normalized = ' '.join(tokens).strip()
     return normalized
+def _compute_f1(predicted_bag, gold_bag):
+    intersection = len(gold_bag.intersection(predicted_bag))
+    if not predicted_bag:
+        precision = 1.0
+    else:
+        precision = intersection / float(len(predicted_bag))
+    if not gold_bag:
+        recall = 1.0
+    else:
+        recall = intersection / float(len(gold_bag))
+    f1 = ((2 * precision * recall) / (precision + recall) if not (precision == 0.0 and recall == 0.0) else 0.0)
+    return f1
+def _match_numbers_if_present(gold_bag, predicted_bag):
+    gold_numbers = {word for word in gold_bag if _is_number(word)}
+    predicted_numbers = {word for word in predicted_bag if _is_number(word)}
+    if (not gold_numbers) or gold_numbers.intersection(predicted_numbers):
+        return True
+    return False
+def _align_bags(predicted, gold):
+    """
+    Takes gold and predicted answer sets and first finds the optimal 1-1 alignment
+    between them and gets maximum metric values over all the answers.
+    """
+    from scipy.optimize import linear_sum_assignment
+    scores = np.zeros([len(gold), len(predicted)])
+    for gold_index, gold_item in enumerate(gold):
+        for pred_index, pred_item in enumerate(predicted):
+            if _match_numbers_if_present(gold_item, pred_item):
+                scores[gold_index, pred_index] = _compute_f1(pred_item, gold_item)
+    row_ind, col_ind = linear_sum_assignment(-scores)
+    max_scores = np.zeros([max(len(gold), len(predicted))])
+    for row, column in zip(row_ind, col_ind):
+        max_scores[row] = max(max_scores[row], scores[row, column])
+    return max_scores
+def parse_answer(answer):
+    # NOTE: Everything is returned as a tuple for uniformity and hashability.
+    if answer['number'] != '':
+        return (str(answer['number']), )
+    if answer['spans'] != []:
+        return tuple(answer['spans'])
+    return (' '.join([answer['date']['day'], answer['date']['month'], answer['date']['year']]).strip(), )
+def _get_gold_answers(input_d: dict) -> List[str]:
+    """
+    Parse the raw input labels (gold).
+    """
+    def _flatten_validated_answers(validated_answers: dict) -> List[dict]:
+        """
+        Flatten the validated_answers structure into a list of answer dictionaries.
+        Expected input:
+            validated_answers: {
+                'number': [...],
+                'date':   [...],
+                'spans':  [...]
+            }
+        Each returned dict has keys: 'number', 'date', 'spans'.
+        If the input lists have different lengths, iteration stops at the shortest.
+        """
+        # Safely read lists from the input dict (default to empty lists)
+        numbers = validated_answers.get('number', [])
+        dates = validated_answers.get('date', [])
+        spans = validated_answers.get('spans', [])
+        # Ensure we only iterate as far as the shortest sequence to avoid IndexError
+        length = min(len(numbers), len(dates), len(spans))
+        flattened: List[dict] = []
+        for num, date, sp in zip(numbers[:length], dates[:length], spans[:length]):
+            flattened.append({'number': num, 'date': date, 'spans': sp})
+        return flattened
+    answers = []
+    answers_set = set()
+    candidates = [input_d['answer']] + _flatten_validated_answers(input_d['validated_answers'])
+    for candidate in candidates:
+        answer = parse_answer(candidate)
+        if answer in answers_set:
+            continue
+        answers_set.add(answer)
+        answers.append(answer)
+    return answers

evalscope/benchmarks/frames/frames_adapter.py CHANGED Viewed

@@ -61,7 +61,8 @@ class FramesAdapter(DefaultDataAdapter):
             sample_fields=self.record_to_sample,
             subset='test',
             limit=self.limit,
-            repeats=self.repeats
+            repeats=self.repeats,
+            shuffle=self.shuffle,
         ).load()
         test_dataset = DatasetDict({'test': dataset})

evalscope/benchmarks/general_arena/general_arena_adapter.py CHANGED Viewed

@@ -31,9 +31,10 @@ GRADER_TEMPLATE = "<|User Prompt|>\n{question}\n\n<|The Start of Assistant A's A
         'GeneralArena is a custom benchmark designed to evaluate the performance of large language models in a competitive setting, '
         'where models are pitted against each other in custom tasks to determine their relative strengths and weaknesses. You should '
         'provide the model outputs in the format of a list of dictionaries, where each dictionary contains the model name and its report path. '
-        'For detailed instructions on how to use this benchmark, please refer to the [Arena User Guide](https://evalscope.readthedocs.io/zh-cn/latest/user_guides/arena.html).',
+        'For detailed instructions on how to use this benchmark, please refer to the [Arena User Guide](https://evalscope.readthedocs.io/en/latest/user_guides/arena.html).',
         dataset_id='general_arena',
         metric_list=['winrate'],
+        aggregation='elo',
         few_shot_num=0,
         train_split=None,
         eval_split='test',
@@ -75,7 +76,11 @@ class GeneralArenaAdapter(DefaultDataAdapter):
         dataset_dict = {}
         for subset_name, samples in datasets.items():
             dataset = DictDataLoader(
-                dict_list=samples, limit=self.limit, repeats=self.repeats, sample_fields=self.record_to_sample
+                dict_list=samples,
+                limit=self.limit,
+                shuffle=self.shuffle,
+                repeats=self.repeats,
+                sample_fields=self.record_to_sample
             ).load()
             dataset_dict[subset_name] = dataset

evalscope/benchmarks/general_arena/utils.py CHANGED Viewed

@@ -34,7 +34,8 @@ def process_review_item(review_result: ReviewResult) -> list:
         'Index': str(review_result.index),
         'Input': review_result.input,
         'Question': review_result.input,  # Use input as question
-        'Generated': prediction if prediction != extracted_prediction else extracted_prediction,
+        'Generated':
+        prediction if prediction != extracted_prediction else extracted_prediction or '',  # Ensure no None value
         'Gold': target,
         'Pred': extracted_prediction,
         'Score': sample_score.score.model_dump(exclude_none=True),

evalscope/benchmarks/general_mcq/general_mcq_adapter.py CHANGED Viewed

@@ -20,7 +20,7 @@ logger = get_logger()
         name='general_mcq',
         pretty_name='General-MCQ',
         description='A general multiple-choice question answering dataset for custom evaluation. '
-        'For detailed instructions on how to use this benchmark, please refer to the [User Guide](https://evalscope.readthedocs.io/zh-cn/latest/advanced_guides/custom_dataset/llm.html#mcq).',
+        'For detailed instructions on how to use this benchmark, please refer to the [User Guide](https://evalscope.readthedocs.io/en/latest/advanced_guides/custom_dataset/llm.html#mcq).',
         tags=[Tags.MULTIPLE_CHOICE, Tags.CUSTOM],
         dataset_id='general_mcq',
         subset_list=['default'],

evalscope/benchmarks/general_qa/general_qa_adapter.py CHANGED Viewed

@@ -20,7 +20,7 @@ PROMPT_TEMPLATE = '请回答问题\n{question}'
         name='general_qa',
         pretty_name='General-QA',
         description='A general question answering dataset for custom evaluation. '
-        'For detailed instructions on how to use this benchmark, please refer to the [User Guide](https://evalscope.readthedocs.io/zh-cn/latest/advanced_guides/custom_dataset/llm.html#qa).',  # noqa: E501
+        'For detailed instructions on how to use this benchmark, please refer to the [User Guide](https://evalscope.readthedocs.io/en/latest/advanced_guides/custom_dataset/llm.html#qa).',  # noqa: E501
         tags=[Tags.QA, Tags.CUSTOM],
         dataset_id='general_qa',
         metric_list=['BLEU', 'Rouge'],

evalscope/benchmarks/gsm8k/gsm8k_adapter.py CHANGED Viewed

@@ -1,5 +1,6 @@
 # Copyright (c) Alibaba, Inc. and its affiliates.
+import re
 from typing import Any, Dict
 from evalscope.api.benchmark import BenchmarkMeta, DefaultDataAdapter
@@ -12,13 +13,26 @@ from evalscope.utils.logger import get_logger
 logger = get_logger()
 PROMPT_TEMPLATE = """
-Solve the following math problem step by step. The last line of your response should be of the form "ANSWER: $ANSWER" (without quotes) where $ANSWER is the answer to the problem.
+Solve the following math problem step by step. The last line of your response should display the answer enclosed within \\boxed{{\\text{{$ANSWER}}}}.
-{question}
+Example:
+Let's solve the problem step by step.
+Problem: Eliza's rate per hour for the first 40 hours she works each week is $10. She also receives an overtime pay of 1.2 times her regular hourly rate. If Eliza worked for 45 hours this week, how much are her earnings for this week?
-Remember to put your answer on its own line at the end in the form "ANSWER: $ANSWER" (without quotes) where $ANSWER is the answer to the problem, and you do not need to use a \\boxed command.
+Step 1: Calculate Eliza's earnings for the first 40 hours. Eliza's hourly rate is $10, so her earnings for the first 40 hours are $10/hour x 40 hours = $400.
+Step 2: Calculate Eliza's overtime pay rate. Eliza's overtime pay rate is 1.2 times her regular hourly rate, so her overtime pay rate is $10/hour x 1.2 = $12/hour.
+Step 3: Calculate Eliza's earnings for the overtime hours. Eliza worked for 45 hours, so her overtime hours are 45 hours - 40 hours = 5 hours. Her earnings for the overtime hours are $12/hour x 5 hours = $60.
+Step 4: Calculate Eliza's total earnings for the week. Eliza's total earnings for the week are her earnings for the first 40 hours plus her earnings for the overtime hours, which is $400 + $60 = $460.
+Answer:
+\\boxed{{\\text{{460}}}}
+question:
+{question}
-Reasoning:
+Remember to put your answer on its own line at the end in the form "\\boxed{{\\text{{$ANSWER}}}}" (without quotes), where $ANSWER is replaced by the actual answer to the problem.
 """.lstrip()  # noqa: E501
 FEWSHOT_TEMPLATE = """
@@ -41,7 +55,11 @@ Here are some examples of how to solve similar problems:
         few_shot_num=4,
         train_split='train',
         eval_split='test',
-        metric_list=['acc'],
+        metric_list=[{
+            'acc': {
+                'numeric': True
+            }
+        }],
         prompt_template=PROMPT_TEMPLATE,
         few_shot_prompt_template=FEWSHOT_TEMPLATE,
     )
@@ -69,8 +87,6 @@ class GSM8KAdapter(DefaultDataAdapter):
             return ''
     def extract_answer(self, prediction: str, task_state: TaskState):
-        from evalscope.filters.extraction import RegexFilter
+        from evalscope.metrics.math_parser import extract_answer
-        regex = RegexFilter(regex_pattern=r'(-?[0-9.,]{2,})|(-?[0-9]+)', group_select=-1)
-        res = regex(prediction)
-        return res.replace(',', '').replace('+', '').strip().strip('.')
+        return extract_answer(prediction)

evalscope/benchmarks/hallusion_bench/__init__.py ADDED Viewed

File without changes

evalscope/benchmarks/hallusion_bench/hallusion_bench_adapter.py ADDED Viewed

@@ -0,0 +1,159 @@
+from collections import defaultdict
+from typing import Any, Dict, List
+from evalscope.api.benchmark import BenchmarkMeta, VisionLanguageAdapter
+from evalscope.api.dataset import Sample
+from evalscope.api.evaluator.state import TaskState
+from evalscope.api.messages import ChatMessageUser, Content, ContentImage, ContentText
+from evalscope.api.metric.scorer import AggScore, SampleScore, Score
+from evalscope.api.registry import register_benchmark
+from evalscope.constants import Tags
+from evalscope.utils.io_utils import bytes_to_base64
+from evalscope.utils.logger import get_logger
+logger = get_logger()
+@register_benchmark(
+    BenchmarkMeta(
+        name='hallusion_bench',
+        pretty_name='HallusionBench',
+        tags=[Tags.MULTI_MODAL, Tags.HALLUCINATION, Tags.YES_NO],
+        description=
+        'HallusionBench is an advanced diagnostic benchmark designed to evaluate image-context reasoning, analyze models\' tendencies for language hallucination and visual illusion in large vision-language models (LVLMs).',  # noqa: E501
+        dataset_id='lmms-lab/HallusionBench',
+        metric_list=['aAcc', 'qAcc', 'fAcc'],
+        aggregation='f1',
+        eval_split='image',
+        prompt_template='{question}\nPlease answer YES or NO without an explanation.',
+    )
+)
+class HallusionBenchAdapter(VisionLanguageAdapter):
+    def __init__(self, **kwargs):
+        super().__init__(**kwargs)
+    def record_to_sample(self, record: Dict[str, Any]) -> Sample:
+        input_text = self.prompt_template.format(question=record['question'])
+        content_list: List[Content] = [ContentText(text=input_text)]
+        image = record.get('image')
+        if image:
+            image_base64 = bytes_to_base64(image['bytes'], format='png', add_header=True)
+            content_list.append(ContentImage(image=image_base64))
+        answer = 'NO' if str(record.get('answer', '0')) == '1' else 'YES'
+        return Sample(
+            input=[ChatMessageUser(content=content_list)],
+            target=answer,
+            metadata={
+                'category': record.get('category'),
+                'subcategory': record.get('subcategory'),
+                'visual_input': record.get('visual_input'),
+                'set_id': record.get('set_id'),
+                'figure_id': record.get('figure_id'),
+                'question_id': record.get('question_id'),
+            }
+        )
+    def match_score(self, original_prediction, filtered_prediction, reference, task_state) -> Score:
+        score = Score(
+            extracted_prediction=filtered_prediction,
+            prediction=original_prediction,
+        )
+        # Check if the reference answer is in the filtered prediction
+        result = 1 if reference in filtered_prediction.strip().upper() else 0
+        score.value = {'acc': result}
+        return score
+    def aggregate_scores(self, sample_scores: List[SampleScore]) -> List[AggScore]:
+        def compute_aAcc(scores: List[SampleScore]):
+            total = len(scores)
+            if total == 0:
+                return 0.0, 0
+            correct = sum(ss.score.main_value for ss in scores)
+            return (correct / total), total
+        def compute_group_accuracy(scores: List[SampleScore], group_type: str):
+            # group_type: 'figure' or 'question'
+            groups = defaultdict(list)
+            for ss in scores:
+                md = ss.sample_metadata
+                subcategory = md.get('subcategory')
+                set_id = md.get('set_id')
+                group_id = md.get('figure_id') if group_type == 'figure' else md.get('question_id')
+                if subcategory is None or set_id is None or group_id is None:
+                    # Skip incomplete records for this grouping
+                    continue
+                key = f'{subcategory}_{set_id}_{group_id}'
+                groups[key].append(ss.score.main_value)
+            if not groups:
+                return 0.0, 0
+            num_correct_groups = sum(1 for vals in groups.values() if all(vals))
+            num_groups = len(groups)
+            return (num_correct_groups / num_groups), num_groups
+        def compute_metrics(scores: List[SampleScore]) -> Dict[str, Dict[str, float]]:
+            a_acc, a_n = compute_aAcc(scores)
+            f_acc, f_n = compute_group_accuracy(scores, 'figure')
+            q_acc, q_n = compute_group_accuracy(scores, 'question')
+            return {
+                'aAcc': {
+                    'score': a_acc,
+                    'num': a_n
+                },
+                'fAcc': {
+                    'score': f_acc,
+                    'num': f_n
+                },
+                'qAcc': {
+                    'score': q_acc,
+                    'num': q_n
+                },
+            }
+        outputs: List[AggScore] = []
+        # By subcategory
+        subcategories = sorted({ss.sample_metadata.get('subcategory') for ss in sample_scores})
+        for subcategory in subcategories:
+            subset = [ss for ss in sample_scores if ss.sample_metadata.get('subcategory') == subcategory]
+            stats = compute_metrics(subset)
+            for metric in ['aAcc', 'fAcc', 'qAcc']:
+                outputs.append(
+                    AggScore(
+                        score=stats[metric]['score'],
+                        metric_name=metric,
+                        aggregation_name=str(subcategory),
+                        num=stats[metric]['num'],
+                    )
+                )
+        # By category
+        categories = sorted({ss.sample_metadata.get('category') for ss in sample_scores})
+        for category in categories:
+            subset = [ss for ss in sample_scores if ss.sample_metadata.get('category') == category]
+            stats = compute_metrics(subset)
+            for metric in ['aAcc', 'fAcc', 'qAcc']:
+                outputs.append(
+                    AggScore(
+                        score=stats[metric]['score'],
+                        metric_name=metric,
+                        aggregation_name=str(category),
+                        num=stats[metric]['num'],
+                    )
+                )
+        # Overall
+        overall = compute_metrics(sample_scores)
+        for metric in ['aAcc', 'fAcc', 'qAcc']:
+            outputs.append(
+                AggScore(
+                    score=overall[metric]['score'],
+                    metric_name=metric,
+                    aggregation_name='Overall',
+                    num=overall[metric]['num'],
+                )
+            )
+        return outputs

evalscope/benchmarks/halu_eval/__init__.py ADDED Viewed

File without changes

evalscope/benchmarks/halu_eval/halu_eval_adapter.py ADDED Viewed

@@ -0,0 +1,128 @@
+# flake8: noqa: E501
+from typing import Any, Dict, List
+from evalscope.api.benchmark import BenchmarkMeta, DefaultDataAdapter
+from evalscope.api.dataset import Sample
+from evalscope.api.messages import ChatMessageUser, Content, ContentText
+from evalscope.api.metric.scorer import AggScore, SampleScore, Score
+from evalscope.api.registry import register_benchmark
+from evalscope.benchmarks.halu_eval.halu_eval_instructions import (
+    DIALOGUE_INSTRUCTIONS,
+    QA_INSTRUCTIONS,
+    SUMMARIZATION_INSTRUCTIONS,
+)
+from evalscope.constants import Tags
+from evalscope.utils.logger import get_logger
+DESCRIPTION = (
+    'HaluEval is a large collection of generated and human-annotated hallucinated samples for evaluating the performance of LLMs in recognizing hallucination.'
+)
+logger = get_logger()
+@register_benchmark(
+    BenchmarkMeta(
+        name='halueval',
+        pretty_name='HaluEval',
+        tags=[Tags.KNOWLEDGE, Tags.HALLUCINATION, Tags.YES_NO],
+        description=DESCRIPTION.strip(),
+        dataset_id='evalscope/HaluEval',
+        subset_list=['dialogue_samples', 'qa_samples', 'summarization_samples'],
+        default_subset='Full',
+        metric_list=['accuracy', 'precision', 'recall', 'f1_score', 'yes_ratio'],
+        few_shot_num=0,
+        eval_split='data',
+        prompt_template='{question}'
+    )
+)
+class HaluEvalAdapter(DefaultDataAdapter):
+    def __init__(self, **kwargs):
+        super().__init__(**kwargs)
+        self.add_overall_metric = False
+    def record_to_sample(self, record: Dict[str, Any]) -> Sample:
+        if self.current_subset_name == 'dialogue_samples':
+            knowledge = record['knowledge']
+            dialogue_history = record['dialogue_history']
+            response = record['response']
+            hallucination = record['hallucination']
+            inputs = f'{DIALOGUE_INSTRUCTIONS}\n\n#Knowledge: {knowledge}\n#Dialogue History#: {dialogue_history}\n#Response#: {response}\n#Your Judgement#:'
+        elif self.current_subset_name == 'qa_samples':
+            knowledge = record['knowledge']
+            question = record['question']
+            answer = record['answer']
+            hallucination = record['hallucination']
+            inputs = f'{QA_INSTRUCTIONS}\n\n#Knowledge: {knowledge}\n#Question#: {question}\n#Answer#: {answer}\n#Your Judgement#:'
+        elif self.current_subset_name == 'summarization_samples':
+            document = record['document']
+            summary = record['summary']
+            hallucination = record['hallucination']
+            inputs = f'{SUMMARIZATION_INSTRUCTIONS}\n\n#Document#: {document}\n#Summary#: {summary}\n#Your Judgement#:'
+        input_text = self.prompt_template.format(question=inputs)
+        content_list: List[Content] = [ContentText(text=input_text)]
+        answer = str(hallucination).upper()  # 'YES' or 'NO'
+        return Sample(
+            input=[ChatMessageUser(content=content_list)], target=answer, metadata={
+                'answer': hallucination,
+            }
+        )
+    def match_score(self, original_prediction, filtered_prediction, reference, task_state) -> Score:
+        score = Score(
+            extracted_prediction=filtered_prediction,
+            prediction=original_prediction,
+        )
+        # Check if the reference answer is in the filtered prediction
+        result = 1 if reference in filtered_prediction.strip().upper() else 0
+        score.value = {'acc': result}
+        return score
+    def aggregate_scores(self, sample_scores: List[SampleScore]) -> List[AggScore]:
+        """
+        Custom aggregation to compute accuracy, precision, recall, f1_score, and yes_ratio.
+        """
+        def compute_metrics(scores: List[SampleScore]):
+            tp = fp = tn = fn = 0
+            yes_count = 0
+            total_count = len(scores)
+            for ss in scores:
+                gt = ss.sample_metadata['answer'].strip().upper()
+                # Get prediction based on score
+                pred = gt if ss.score.main_value == 1 else ('NO' if gt == 'YES' else 'YES')
+                if pred == 'YES':
+                    yes_count += 1
+                if pred == 'YES' and gt == 'YES':
+                    tp += 1
+                elif pred == 'YES' and gt == 'NO':
+                    fp += 1
+                elif pred == 'NO' and gt == 'NO':
+                    tn += 1
+                elif pred == 'NO' and gt == 'YES':
+                    fn += 1
+            accuracy = (tp + tn) / total_count if total_count > 0 else 0.0
+            precision = tp / (tp + fp) if (tp + fp) > 0 else 0.0
+            recall = tp / (tp + fn) if (tp + fn) > 0 else 0.0
+            f1_score = (2 * precision * recall) / (precision + recall) if (precision + recall) > 0 else 0.0
+            yes_ratio = yes_count / total_count if total_count > 0 else 0.0
+            return {
+                'accuracy': accuracy,
+                'precision': precision,
+                'recall': recall,
+                'f1_score': f1_score,
+                'yes_ratio': yes_ratio
+            }
+        overall_metrics = compute_metrics(sample_scores)
+        agg_scores = []
+        for metric_name, value in overall_metrics.items():
+            agg_scores.append(AggScore(metric_name=metric_name, score=value, num=len(sample_scores), metadata={}))
+        return agg_scores

evalscope/benchmarks/halu_eval/halu_eval_instructions.py ADDED Viewed

@@ -0,0 +1,84 @@
+# flake8: noqa: E501
+DIALOGUE_INSTRUCTIONS = """I want you act as a response judge. Given a dialogue history and a response, your objective is to determine if the provided response contains non-factual or hallucinated information. You SHOULD give your judgement based on the following hallucination types and the world knowledge.
+You are trying to determine if the true entity in the response is replaced with a highly similar entity.
+#Dialogue History#: [Human]: Could you recommand movies similar to The Dark Knight? [Assistant]: The sequel to Batman Begins is The Dark Knight. [Human]: Okay. Who is the director of The Dark Knight and any other movies from him not related to Batman?
+#Response#: Christopher Nolan was the director. He also directed insomnia and inception.
+#Your Judgement#: No
+#Dialogue History#: [Human]: Could you recommand movies similar to The Dark Knight? [Assistant]: The sequel to Batman Begins is The Dark Knight. [Human]: Okay. Who is the director of The Dark Knight and any other movies from him not related to Batman?
+#Response#: Steven Spielberg was the director. He also directed insomnia and inception.
+#Your Judgement#: Yes
+You are trying to determine if the true entity in the response is replaced with a dissimilar entity.
+#Dialogue History#: [Human]: Could you recommand movies similar to The Dark Knight? [Assistant]: The sequel to Batman Begins is The Dark Knight. [Human]: Okay. Who is the director of The Dark Knight and any other movies from him not related to Batman?
+#Response#: Christopher Nolan was the director. He also directed insomnia and inception.
+#Your Judgement#: No
+#Dialogue History#: [Human]: Could you recommand movies similar to The Dark Knight? [Assistant]: The sequel to Batman Begins is The Dark Knight. [Human]: Okay. Who is the director of The Dark Knight and any other movies from him not related to Batman?
+#Response#: Batman Begins was the director. He also directed insomnia and inception.
+#Your Judgement#: Yes
+You are trying to determine if the true entity in the response is replaced with a dissimilar entity in a different entity type.
+#Dialogue History#: [Human]: Could you recommand movies similar to The Dark Knight? [Assistant]: The sequel to Batman Begins is The Dark Knight. [Human]: Okay. Who is the director of The Dark Knight and any other movies from him not related to Batman?
+#Response#: Christopher Nolan was the director. He also directed insomnia and inception.
+#Your Judgement#: No
+#Dialogue History#: [Human]: Could you recommand movies similar to The Dark Knight? [Assistant]: The sequel to Batman Begins is The Dark Knight. [Human]: Okay. Who is the director of The Dark Knight and any other movies from him not related to Batman?
+#Response#: United States of America was the director. He also directed insomnia and inception.
+#Your Judgement#: Yes
+You should try your best to determine if the response contains non-factual or hallucinated information according to the above hallucination types. The answer you give MUST be \"Yes\" or \"No\"".
+"""
+QA_INSTRUCTIONS = """I want you act as an answer judge. Given a question and an answer, your objective is to determine if the provided answer contains non-factual or hallucinated information. You SHOULD give your judgement based on the following hallucination types and the world knowledge.
+You are trying to determine if the answer misunderstands the question context and intention.
+#Question#: What is a rare breed of dog that was derived as a variant of Rat Terrier, Shiloh Shepherd dog or American Hairless Terrier?
+#Answer#: American Hairless Terrier
+#Your Judgement#: No
+You are trying to determine if there is a factual contradiction between the answer and the world knowledge. Some information in the answer might be fabricated.
+#Question#: Are the New Orleans Outfall Canals the same length as the Augusta Canal?
+#Answer#: No, the New Orleans Outfall Canals and the Augusta Canal are not the same length. The Orleans Canal is approximately 3.6 miles (5.8 kilometers) long while the Augusta Canal is approximately 7 miles (11.3 kilometers) long.
+#Your Judgement#: Yes
+#Question#: What U.S Highway gives access to Zilpo Road, and is also known as Midland Trail?
+#Answer#: U.S Highway 70
+#Your Judgement#: Yes
+You are trying to determine if the answer is too general or too specific to answer the question at an appropriate level of specificity.
+#Question#: What genre do Superheaven and Oceansize belong to?
+#Answer#: Superheaven and Oceansize belong to the rock genre.
+#Your Judgement#: No
+#Question#: What profession do Kōbō Abe and Agatha Christie share?
+#Answer#: Playwright.
+#Your Judgement#: No
+You are trying to determine if the answer can be correctly inferred from the knowledge.
+#Question#: Which band has more members, Muse or The Raconteurs?
+#Answer#: Muse has more members than The Raconteurs.
+#Your Judgement#: Yes
+#Question#: Which is currently more valuable, Temagami-Lorrain Mine or Meadowbank Gold Mine?
+#Answer#: Meadowbank Gold Mine, since Meadowbank Gold Mine is still producing gold and the TemagamiLorrain Mine has been inactive for years.
+#Your Judgement#: No
+You should try your best to determine if the answer contains non-factual or hallucinated information according to the above hallucination types. The answer you give MUST be \"Yes\" or \"No\"".
+"""
+SUMMARIZATION_INSTRUCTIONS = """I want you act as a summary judge. Given a document and a summary, your objective is to determine if the provided summary contains non-factual or hallucinated information. You SHOULD give your judgement based on the following hallucination types and the world knowledge.
+You are trying to determine if the summary is factual but some information cannot be directly inferred or entailed from the document.
+#Document#: The panther chameleon was found on Monday by a dog walker in the wooded area at Marl Park. It had to be put down after X-rays showed all of its legs were broken and it had a deformed spine. RSPCA Cymru said it was an "extremely sad example of an abandoned and neglected exotic pet". Inspector Selina Chan said: "It is a possibility that the owners took on this animal but were unable to provide the care he needs and decided to release him to the wild. "We are urging potential owners of exotic animals to thoroughly research what is required in the care of the particular species before taking one on. "Potential owners need to make sure they can give their animal the environment it needs and they have the facilities, time, financial means and long-term commitment to maintain a good standard of care, as required under the Animal Welfare Act 2006." She added it was illegal to release non-native species into the wild.
+#Summary#: A chameleon that was found in a Cardiff park has been put down after being abandoned and neglected by its owners.
+#Your Judgement#: Yes
+You are trying to determine if there exists some non-factual and incorrect information in the summary.
+#Document#: The city was brought to a standstill on 15 December last year when a gunman held 18 hostages for 17 hours. Family members of victims Tori Johnson and Katrina Dawson were in attendance. Images of the floral tributes that filled the city centre in the wake of the siege were projected on to the cafe and surrounding buildings in an emotional twilight ceremony. Prime Minister Malcolm Turnbull gave an address saying a "whole nation resolved to answer hatred with love". "Testament to the spirit of Australians is that with such unnecessary, thoughtless tragedy, an amazing birth of mateship, unity and love occurs. Proud to be Australian," he said. How the Sydney siege unfolded New South Wales Premier Mike Baird has also announced plans for a permanent memorial to be built into the pavement in Martin Place. Clear cubes containing flowers will be embedded into the concrete and will shine with specialised lighting. It is a project inspired by the massive floral tributes that were left in the days after the siege. "Something remarkable happened here. As a city we were drawn to Martin Place. We came in shock and in sorrow but every step we took was with purpose," he said on Tuesday.
+#Summary#: Crowds have gathered in Sydney's Martin Place to honour the victims of the Lindt cafe siege, one year on.
+#Your Judgement#: No
+You are trying to determine if there is a factual contradiction between the summary and the document.
+#Document#: Christopher Huxtable, 34, from Swansea, had been missing since the collapse in February. His body was found on Wednesday and workers who carried out the search formed a guard of honour as it was driven from the site in the early hours of the morning. Ken Cresswell, 57, and John Shaw, 61, both from Rotherham, remain missing. The body of a fourth man, Michael Collings, 53, from Brotton, Teesside, was previously recovered from the site. Swansea East MP Carolyn Harris, who has been involved with the family since the incident, said they still did not know all the facts about the collapse. She said: "I feel very sad. My heart and my prayers go out to the family who have waited desperately for Christopher's body to be found. They can finally have closure, and say goodbye to him and grieve his loss. "But let's not forget that there's two other families who are still waiting for their loved ones to be returned." The building was due for demolition when it partially collapsed in February.
+#Summary#: The body of a man whose body was found at the site of the Swansea Bay Power Station collapse has been removed from the site.
+#Your Judgement#: Yes
+You should try your best to determine if the summary contains non-factual or hallucinated information according to the above hallucination types. The answer you give MUST be \"Yes\" or \"No\"".
+"""

evalscope/benchmarks/healthbench/__init__.py ADDED Viewed

File without changes

evalscope 1.0.0__py3-none-any.whl → 1.2.0__py3-none-any.whl

evalscope 1.0.0py3-none-any.whl → 1.2.0py3-none-any.whl