flexeval 0.17.1__tar.gz → 0.17.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {flexeval-0.17.1 → flexeval-0.17.2}/PKG-INFO +1 -1
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/__init__.py +1 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/evaluate_chat_response.py +11 -7
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/evaluate_from_data.py +21 -3
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/language_model/base.py +15 -1
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/language_model/hf_lm.py +28 -6
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/language_model/vllm_model.py +28 -5
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/metric/llm_geval_score.py +1 -5
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/metric/llm_label.py +1 -7
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/metric/llm_score.py +11 -9
- flexeval-0.17.2/flexeval/core/reasoning_parser/__init__.py +2 -0
- flexeval-0.17.2/flexeval/core/reasoning_parser/base.py +30 -0
- flexeval-0.17.2/flexeval/core/reasoning_parser/regex_reasoning_parser.py +56 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/scripts/flexeval_file.py +1 -1
- {flexeval-0.17.1 → flexeval-0.17.2}/pyproject.toml +1 -1
- {flexeval-0.17.1 → flexeval-0.17.2}/LICENSE +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/README.md +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/__init__.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/chat_dataset/__init__.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/chat_dataset/base.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/chat_dataset/chatbot_bench.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/chat_dataset/chatbot_bench_datasets/README.md +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/chat_dataset/chatbot_bench_datasets/mt-en-ref-gpt4.jsonl +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/chat_dataset/chatbot_bench_datasets/mt-en.jsonl +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/chat_dataset/chatbot_bench_datasets/mt-ja-ref-gpt4.jsonl +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/chat_dataset/chatbot_bench_datasets/mt-ja-ref-gpt4o-with-human-annotation.jsonl +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/chat_dataset/chatbot_bench_datasets/mt-ja.jsonl +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/chat_dataset/chatbot_bench_datasets/rakuda-v2-ja.jsonl +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/chat_dataset/chatbot_bench_datasets/vicuna-en-ref-gpt4.jsonl +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/chat_dataset/chatbot_bench_datasets/vicuna-en.jsonl +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/chat_dataset/chatbot_bench_datasets/vicuna-ja-ref-gpt4.jsonl +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/chat_dataset/chatbot_bench_datasets/vicuna-ja.jsonl +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/chat_dataset/openai_messages.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/chat_dataset/sacrebleu_dataset.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/chat_dataset/template_based.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/eval_setups.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/evaluate_generation.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/evaluate_multiple_choice.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/evaluate_pairwise.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/evaluate_perplexity.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/evaluate_reward_model.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/few_shot_generator/__init__.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/few_shot_generator/balanced.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/few_shot_generator/base.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/few_shot_generator/fixed.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/few_shot_generator/rand.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/generation_dataset/__init__.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/generation_dataset/base.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/generation_dataset/sacrebleu_dataset.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/generation_dataset/template_based.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/language_model/__init__.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/language_model/litellm_api.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/language_model/openai_api.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/language_model/openai_batch_api.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/language_model/vllm_serve_lm.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/metric/__init__.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/metric/base.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/metric/bleu.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/metric/char_f1.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/metric/code_eval.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/metric/common_prefix_length.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/metric/common_string_length.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/metric/correlation.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/metric/exact_match.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/metric/finish_reason.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/metric/math.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/metric/output_length_stats.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/metric/perspective_api.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/metric/repetition_count.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/metric/repetition_n.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/metric/rouge.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/metric/sari.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/metric/substring_match.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/metric/tool_call.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/metric/utils.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/metric/xer.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/multiple_choice_dataset/__init__.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/multiple_choice_dataset/base.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/multiple_choice_dataset/template_based.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/pairwise_comparison/__init__.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/pairwise_comparison/judge/__init__.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/pairwise_comparison/judge/base.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/pairwise_comparison/judge/llm_judge.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/pairwise_comparison/match.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/pairwise_comparison/match_maker/__init__.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/pairwise_comparison/match_maker/all_combinations.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/pairwise_comparison/match_maker/base.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/pairwise_comparison/match_maker/random_combinations.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/pairwise_comparison/scorer/__init__.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/pairwise_comparison/scorer/base.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/pairwise_comparison/scorer/bradley_terry.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/pairwise_comparison/scorer/win_rate.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/prompt_template/__init__.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/prompt_template/base.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/prompt_template/jinja2.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/result_recorder/__init__.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/result_recorder/base.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/result_recorder/local_recorder.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/result_recorder/wandb_recorder.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/reward_bench_dataset/__init__.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/reward_bench_dataset/base.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/reward_bench_dataset/template_based.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/reward_model/__init__.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/reward_model/base.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/reward_model/log_prob.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/reward_model/pairwise_judge_reward_model.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/reward_model/sequence_classification.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/string_processor/__init__.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/string_processor/aio.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/string_processor/base.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/string_processor/last_line.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/string_processor/lower.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/string_processor/mgsm.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/string_processor/nfkc.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/string_processor/regex.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/string_processor/string_strip.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/string_processor/template.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/text_dataset/__init__.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/text_dataset/base.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/text_dataset/hf.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/text_dataset/jsonl.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/tokenizer/__init__.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/tokenizer/base.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/tokenizer/mecab.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/tokenizer/sacrebleu_tokenizer.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/tokenizer/tiktoken_tokenizer.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/tokenizer/transformers_tokenizer.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/tokenizer/whitespace.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/tool_parser/__init__.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/tool_parser/base.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/utils/__init__.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/utils/chat_util.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/utils/data_util.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/utils/jinja2_utils.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/utils/json_util.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/preset_configs/EvalSetup/code_chat/mbpp_chat.jsonnet +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/preset_configs/EvalSetup/code_generation/jhumaneval.jsonnet +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/preset_configs/EvalSetup/code_generation/jhumaneval_tab_indent.jsonnet +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/preset_configs/EvalSetup/code_generation/mbpp.jsonnet +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/preset_configs/EvalSetup/code_generation/mbpp_tab_indent.jsonnet +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/preset_configs/EvalSetup/code_generation/openai_humaneval.jsonnet +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/preset_configs/EvalSetup/code_generation/openai_humaneval_tab_indent.jsonnet +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/preset_configs/EvalSetup/en_chat/mt-en.jsonnet +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/preset_configs/EvalSetup/en_chat/vicuna-en.jsonnet +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/preset_configs/EvalSetup/en_generation/babi.jsonnet +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/preset_configs/EvalSetup/en_generation/commonsense_qa.jsonnet +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/preset_configs/EvalSetup/en_generation/gsm8k.jsonnet +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/preset_configs/EvalSetup/en_generation/squad_v1.jsonnet +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/preset_configs/EvalSetup/en_generation/trivia_qa.jsonnet +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/preset_configs/EvalSetup/en_generation/twitter_sentiment.jsonnet +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/preset_configs/EvalSetup/en_multiple_choice/arc_challenge.jsonnet +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/preset_configs/EvalSetup/en_multiple_choice/arc_easy.jsonnet +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/preset_configs/EvalSetup/en_multiple_choice/commonsense_qa_mc.jsonnet +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/preset_configs/EvalSetup/en_multiple_choice/hellaswag.jsonnet +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/preset_configs/EvalSetup/en_multiple_choice/openbookqa.jsonnet +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/preset_configs/EvalSetup/en_multiple_choice/piqa.jsonnet +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/preset_configs/EvalSetup/en_multiple_choice/xwinograd_en.jsonnet +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/preset_configs/EvalSetup/en_perplexity/tiny_shakespeare.jsonnet +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/preset_configs/EvalSetup/ja_chat/aio_chat.jsonnet +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/preset_configs/EvalSetup/ja_chat/elyza_tasks_100.jsonnet +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/preset_configs/EvalSetup/ja_chat/mgsm_ja_chat.jsonnet +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/preset_configs/EvalSetup/ja_chat/mt-ja.jsonnet +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/preset_configs/EvalSetup/ja_chat/rakuda-v2-ja.jsonnet +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/preset_configs/EvalSetup/ja_chat/vicuna-ja.jsonnet +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/preset_configs/EvalSetup/ja_generation/aio.jsonnet +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/preset_configs/EvalSetup/ja_generation/jamcqa.jsonnet +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/preset_configs/EvalSetup/ja_generation/jcommonsenseqa.jsonnet +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/preset_configs/EvalSetup/ja_generation/jnli.jsonnet +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/preset_configs/EvalSetup/ja_generation/jsquad.jsonnet +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/preset_configs/EvalSetup/ja_generation/mgsm_ja.jsonnet +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/preset_configs/EvalSetup/ja_generation/wrime_pos_neg.jsonnet +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/preset_configs/EvalSetup/ja_generation/xlsum_ja.jsonnet +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/preset_configs/EvalSetup/ja_multiple_choice/jcommonsenseqa_mc.jsonnet +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/preset_configs/EvalSetup/ja_multiple_choice/xwinograd_ja.jsonnet +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/preset_configs/EvalSetup/translation/wmt20_en_ja.jsonnet +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/preset_configs/EvalSetup/translation/wmt20_ja_en.jsonnet +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/preset_configs/EvalSetup/translation_chat/wmt20_en_ja_chat.jsonnet +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/preset_configs/EvalSetup/translation_chat/wmt20_ja_en_chat.jsonnet +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/preset_configs/Metric/assistant_eval_en_single_turn.jsonnet +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/preset_configs/Metric/assistant_eval_ja_single_turn.jsonnet +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/preset_configs/Metric/elyza_tasks_100_eval.jsonnet +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/preset_configs/PairwiseJudge/assistant_judge_en_single_turn.jsonnet +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/preset_configs/PairwiseJudge/assistant_judge_ja_single_turn.jsonnet +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/scripts/__init__.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/scripts/common.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/scripts/flexeval_lm.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/scripts/flexeval_pairwise.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/scripts/flexeval_presets.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/scripts/flexeval_reward.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/utils/__init__.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/utils/hf_utils.py +0 -0
- {flexeval-0.17.1 → flexeval-0.17.2}/flexeval/utils/module_utils.py +0 -0
|
@@ -13,6 +13,7 @@ from .core.metric import *
|
|
|
13
13
|
from .core.multiple_choice_dataset import *
|
|
14
14
|
from .core.pairwise_comparison import *
|
|
15
15
|
from .core.prompt_template import *
|
|
16
|
+
from .core.reasoning_parser import *
|
|
16
17
|
from .core.result_recorder import *
|
|
17
18
|
from .core.reward_bench_dataset import *
|
|
18
19
|
from .core.reward_model import *
|
|
@@ -122,6 +122,7 @@ def execute_conversation_flow(
|
|
|
122
122
|
"finish_reason": lm_output.finish_reason,
|
|
123
123
|
}
|
|
124
124
|
| ({"raw_content": lm_output.raw_text} if lm_output.raw_text else {})
|
|
125
|
+
| ({"reasoning_content": lm_output.reasoning_text} if lm_output.reasoning_text else {})
|
|
125
126
|
| ({"tool_calls": lm_output.tool_calls} if lm_output.tool_calls else {})
|
|
126
127
|
| (
|
|
127
128
|
{"tool_call_validation_result": lm_output.tool_call_validation_result}
|
|
@@ -215,22 +216,25 @@ def evaluate_chat_response( # noqa: C901, PLR0912
|
|
|
215
216
|
# legacy: `{"lm_output": "...", "finish_reason": "...", "task_inputs": {...}, "references": [...], **metrics}`
|
|
216
217
|
restructured_outputs: list[dict[str, Any]] = []
|
|
217
218
|
for output in outputs:
|
|
219
|
+
lm_output: LMOutput = output["lm_output"]
|
|
218
220
|
extra_info = output["chat_instance"].extra_info | {"messages": output["messages"]}
|
|
219
|
-
if output["lm_output"].tool_calls:
|
|
220
|
-
extra_info["tool_calls"] = output["lm_output"].tool_calls
|
|
221
221
|
if output["chat_instance"].tools:
|
|
222
222
|
extra_info["tools"] = output["chat_instance"].tools
|
|
223
223
|
restructured_output = {
|
|
224
|
-
"lm_output":
|
|
224
|
+
"lm_output": lm_output.text,
|
|
225
225
|
"finish_reason": output["lm_output"].finish_reason,
|
|
226
226
|
"extra_info": extra_info,
|
|
227
227
|
"references": output["chat_instance"].references,
|
|
228
228
|
**output["metrics"],
|
|
229
229
|
}
|
|
230
|
-
if
|
|
231
|
-
restructured_output["raw_lm_output"] =
|
|
232
|
-
if
|
|
233
|
-
restructured_output["reasoning_text"] =
|
|
230
|
+
if lm_output.raw_text:
|
|
231
|
+
restructured_output["raw_lm_output"] = lm_output.raw_text
|
|
232
|
+
if lm_output.reasoning_text:
|
|
233
|
+
restructured_output["reasoning_text"] = lm_output.reasoning_text
|
|
234
|
+
if lm_output.tool_calls:
|
|
235
|
+
restructured_output["tool_calls"] = lm_output.tool_calls
|
|
236
|
+
if lm_output.tool_call_validation_result:
|
|
237
|
+
restructured_output["tool_call_validation_result"] = lm_output.tool_call_validation_result
|
|
234
238
|
restructured_outputs.append(restructured_output)
|
|
235
239
|
|
|
236
240
|
return metrics_summary_dict, restructured_outputs
|
|
@@ -5,18 +5,20 @@ from typing import Any
|
|
|
5
5
|
|
|
6
6
|
from loguru import logger
|
|
7
7
|
|
|
8
|
+
from flexeval.core.language_model.base import LMOutput
|
|
9
|
+
|
|
8
10
|
from .chat_dataset import ChatDataset
|
|
9
11
|
from .generation_dataset import GenerationDataset
|
|
10
12
|
from .metric import Metric
|
|
11
13
|
|
|
12
14
|
|
|
13
|
-
def evaluate_from_data(
|
|
15
|
+
def evaluate_from_data( # noqa: C901
|
|
14
16
|
eval_data: Iterable[dict[str, Any]],
|
|
15
17
|
metrics: list[Metric],
|
|
16
18
|
eval_dataset: GenerationDataset | ChatDataset | None = None,
|
|
17
19
|
) -> tuple[dict[str, float], list[dict[str, Any]]]:
|
|
18
20
|
extra_info_list: list[dict[str, Any]] = []
|
|
19
|
-
lm_output_list: list[str] = []
|
|
21
|
+
lm_output_list: list[str | LMOutput] = []
|
|
20
22
|
references_list: list[list[str]] = []
|
|
21
23
|
for item in eval_data:
|
|
22
24
|
# ignore extra info if empty for backward compatibility
|
|
@@ -24,7 +26,23 @@ def evaluate_from_data(
|
|
|
24
26
|
if "extra_info" not in item:
|
|
25
27
|
item["extra_info"] = {}
|
|
26
28
|
extra_info_list.append(item["extra_info"])
|
|
27
|
-
|
|
29
|
+
# We adapt the current output format of evaluate_chat_response(). This will be changed in the future.
|
|
30
|
+
if isinstance(item["lm_output"], str):
|
|
31
|
+
lm_output = LMOutput(
|
|
32
|
+
text=item["lm_output"],
|
|
33
|
+
raw_text=item.get("raw_lm_output"),
|
|
34
|
+
reasoning_text=item.get("reasoning_text"),
|
|
35
|
+
finish_reason=item.get("finish_reason"),
|
|
36
|
+
tool_calls=item.get("tool_calls"),
|
|
37
|
+
tool_call_validation_result=item.get("tool_call_validation_result"),
|
|
38
|
+
)
|
|
39
|
+
elif isinstance(item["lm_output"], LMOutput):
|
|
40
|
+
lm_output = item["lm_output"]
|
|
41
|
+
else:
|
|
42
|
+
msg = f"Invalid type for lm_output: {type(item['lm_output'])}"
|
|
43
|
+
raise TypeError(msg)
|
|
44
|
+
|
|
45
|
+
lm_output_list.append(lm_output)
|
|
28
46
|
references_list.append(item["references"])
|
|
29
47
|
|
|
30
48
|
if eval_dataset is not None:
|
|
@@ -46,6 +46,17 @@ class LMOutput:
|
|
|
46
46
|
msg = "Both `text` and `tool_calls` are empty."
|
|
47
47
|
logger.warning(msg)
|
|
48
48
|
|
|
49
|
+
def __str__(self) -> str:
|
|
50
|
+
"""Return the `text` attribute as a string for backward compatibility.
|
|
51
|
+
|
|
52
|
+
Before `LMOutput` was introduced, outputs were saved as plain strings.
|
|
53
|
+
Therefore, several existing configs using LLM-judge metrics such as `ChatLLMScore` assume
|
|
54
|
+
that `lm_output` is a plain string. By implementing `__str__`, an `LMOutput`
|
|
55
|
+
instance embedded in a Jinja2 template renders as its `text` attribute,
|
|
56
|
+
preserving that contract.
|
|
57
|
+
"""
|
|
58
|
+
return self.text if self.text is not None else ""
|
|
59
|
+
|
|
49
60
|
|
|
50
61
|
class LanguageModel:
|
|
51
62
|
"""LanguageModel is what you want to evaluate with this library.
|
|
@@ -226,7 +237,10 @@ class LanguageModel:
|
|
|
226
237
|
# Post-process the generated text
|
|
227
238
|
if self.string_processors:
|
|
228
239
|
for lm_output in lm_outputs:
|
|
229
|
-
|
|
240
|
+
# If a subclass preprocesses the output (e.g., via ToolParser or ReasoningParser),
|
|
241
|
+
# lm_output.raw_text should already hold the unprocessed text.
|
|
242
|
+
if lm_output.raw_text is None:
|
|
243
|
+
lm_output.raw_text = lm_output.text
|
|
230
244
|
for string_processor in self.string_processors:
|
|
231
245
|
lm_output.text = string_processor(lm_output.text)
|
|
232
246
|
|
|
@@ -13,6 +13,7 @@ import transformers
|
|
|
13
13
|
from loguru import logger
|
|
14
14
|
from transformers import AutoModelForCausalLM, AutoTokenizer, BatchEncoding, PreTrainedModel, PreTrainedTokenizer
|
|
15
15
|
|
|
16
|
+
from flexeval.core.reasoning_parser.base import ReasoningParser
|
|
16
17
|
from flexeval.core.string_processor import StringProcessor
|
|
17
18
|
from flexeval.core.tool_parser.base import ToolParser
|
|
18
19
|
from flexeval.utils.hf_utils import get_default_model_kwargs
|
|
@@ -186,8 +187,13 @@ class HuggingFaceLM(LanguageModel):
|
|
|
186
187
|
If this value is set to less than or equal to the model's capacity and the input exceeds it,
|
|
187
188
|
an empty string is returned instead of raising an error.
|
|
188
189
|
If set to “default”, the value will be automatically determined when possible.
|
|
190
|
+
reasoning_parser: A ReasoningParser object to extract the reasoning_text from the model's output.
|
|
189
191
|
tool_parser: A ToolParser object to extract the tool_calls from the model's output.
|
|
190
192
|
tools: Default tools to use in chat responses when no tools are explicitly provided.
|
|
193
|
+
prefix_str_for_chat: A string to prepend to the assistant's response in chat generation.
|
|
194
|
+
This string is appended to the prompt (after `apply_chat_template`) so that the model
|
|
195
|
+
continues generation from it. The same string is also prepended to the final output
|
|
196
|
+
so that the returned text includes the forced prefix.
|
|
191
197
|
"""
|
|
192
198
|
|
|
193
199
|
def __init__(
|
|
@@ -206,8 +212,10 @@ class HuggingFaceLM(LanguageModel):
|
|
|
206
212
|
default_gen_kwargs: dict[str, Any] | None = None,
|
|
207
213
|
string_processors: StringProcessor | list[StringProcessor] | None = None,
|
|
208
214
|
model_limit_tokens: int | None | Literal["default"] = "default",
|
|
215
|
+
reasoning_parser: ReasoningParser | None = None,
|
|
209
216
|
tool_parser: ToolParser | None = None,
|
|
210
217
|
tools: list[dict[str, Any]] | None = None,
|
|
218
|
+
prefix_str_for_chat: str = "",
|
|
211
219
|
) -> None:
|
|
212
220
|
super().__init__(string_processors=string_processors, tools=tools)
|
|
213
221
|
self._model_name_or_path = model
|
|
@@ -225,7 +233,9 @@ class HuggingFaceLM(LanguageModel):
|
|
|
225
233
|
self.load_peft = load_peft
|
|
226
234
|
self.amp_dtype = amp_dtype
|
|
227
235
|
self.model_limit_tokens = model_limit_tokens
|
|
236
|
+
self.reasoning_parser = reasoning_parser
|
|
228
237
|
self.tool_parser = tool_parser
|
|
238
|
+
self.prefix_str_for_chat = prefix_str_for_chat
|
|
229
239
|
logger.info(f"amp_dtype: {amp_dtype}")
|
|
230
240
|
logger.info(f"random seed: {random_seed}")
|
|
231
241
|
transformers.set_seed(random_seed)
|
|
@@ -411,18 +421,30 @@ class HuggingFaceLM(LanguageModel):
|
|
|
411
421
|
chat_template=self.custom_chat_template,
|
|
412
422
|
**self.chat_template_kwargs,
|
|
413
423
|
)
|
|
424
|
+
+ self.prefix_str_for_chat
|
|
414
425
|
for chat_messages, tools in zip(chat_messages_list, tools_list)
|
|
415
426
|
]
|
|
416
427
|
lm_outputs = self._batch_complete_text(chat_messages_as_string, **kwargs)
|
|
417
|
-
|
|
418
|
-
|
|
419
|
-
|
|
420
|
-
|
|
421
|
-
|
|
428
|
+
|
|
429
|
+
for lm_output, tools in zip(lm_outputs, tools_list):
|
|
430
|
+
if lm_output.text is None:
|
|
431
|
+
continue
|
|
432
|
+
|
|
433
|
+
lm_output.text = self.prefix_str_for_chat + lm_output.text
|
|
434
|
+
|
|
435
|
+
if self.reasoning_parser:
|
|
436
|
+
if lm_output.raw_text is None:
|
|
437
|
+
lm_output.raw_text = lm_output.text
|
|
438
|
+
reasoning = self.reasoning_parser(lm_output.text)
|
|
439
|
+
lm_output.text = reasoning.text
|
|
440
|
+
lm_output.reasoning_text = reasoning.reasoning_text
|
|
441
|
+
|
|
442
|
+
if self.tool_parser and tools is not None:
|
|
422
443
|
parsed_tool_calling_message = self.tool_parser(lm_output.text)
|
|
423
444
|
lm_output.tool_calls = parsed_tool_calling_message.tool_call_dicts
|
|
424
|
-
lm_output.raw_text = parsed_tool_calling_message.raw_text
|
|
425
445
|
lm_output.text = parsed_tool_calling_message.text
|
|
446
|
+
if lm_output.raw_text is None:
|
|
447
|
+
lm_output.raw_text = parsed_tool_calling_message.raw_text
|
|
426
448
|
lm_output.tool_call_validation_result = parsed_tool_calling_message.validation_result
|
|
427
449
|
|
|
428
450
|
return lm_outputs
|
|
@@ -8,6 +8,7 @@ import torch
|
|
|
8
8
|
from loguru import logger
|
|
9
9
|
from transformers import AutoTokenizer, PreTrainedTokenizer
|
|
10
10
|
|
|
11
|
+
from flexeval.core.reasoning_parser.base import ReasoningParser
|
|
11
12
|
from flexeval.core.string_processor import StringProcessor
|
|
12
13
|
from flexeval.core.tool_parser.base import ToolParser
|
|
13
14
|
|
|
@@ -94,8 +95,13 @@ class VLLM(LanguageModel):
|
|
|
94
95
|
If this value is set to less than or equal to the model's capacity and the input exceeds it,
|
|
95
96
|
an empty string is returned instead of raising an error.
|
|
96
97
|
If set to “default”, the value will be automatically determined when possible.
|
|
98
|
+
reasoning_parser: A ReasoningParser object to extract the reasoning and content from the model's output.
|
|
97
99
|
tool_parser: A ToolParser object to extract the tool_calls from the model's output.
|
|
98
100
|
tools: Default tools to use in chat responses when no tools are explicitly provided.
|
|
101
|
+
prefix_str_for_chat: A string to prepend to the assistant's response in chat generation.
|
|
102
|
+
This string is appended to the prompt (after `apply_chat_template`) so that the model
|
|
103
|
+
continues generation from it. The same string is also prepended to the final output
|
|
104
|
+
so that the returned text includes the forced prefix.
|
|
99
105
|
"""
|
|
100
106
|
|
|
101
107
|
def __init__(
|
|
@@ -111,8 +117,10 @@ class VLLM(LanguageModel):
|
|
|
111
117
|
default_gen_kwargs: dict[str, Any] | None = None,
|
|
112
118
|
string_processors: StringProcessor | list[StringProcessor] | None = None,
|
|
113
119
|
model_limit_tokens: int | None | Literal["default"] = "default",
|
|
120
|
+
reasoning_parser: ReasoningParser | None = None,
|
|
114
121
|
tool_parser: ToolParser | None = None,
|
|
115
122
|
tools: list[dict[str, Any]] | None = None,
|
|
123
|
+
prefix_str_for_chat: str = "",
|
|
116
124
|
) -> None:
|
|
117
125
|
super().__init__(string_processors=string_processors, tools=tools)
|
|
118
126
|
self.model_name = model
|
|
@@ -141,6 +149,8 @@ class VLLM(LanguageModel):
|
|
|
141
149
|
self.llm: LLM | None = None
|
|
142
150
|
self.model_limit_tokens = model_limit_tokens
|
|
143
151
|
self.tool_parser = tool_parser
|
|
152
|
+
self.reasoning_parser = reasoning_parser
|
|
153
|
+
self.prefix_str_for_chat = prefix_str_for_chat
|
|
144
154
|
|
|
145
155
|
@staticmethod
|
|
146
156
|
def load_model(method: Callable) -> Callable:
|
|
@@ -259,17 +269,30 @@ class VLLM(LanguageModel):
|
|
|
259
269
|
chat_template=self.custom_chat_template,
|
|
260
270
|
**self.chat_template_kwargs,
|
|
261
271
|
)
|
|
272
|
+
+ self.prefix_str_for_chat
|
|
262
273
|
for chat_messages, tools in zip(chat_messages_list, tools_list)
|
|
263
274
|
]
|
|
264
275
|
lm_outputs = self._batch_complete_text(chat_messages_as_string, **kwargs)
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
|
|
276
|
+
|
|
277
|
+
for lm_output, tools in zip(lm_outputs, tools_list):
|
|
278
|
+
if lm_output.text is None:
|
|
279
|
+
continue
|
|
280
|
+
|
|
281
|
+
lm_output.text = self.prefix_str_for_chat + lm_output.text
|
|
282
|
+
|
|
283
|
+
if self.reasoning_parser:
|
|
284
|
+
if lm_output.raw_text is None:
|
|
285
|
+
lm_output.raw_text = lm_output.text
|
|
286
|
+
reasoning = self.reasoning_parser(lm_output.text)
|
|
287
|
+
lm_output.text = reasoning.text
|
|
288
|
+
lm_output.reasoning_text = reasoning.reasoning_text
|
|
289
|
+
|
|
290
|
+
if self.tool_parser and tools is not None:
|
|
269
291
|
parsed_tool_calling_message = self.tool_parser(lm_output.text)
|
|
270
292
|
lm_output.tool_calls = parsed_tool_calling_message.tool_call_dicts
|
|
271
|
-
lm_output.raw_text = parsed_tool_calling_message.raw_text
|
|
272
293
|
lm_output.text = parsed_tool_calling_message.text
|
|
294
|
+
if lm_output.raw_text is None:
|
|
295
|
+
lm_output.raw_text = parsed_tool_calling_message.raw_text
|
|
273
296
|
lm_output.tool_call_validation_result = parsed_tool_calling_message.validation_result
|
|
274
297
|
|
|
275
298
|
return lm_outputs
|
|
@@ -15,7 +15,7 @@ from flexeval.core.utils.data_util import batch_iter
|
|
|
15
15
|
|
|
16
16
|
from .base import Metric, MetricResult
|
|
17
17
|
from .llm_score import prepare_chat_input_for_evaluator, prepare_text_input_for_evaluator
|
|
18
|
-
from .utils import
|
|
18
|
+
from .utils import validate_inputs
|
|
19
19
|
|
|
20
20
|
|
|
21
21
|
def calculate_weighted_average(
|
|
@@ -238,8 +238,6 @@ class LLMGEvalScore(Metric):
|
|
|
238
238
|
|
|
239
239
|
validate_inputs(lm_outputs, references_list, extra_info_list)
|
|
240
240
|
|
|
241
|
-
lm_outputs = extract_text_from_outputs(lm_outputs)
|
|
242
|
-
|
|
243
241
|
# Compute metrics
|
|
244
242
|
evaluator_input_list: list[str] = prepare_text_input_for_evaluator(
|
|
245
243
|
lm_outputs, references_list, extra_info_list, self.prompt_template
|
|
@@ -411,8 +409,6 @@ class ChatLLMGEvalScore(Metric):
|
|
|
411
409
|
if references_list is None:
|
|
412
410
|
references_list = [[] for _ in lm_outputs]
|
|
413
411
|
|
|
414
|
-
lm_outputs = extract_text_from_outputs(lm_outputs)
|
|
415
|
-
|
|
416
412
|
# Compute metrics
|
|
417
413
|
evaluator_input_list = prepare_chat_input_for_evaluator(
|
|
418
414
|
lm_outputs, references_list, extra_info_list, self.prompt_template, self.system_message
|
|
@@ -14,7 +14,7 @@ from flexeval.core.metric.llm_score import (
|
|
|
14
14
|
from flexeval.core.prompt_template import PromptTemplate
|
|
15
15
|
|
|
16
16
|
from .base import Metric, MetricResult
|
|
17
|
-
from .utils import
|
|
17
|
+
from .utils import validate_inputs
|
|
18
18
|
|
|
19
19
|
|
|
20
20
|
def parse_label_from_evaluator_output(evaluator_output: str, label_names: list[str]) -> str | None:
|
|
@@ -180,9 +180,6 @@ class LLMLabel(Metric):
|
|
|
180
180
|
|
|
181
181
|
validate_inputs(lm_outputs, references_list, extra_info_list)
|
|
182
182
|
|
|
183
|
-
# Extract text from LMOutput objects
|
|
184
|
-
lm_outputs = extract_text_from_outputs(lm_outputs)
|
|
185
|
-
|
|
186
183
|
# Compute metrics
|
|
187
184
|
evaluator_input_list: list[str] = prepare_text_input_for_evaluator(
|
|
188
185
|
lm_outputs, references_list, extra_info_list, self.prompt_template
|
|
@@ -331,9 +328,6 @@ class ChatLLMLabel(Metric):
|
|
|
331
328
|
|
|
332
329
|
validate_inputs(lm_outputs, references_list, extra_info_list)
|
|
333
330
|
|
|
334
|
-
# Extract text from LMOutput objects
|
|
335
|
-
lm_outputs = extract_text_from_outputs(lm_outputs)
|
|
336
|
-
|
|
337
331
|
# Compute metrics
|
|
338
332
|
evaluator_input_list = prepare_chat_input_for_evaluator(
|
|
339
333
|
lm_outputs, references_list, extra_info_list, self.prompt_template, self.system_message
|
|
@@ -11,7 +11,7 @@ from flexeval.core.prompt_template import PromptTemplate
|
|
|
11
11
|
from flexeval.core.utils.data_util import batch_iter
|
|
12
12
|
|
|
13
13
|
from .base import Metric, MetricResult
|
|
14
|
-
from .utils import
|
|
14
|
+
from .utils import validate_inputs
|
|
15
15
|
|
|
16
16
|
|
|
17
17
|
def parse_score_from_evaluator_output(
|
|
@@ -71,7 +71,7 @@ def summarize_evaluator_scores(
|
|
|
71
71
|
|
|
72
72
|
|
|
73
73
|
def prepare_text_input_for_evaluator(
|
|
74
|
-
lm_outputs: list[str],
|
|
74
|
+
lm_outputs: list[LMOutput | str],
|
|
75
75
|
references_list: list[list[str]],
|
|
76
76
|
extra_info_list: list[dict[str, str]],
|
|
77
77
|
prompt_template: PromptTemplate,
|
|
@@ -80,6 +80,10 @@ def prepare_text_input_for_evaluator(
|
|
|
80
80
|
by integrating the extra_info, the model outputs, and the prompt template for evaluator.
|
|
81
81
|
"""
|
|
82
82
|
|
|
83
|
+
lm_outputs: list[LMOutput] = [
|
|
84
|
+
LMOutput(text=lm_output) if isinstance(lm_output, str) else lm_output for lm_output in lm_outputs
|
|
85
|
+
]
|
|
86
|
+
|
|
83
87
|
evaluator_input_list: list[str] = []
|
|
84
88
|
for lm_output, extra_info, references in zip(
|
|
85
89
|
lm_outputs,
|
|
@@ -97,7 +101,7 @@ def prepare_text_input_for_evaluator(
|
|
|
97
101
|
|
|
98
102
|
|
|
99
103
|
def prepare_chat_input_for_evaluator(
|
|
100
|
-
lm_outputs: list[str],
|
|
104
|
+
lm_outputs: list[LMOutput | str],
|
|
101
105
|
references_list: list[list[str]],
|
|
102
106
|
extra_info_list: list[dict[str, str]],
|
|
103
107
|
prompt_template: PromptTemplate,
|
|
@@ -107,6 +111,10 @@ def prepare_chat_input_for_evaluator(
|
|
|
107
111
|
by integrating the extra_info, the model outputs, and the prompt template for evaluator.
|
|
108
112
|
"""
|
|
109
113
|
|
|
114
|
+
lm_outputs: list[LMOutput] = [
|
|
115
|
+
LMOutput(text=lm_output) if isinstance(lm_output, str) else lm_output for lm_output in lm_outputs
|
|
116
|
+
]
|
|
117
|
+
|
|
110
118
|
evaluator_input_list: list[list[dict[str, str]]] = []
|
|
111
119
|
for lm_output, extra_info, references in zip(
|
|
112
120
|
lm_outputs,
|
|
@@ -246,9 +254,6 @@ class LLMScore(Metric):
|
|
|
246
254
|
|
|
247
255
|
validate_inputs(lm_outputs, references_list, extra_info_list)
|
|
248
256
|
|
|
249
|
-
# Extract text from LMOutput objects
|
|
250
|
-
lm_outputs = extract_text_from_outputs(lm_outputs)
|
|
251
|
-
|
|
252
257
|
# Compute metrics
|
|
253
258
|
evaluator_input_list: list[str] = prepare_text_input_for_evaluator(
|
|
254
259
|
lm_outputs, references_list, extra_info_list, self.prompt_template
|
|
@@ -374,9 +379,6 @@ class ChatLLMScore(Metric):
|
|
|
374
379
|
if references_list is None:
|
|
375
380
|
references_list = [[] for _ in lm_outputs]
|
|
376
381
|
|
|
377
|
-
# Extract text from LMOutput objects
|
|
378
|
-
lm_outputs = extract_text_from_outputs(lm_outputs)
|
|
379
|
-
|
|
380
382
|
# Compute metrics
|
|
381
383
|
evaluator_input_list = prepare_chat_input_for_evaluator(
|
|
382
384
|
lm_outputs, references_list, extra_info_list, self.prompt_template, self.system_message
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from abc import ABC, abstractmethod
|
|
4
|
+
from dataclasses import dataclass
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
@dataclass
|
|
8
|
+
class Reasoning:
|
|
9
|
+
text: str | None
|
|
10
|
+
reasoning_text: str | None
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class ReasoningParser(ABC):
|
|
14
|
+
"""Base class for parsing raw LLM output text into reasoning and content parts.
|
|
15
|
+
|
|
16
|
+
Subclasses must implement `__call__`, which receives the raw text and returns
|
|
17
|
+
an instance of `Reasoning` class.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
@abstractmethod
|
|
21
|
+
def __call__(self, raw_text: str) -> Reasoning:
|
|
22
|
+
"""Parse raw LLM output and return the reasoning and content parts.
|
|
23
|
+
|
|
24
|
+
Args:
|
|
25
|
+
raw_text: Raw output string produced by the language model.
|
|
26
|
+
|
|
27
|
+
Returns:
|
|
28
|
+
An instance of `Reasoning` class.
|
|
29
|
+
"""
|
|
30
|
+
raise NotImplementedError
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
import re
|
|
2
|
+
|
|
3
|
+
from flexeval.core.reasoning_parser.base import Reasoning, ReasoningParser
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
class UnifiedRegexReasoningParser(ReasoningParser):
|
|
7
|
+
"""Extracts reasoning and content using a single regex with named groups.
|
|
8
|
+
|
|
9
|
+
The pattern must contain named groups ``(?P<reasoning_content>...)`` and/or
|
|
10
|
+
``(?P<content>...)`` to populate the corresponding fields of :class:`Reasoning`.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
def __init__(self, pattern: str) -> None:
|
|
14
|
+
super().__init__()
|
|
15
|
+
self.pattern = pattern
|
|
16
|
+
|
|
17
|
+
def __call__(self, raw_text: str) -> Reasoning:
|
|
18
|
+
m = re.search(self.pattern, raw_text, re.DOTALL)
|
|
19
|
+
if m is None:
|
|
20
|
+
return Reasoning(text=None, reasoning_text=None)
|
|
21
|
+
gd = m.groupdict()
|
|
22
|
+
return Reasoning(
|
|
23
|
+
text=gd.get("content"),
|
|
24
|
+
reasoning_text=gd.get("reasoning_content"),
|
|
25
|
+
)
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
class SeparatedRegexReasoningParser(ReasoningParser):
|
|
29
|
+
"""Extracts reasoning and content using two independent regex patterns.
|
|
30
|
+
|
|
31
|
+
Each pattern is applied separately. If a pattern has a capture group,
|
|
32
|
+
the last group is used; otherwise the full match is used.
|
|
33
|
+
"""
|
|
34
|
+
|
|
35
|
+
def __init__(
|
|
36
|
+
self,
|
|
37
|
+
pattern_for_content: str,
|
|
38
|
+
pattern_for_reasoning_content: str,
|
|
39
|
+
) -> None:
|
|
40
|
+
super().__init__()
|
|
41
|
+
self.pattern_for_content = re.compile(pattern_for_content, re.DOTALL)
|
|
42
|
+
self.pattern_for_reasoning_content = re.compile(pattern_for_reasoning_content, re.DOTALL)
|
|
43
|
+
|
|
44
|
+
def __call__(self, raw_text: str) -> Reasoning:
|
|
45
|
+
content = None
|
|
46
|
+
reasoning_text = None
|
|
47
|
+
|
|
48
|
+
found_content = self.pattern_for_content.findall(raw_text)
|
|
49
|
+
if found_content:
|
|
50
|
+
content = found_content[-1]
|
|
51
|
+
|
|
52
|
+
found_reasoning = self.pattern_for_reasoning_content.findall(raw_text)
|
|
53
|
+
if found_reasoning:
|
|
54
|
+
reasoning_text = found_reasoning[-1]
|
|
55
|
+
|
|
56
|
+
return Reasoning(text=content, reasoning_text=reasoning_text)
|
|
@@ -26,7 +26,7 @@ class EvalDataLoader(ABC):
|
|
|
26
26
|
A class to load evaluation data.
|
|
27
27
|
The evaluation data should be a list of dictionaries with the following keys:
|
|
28
28
|
- extra_info (dict[str, Any]): A dictionary containing the input data for the task and any other informations.
|
|
29
|
-
- lm_output (str): The output of the language model.
|
|
29
|
+
- lm_output (str|LMOutput): The output of the language model.
|
|
30
30
|
- references (list[str]): A list of reference outputs.
|
|
31
31
|
- extra_info (dict[str, Any]): alias for "extra_info". Older versions used this key.
|
|
32
32
|
"""
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[tool.poetry]
|
|
2
2
|
name = "flexeval"
|
|
3
|
-
version = "0.17.
|
|
3
|
+
version = "0.17.2" # This will be automatically set from git tag by poetry-dynamic-versioning
|
|
4
4
|
description = ""
|
|
5
5
|
authors = ["ryokan-ri <ryokan.ri@sbintuitions.co.jp>"]
|
|
6
6
|
readme = "README.md"
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/chat_dataset/chatbot_bench_datasets/README.md
RENAMED
|
File without changes
|
|
File without changes
|
{flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/chat_dataset/chatbot_bench_datasets/mt-en.jsonl
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{flexeval-0.17.1 → flexeval-0.17.2}/flexeval/core/chat_dataset/chatbot_bench_datasets/mt-ja.jsonl
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|