flexeval 0.17.2__tar.gz → 0.17.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {flexeval-0.17.2 → flexeval-0.17.3}/PKG-INFO +1 -1
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/language_model/hf_lm.py +31 -6
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/language_model/openai_api.py +18 -9
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/language_model/openai_batch_api.py +19 -8
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/language_model/vllm_model.py +6 -2
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/language_model/vllm_serve_lm.py +1 -1
- {flexeval-0.17.2 → flexeval-0.17.3}/pyproject.toml +1 -1
- {flexeval-0.17.2 → flexeval-0.17.3}/LICENSE +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/README.md +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/__init__.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/__init__.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/chat_dataset/__init__.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/chat_dataset/base.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/chat_dataset/chatbot_bench.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/chat_dataset/chatbot_bench_datasets/README.md +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/chat_dataset/chatbot_bench_datasets/mt-en-ref-gpt4.jsonl +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/chat_dataset/chatbot_bench_datasets/mt-en.jsonl +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/chat_dataset/chatbot_bench_datasets/mt-ja-ref-gpt4.jsonl +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/chat_dataset/chatbot_bench_datasets/mt-ja-ref-gpt4o-with-human-annotation.jsonl +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/chat_dataset/chatbot_bench_datasets/mt-ja.jsonl +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/chat_dataset/chatbot_bench_datasets/rakuda-v2-ja.jsonl +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/chat_dataset/chatbot_bench_datasets/vicuna-en-ref-gpt4.jsonl +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/chat_dataset/chatbot_bench_datasets/vicuna-en.jsonl +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/chat_dataset/chatbot_bench_datasets/vicuna-ja-ref-gpt4.jsonl +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/chat_dataset/chatbot_bench_datasets/vicuna-ja.jsonl +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/chat_dataset/openai_messages.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/chat_dataset/sacrebleu_dataset.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/chat_dataset/template_based.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/eval_setups.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/evaluate_chat_response.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/evaluate_from_data.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/evaluate_generation.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/evaluate_multiple_choice.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/evaluate_pairwise.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/evaluate_perplexity.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/evaluate_reward_model.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/few_shot_generator/__init__.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/few_shot_generator/balanced.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/few_shot_generator/base.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/few_shot_generator/fixed.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/few_shot_generator/rand.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/generation_dataset/__init__.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/generation_dataset/base.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/generation_dataset/sacrebleu_dataset.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/generation_dataset/template_based.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/language_model/__init__.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/language_model/base.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/language_model/litellm_api.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/metric/__init__.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/metric/base.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/metric/bleu.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/metric/char_f1.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/metric/code_eval.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/metric/common_prefix_length.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/metric/common_string_length.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/metric/correlation.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/metric/exact_match.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/metric/finish_reason.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/metric/llm_geval_score.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/metric/llm_label.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/metric/llm_score.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/metric/math.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/metric/output_length_stats.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/metric/perspective_api.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/metric/repetition_count.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/metric/repetition_n.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/metric/rouge.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/metric/sari.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/metric/substring_match.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/metric/tool_call.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/metric/utils.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/metric/xer.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/multiple_choice_dataset/__init__.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/multiple_choice_dataset/base.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/multiple_choice_dataset/template_based.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/pairwise_comparison/__init__.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/pairwise_comparison/judge/__init__.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/pairwise_comparison/judge/base.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/pairwise_comparison/judge/llm_judge.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/pairwise_comparison/match.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/pairwise_comparison/match_maker/__init__.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/pairwise_comparison/match_maker/all_combinations.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/pairwise_comparison/match_maker/base.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/pairwise_comparison/match_maker/random_combinations.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/pairwise_comparison/scorer/__init__.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/pairwise_comparison/scorer/base.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/pairwise_comparison/scorer/bradley_terry.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/pairwise_comparison/scorer/win_rate.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/prompt_template/__init__.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/prompt_template/base.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/prompt_template/jinja2.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/reasoning_parser/__init__.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/reasoning_parser/base.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/reasoning_parser/regex_reasoning_parser.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/result_recorder/__init__.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/result_recorder/base.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/result_recorder/local_recorder.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/result_recorder/wandb_recorder.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/reward_bench_dataset/__init__.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/reward_bench_dataset/base.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/reward_bench_dataset/template_based.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/reward_model/__init__.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/reward_model/base.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/reward_model/log_prob.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/reward_model/pairwise_judge_reward_model.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/reward_model/sequence_classification.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/string_processor/__init__.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/string_processor/aio.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/string_processor/base.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/string_processor/last_line.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/string_processor/lower.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/string_processor/mgsm.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/string_processor/nfkc.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/string_processor/regex.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/string_processor/string_strip.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/string_processor/template.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/text_dataset/__init__.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/text_dataset/base.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/text_dataset/hf.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/text_dataset/jsonl.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/tokenizer/__init__.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/tokenizer/base.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/tokenizer/mecab.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/tokenizer/sacrebleu_tokenizer.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/tokenizer/tiktoken_tokenizer.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/tokenizer/transformers_tokenizer.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/tokenizer/whitespace.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/tool_parser/__init__.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/tool_parser/base.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/utils/__init__.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/utils/chat_util.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/utils/data_util.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/utils/jinja2_utils.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/utils/json_util.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/code_chat/mbpp_chat.jsonnet +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/code_generation/jhumaneval.jsonnet +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/code_generation/jhumaneval_tab_indent.jsonnet +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/code_generation/mbpp.jsonnet +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/code_generation/mbpp_tab_indent.jsonnet +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/code_generation/openai_humaneval.jsonnet +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/code_generation/openai_humaneval_tab_indent.jsonnet +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/en_chat/mt-en.jsonnet +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/en_chat/vicuna-en.jsonnet +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/en_generation/babi.jsonnet +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/en_generation/commonsense_qa.jsonnet +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/en_generation/gsm8k.jsonnet +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/en_generation/squad_v1.jsonnet +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/en_generation/trivia_qa.jsonnet +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/en_generation/twitter_sentiment.jsonnet +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/en_multiple_choice/arc_challenge.jsonnet +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/en_multiple_choice/arc_easy.jsonnet +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/en_multiple_choice/commonsense_qa_mc.jsonnet +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/en_multiple_choice/hellaswag.jsonnet +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/en_multiple_choice/openbookqa.jsonnet +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/en_multiple_choice/piqa.jsonnet +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/en_multiple_choice/xwinograd_en.jsonnet +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/en_perplexity/tiny_shakespeare.jsonnet +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/ja_chat/aio_chat.jsonnet +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/ja_chat/elyza_tasks_100.jsonnet +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/ja_chat/mgsm_ja_chat.jsonnet +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/ja_chat/mt-ja.jsonnet +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/ja_chat/rakuda-v2-ja.jsonnet +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/ja_chat/vicuna-ja.jsonnet +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/ja_generation/aio.jsonnet +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/ja_generation/jamcqa.jsonnet +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/ja_generation/jcommonsenseqa.jsonnet +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/ja_generation/jnli.jsonnet +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/ja_generation/jsquad.jsonnet +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/ja_generation/mgsm_ja.jsonnet +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/ja_generation/wrime_pos_neg.jsonnet +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/ja_generation/xlsum_ja.jsonnet +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/ja_multiple_choice/jcommonsenseqa_mc.jsonnet +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/ja_multiple_choice/xwinograd_ja.jsonnet +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/translation/wmt20_en_ja.jsonnet +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/translation/wmt20_ja_en.jsonnet +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/translation_chat/wmt20_en_ja_chat.jsonnet +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/translation_chat/wmt20_ja_en_chat.jsonnet +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/Metric/assistant_eval_en_single_turn.jsonnet +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/Metric/assistant_eval_ja_single_turn.jsonnet +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/Metric/elyza_tasks_100_eval.jsonnet +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/PairwiseJudge/assistant_judge_en_single_turn.jsonnet +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/PairwiseJudge/assistant_judge_ja_single_turn.jsonnet +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/scripts/__init__.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/scripts/common.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/scripts/flexeval_file.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/scripts/flexeval_lm.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/scripts/flexeval_pairwise.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/scripts/flexeval_presets.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/scripts/flexeval_reward.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/utils/__init__.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/utils/hf_utils.py +0 -0
- {flexeval-0.17.2 → flexeval-0.17.3}/flexeval/utils/module_utils.py +0 -0
|
@@ -383,15 +383,36 @@ class HuggingFaceLM(LanguageModel):
|
|
|
383
383
|
|
|
384
384
|
# We strip the input text and stop sequences from the output text.
|
|
385
385
|
lm_outputs: list[LMOutput] = []
|
|
386
|
-
for generated_tensor in generated_tokens:
|
|
386
|
+
for attention_mask, generated_tensor in zip(model_inputs["attention_mask"], generated_tokens):
|
|
387
387
|
input_tensor = generated_tensor[:input_token_length]
|
|
388
388
|
output_tensor = generated_tensor[input_token_length:]
|
|
389
389
|
|
|
390
|
-
|
|
391
|
-
|
|
390
|
+
# Use the attention mask (rather than filtering by `pad_token_id`) to strip the
|
|
391
|
+
# left-padding from the input, since `pad_token_id` may equal `eos_token_id` (the
|
|
392
|
+
# tokenizer has no native pad token), in which case a value-based filter would also
|
|
393
|
+
# strip genuine EOS tokens that are part of the actual prompt content.
|
|
394
|
+
input_tokens = [t for t, m in zip(input_tensor.tolist(), attention_mask.tolist()) if m != 0]
|
|
395
|
+
|
|
396
|
+
# Find where generation actually stopped: the first occurrence of a stop token id in
|
|
397
|
+
# the output. Everything from that point on (the stop token itself and the padding
|
|
398
|
+
# that follows it) is discarded. Note that `pad_token_id` may equal one of the
|
|
399
|
+
# `stop_token_ids` (when the tokenizer has no native pad token), so a value-based pad
|
|
400
|
+
# filter would incorrectly strip the genuine stop token as well, making it impossible
|
|
401
|
+
# to tell "stop" from "length" below.
|
|
402
|
+
stop_token_pos = next(
|
|
403
|
+
(i for i, t in enumerate(output_tensor.tolist()) if t in stop_token_ids),
|
|
404
|
+
None,
|
|
405
|
+
)
|
|
406
|
+
if stop_token_pos is not None:
|
|
407
|
+
finish_reason = "stop"
|
|
408
|
+
output_tokens = output_tensor[:stop_token_pos].tolist()
|
|
409
|
+
else:
|
|
410
|
+
finish_reason = "length"
|
|
411
|
+
# Fall back to filtering by `pad_token_id`, as a safety net for models with a
|
|
412
|
+
# dedicated pad token (distinct from eos) that may still pad the output.
|
|
413
|
+
output_tokens = [t for t in output_tensor.tolist() if t != self.tokenizer.pad_token_id]
|
|
392
414
|
decoded_text = decode_for_lm_continuation(output_tokens, input_tokens, self.tokenizer)
|
|
393
415
|
|
|
394
|
-
finish_reason = "length"
|
|
395
416
|
for stop_seq in stop_sequences:
|
|
396
417
|
idx = decoded_text.find(stop_seq)
|
|
397
418
|
if idx != -1:
|
|
@@ -436,13 +457,17 @@ class HuggingFaceLM(LanguageModel):
|
|
|
436
457
|
if lm_output.raw_text is None:
|
|
437
458
|
lm_output.raw_text = lm_output.text
|
|
438
459
|
reasoning = self.reasoning_parser(lm_output.text)
|
|
439
|
-
|
|
460
|
+
# `reasoning.text` is None when the reasoning pattern fails to match.
|
|
461
|
+
# LMOutput.text must never be None (see LMOutput.__post_init__), so fall back to "".
|
|
462
|
+
lm_output.text = reasoning.text if reasoning.text is not None else ""
|
|
440
463
|
lm_output.reasoning_text = reasoning.reasoning_text
|
|
441
464
|
|
|
442
465
|
if self.tool_parser and tools is not None:
|
|
443
466
|
parsed_tool_calling_message = self.tool_parser(lm_output.text)
|
|
444
467
|
lm_output.tool_calls = parsed_tool_calling_message.tool_call_dicts
|
|
445
|
-
lm_output.text =
|
|
468
|
+
lm_output.text = (
|
|
469
|
+
parsed_tool_calling_message.text if parsed_tool_calling_message.text is not None else ""
|
|
470
|
+
)
|
|
446
471
|
if lm_output.raw_text is None:
|
|
447
472
|
lm_output.raw_text = parsed_tool_calling_message.raw_text
|
|
448
473
|
lm_output.tool_call_validation_result = parsed_tool_calling_message.validation_result
|
|
@@ -226,7 +226,7 @@ class OpenAIChatAPI(LanguageModel):
|
|
|
226
226
|
LMOutput(
|
|
227
227
|
text=res.choices[0].message.content,
|
|
228
228
|
reasoning_text=get_reasoning_text(res.choices[0].message),
|
|
229
|
-
finish_reason=res.choices[0].finish_reason,
|
|
229
|
+
finish_reason="empty" if res is self.empty_response else res.choices[0].finish_reason,
|
|
230
230
|
)
|
|
231
231
|
for res in api_responses
|
|
232
232
|
]
|
|
@@ -246,7 +246,7 @@ class OpenAIChatAPI(LanguageModel):
|
|
|
246
246
|
LMOutput(
|
|
247
247
|
text=res.choices[0].message.content,
|
|
248
248
|
reasoning_text=get_reasoning_text(res.choices[0].message),
|
|
249
|
-
finish_reason=res.choices[0].finish_reason,
|
|
249
|
+
finish_reason="empty" if res is self.empty_response else res.choices[0].finish_reason,
|
|
250
250
|
tool_calls=[tool_call.to_dict() for tool_call in res.choices[0].message.tool_calls]
|
|
251
251
|
if res.choices[0].message.tool_calls
|
|
252
252
|
else None,
|
|
@@ -295,17 +295,21 @@ class OpenAIChatAPI(LanguageModel):
|
|
|
295
295
|
)
|
|
296
296
|
|
|
297
297
|
log_probs = []
|
|
298
|
-
top_logprobs_list = [
|
|
298
|
+
top_logprobs_list = [
|
|
299
|
+
None if res is self.empty_response else res.choices[0].logprobs.content[0].top_logprobs
|
|
300
|
+
for res in api_responses
|
|
301
|
+
]
|
|
299
302
|
for index, prompt in enumerate(prompt_list):
|
|
300
303
|
target_token = response_contents[index]
|
|
301
304
|
index_in_unique = unique_prompt_list.index(prompt)
|
|
302
305
|
|
|
303
|
-
log_prob = None # if target token not in top_logprobs,
|
|
306
|
+
log_prob = None # if target token not in top_logprobs, or the request errored, return None
|
|
304
307
|
top_logprobs = top_logprobs_list[index_in_unique]
|
|
305
|
-
|
|
306
|
-
|
|
307
|
-
|
|
308
|
-
|
|
308
|
+
if top_logprobs is not None:
|
|
309
|
+
for token_logprob in top_logprobs:
|
|
310
|
+
if token_logprob.token == target_token:
|
|
311
|
+
log_prob = token_logprob.logprob
|
|
312
|
+
break
|
|
309
313
|
log_probs.append(log_prob)
|
|
310
314
|
|
|
311
315
|
return log_probs
|
|
@@ -450,7 +454,12 @@ class OpenAICompletionAPI(LanguageModel):
|
|
|
450
454
|
**kwargs,
|
|
451
455
|
)
|
|
452
456
|
|
|
453
|
-
return [
|
|
457
|
+
return [
|
|
458
|
+
LMOutput(text="", finish_reason="empty")
|
|
459
|
+
if res is self.empty_response
|
|
460
|
+
else LMOutput(text=res.choices[0].text, finish_reason=res.choices[0].finish_reason)
|
|
461
|
+
for res in api_responses
|
|
462
|
+
]
|
|
454
463
|
|
|
455
464
|
def __repr__(self) -> str:
|
|
456
465
|
return f"{self.__class__.__name__}(model={self.model})"
|
|
@@ -257,7 +257,12 @@ class OpenAIChatBatchAPI(LanguageModel):
|
|
|
257
257
|
**kwargs,
|
|
258
258
|
)
|
|
259
259
|
return [
|
|
260
|
-
LMOutput(text=
|
|
260
|
+
LMOutput(text="", finish_reason="empty")
|
|
261
|
+
if isinstance(res, str)
|
|
262
|
+
else LMOutput(
|
|
263
|
+
text=res["choices"][0]["message"]["content"],
|
|
264
|
+
finish_reason=res["choices"][0]["finish_reason"],
|
|
265
|
+
)
|
|
261
266
|
for res in api_responses
|
|
262
267
|
]
|
|
263
268
|
|
|
@@ -273,7 +278,9 @@ class OpenAIChatBatchAPI(LanguageModel):
|
|
|
273
278
|
**kwargs,
|
|
274
279
|
)
|
|
275
280
|
return [
|
|
276
|
-
LMOutput(
|
|
281
|
+
LMOutput(text="", finish_reason="empty")
|
|
282
|
+
if isinstance(res, str)
|
|
283
|
+
else LMOutput(
|
|
277
284
|
text=res["choices"][0]["message"]["content"],
|
|
278
285
|
finish_reason=res["choices"][0]["finish_reason"],
|
|
279
286
|
tool_calls=res["choices"][0]["message"].get("tool_calls", None),
|
|
@@ -319,17 +326,21 @@ class OpenAIChatBatchAPI(LanguageModel):
|
|
|
319
326
|
)
|
|
320
327
|
|
|
321
328
|
log_probs = []
|
|
322
|
-
top_logprobs_list = [
|
|
329
|
+
top_logprobs_list = [
|
|
330
|
+
None if isinstance(res, str) else res["choices"][0]["logprobs"]["content"][0]["top_logprobs"]
|
|
331
|
+
for res in api_responses
|
|
332
|
+
]
|
|
323
333
|
for index, prompt in enumerate(prompt_list):
|
|
324
334
|
target_token = response_contents[index]
|
|
325
335
|
index_in_unique = unique_prompt_list.index(prompt)
|
|
326
336
|
|
|
327
|
-
log_prob = None # if target token not in top_logprobs,
|
|
337
|
+
log_prob = None # if target token not in top_logprobs, or the request errored, return None
|
|
328
338
|
top_logprobs = top_logprobs_list[index_in_unique]
|
|
329
|
-
|
|
330
|
-
|
|
331
|
-
|
|
332
|
-
|
|
339
|
+
if top_logprobs is not None:
|
|
340
|
+
for token_logprob in top_logprobs:
|
|
341
|
+
if token_logprob["token"] == target_token:
|
|
342
|
+
log_prob = token_logprob["logprob"]
|
|
343
|
+
break
|
|
333
344
|
log_probs.append(log_prob)
|
|
334
345
|
|
|
335
346
|
return log_probs
|
|
@@ -284,13 +284,17 @@ class VLLM(LanguageModel):
|
|
|
284
284
|
if lm_output.raw_text is None:
|
|
285
285
|
lm_output.raw_text = lm_output.text
|
|
286
286
|
reasoning = self.reasoning_parser(lm_output.text)
|
|
287
|
-
|
|
287
|
+
# `reasoning.text` is None when the reasoning pattern fails to match.
|
|
288
|
+
# LMOutput.text must never be None (see LMOutput.__post_init__), so fall back to "".
|
|
289
|
+
lm_output.text = reasoning.text if reasoning.text is not None else ""
|
|
288
290
|
lm_output.reasoning_text = reasoning.reasoning_text
|
|
289
291
|
|
|
290
292
|
if self.tool_parser and tools is not None:
|
|
291
293
|
parsed_tool_calling_message = self.tool_parser(lm_output.text)
|
|
292
294
|
lm_output.tool_calls = parsed_tool_calling_message.tool_call_dicts
|
|
293
|
-
lm_output.text =
|
|
295
|
+
lm_output.text = (
|
|
296
|
+
parsed_tool_calling_message.text if parsed_tool_calling_message.text is not None else ""
|
|
297
|
+
)
|
|
294
298
|
if lm_output.raw_text is None:
|
|
295
299
|
lm_output.raw_text = parsed_tool_calling_message.raw_text
|
|
296
300
|
lm_output.tool_call_validation_result = parsed_tool_calling_message.validation_result
|
|
@@ -193,7 +193,7 @@ class VLLMServeLM(OpenAIChatAPI):
|
|
|
193
193
|
logging.getLogger("httpx").setLevel(logging.WARNING)
|
|
194
194
|
logging.getLogger("httpcore").setLevel(logging.WARNING)
|
|
195
195
|
model_kwargs = model_kwargs or {}
|
|
196
|
-
if "tensor_parallel_size" not in model_kwargs:
|
|
196
|
+
if "tensor_parallel_size" not in model_kwargs and "tensor-parallel-size" not in model_kwargs:
|
|
197
197
|
model_kwargs["tensor_parallel_size"] = torch.cuda.device_count()
|
|
198
198
|
self.manager = VLLMServerManager(model=model, model_kwargs=model_kwargs, timeout=booting_timeout)
|
|
199
199
|
if api_headers is None:
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[tool.poetry]
|
|
2
2
|
name = "flexeval"
|
|
3
|
-
version = "0.17.
|
|
3
|
+
version = "0.17.3" # This will be automatically set from git tag by poetry-dynamic-versioning
|
|
4
4
|
description = ""
|
|
5
5
|
authors = ["ryokan-ri <ryokan.ri@sbintuitions.co.jp>"]
|
|
6
6
|
readme = "README.md"
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/chat_dataset/chatbot_bench_datasets/README.md
RENAMED
|
File without changes
|
|
File without changes
|
{flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/chat_dataset/chatbot_bench_datasets/mt-en.jsonl
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/chat_dataset/chatbot_bench_datasets/mt-ja.jsonl
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/pairwise_comparison/match_maker/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/pairwise_comparison/scorer/bradley_terry.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/reasoning_parser/regex_reasoning_parser.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{flexeval-0.17.2 → flexeval-0.17.3}/flexeval/core/reward_model/pairwise_judge_reward_model.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/code_chat/mbpp_chat.jsonnet
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/code_generation/mbpp.jsonnet
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/en_chat/vicuna-en.jsonnet
RENAMED
|
File without changes
|
{flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/en_generation/babi.jsonnet
RENAMED
|
File without changes
|
|
File without changes
|
{flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/en_generation/gsm8k.jsonnet
RENAMED
|
File without changes
|
{flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/en_generation/squad_v1.jsonnet
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/ja_chat/aio_chat.jsonnet
RENAMED
|
File without changes
|
|
File without changes
|
{flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/ja_chat/mgsm_ja_chat.jsonnet
RENAMED
|
File without changes
|
|
File without changes
|
{flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/ja_chat/rakuda-v2-ja.jsonnet
RENAMED
|
File without changes
|
{flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/ja_chat/vicuna-ja.jsonnet
RENAMED
|
File without changes
|
{flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/ja_generation/aio.jsonnet
RENAMED
|
File without changes
|
{flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/ja_generation/jamcqa.jsonnet
RENAMED
|
File without changes
|
|
File without changes
|
{flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/ja_generation/jnli.jsonnet
RENAMED
|
File without changes
|
{flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/ja_generation/jsquad.jsonnet
RENAMED
|
File without changes
|
{flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/ja_generation/mgsm_ja.jsonnet
RENAMED
|
File without changes
|
|
File without changes
|
{flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/EvalSetup/ja_generation/xlsum_ja.jsonnet
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{flexeval-0.17.2 → flexeval-0.17.3}/flexeval/preset_configs/Metric/elyza_tasks_100_eval.jsonnet
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|