AutoRAG 0.2.9__tar.gz → 0.2.10__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {autorag-0.2.9 → autorag-0.2.10}/AutoRAG.egg-info/PKG-INFO +1 -1
- {autorag-0.2.9 → autorag-0.2.10}/AutoRAG.egg-info/SOURCES.txt +0 -7
- {autorag-0.2.9 → autorag-0.2.10}/PKG-INFO +1 -1
- autorag-0.2.10/autorag/VERSION +1 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/retrieval/__init__.py +0 -2
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/retrieval/base.py +3 -4
- autorag-0.2.10/autorag/nodes/retrieval/hybrid_cc.py +137 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/retrieval/hybrid_rrf.py +14 -5
- autorag-0.2.10/autorag/nodes/retrieval/run.py +285 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/support.py +0 -2
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/api_spec/autorag.nodes.retrieval.rst +0 -16
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/evaluate_metrics/generation.md +16 -5
- autorag-0.2.10/docs/source/nodes/retrieval/hybrid_cc.md +59 -0
- autorag-0.2.10/docs/source/nodes/retrieval/hybrid_rrf.md +34 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/retrieval/retrieval.md +5 -10
- {autorag-0.2.9 → autorag-0.2.10}/sample_config/compact_local.yaml +1 -7
- autorag-0.2.10/sample_config/compact_openai.yaml +59 -0
- {autorag-0.2.9 → autorag-0.2.10}/sample_config/full.yaml +4 -19
- {autorag-0.2.9 → autorag-0.2.10}/sample_config/simple_ollama.yaml +2 -19
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/retrieval/test_hybrid_base.py +18 -1
- autorag-0.2.10/tests/autorag/nodes/retrieval/test_hybrid_cc.py +54 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/retrieval/test_hybrid_rrf.py +3 -3
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/retrieval/test_run_retrieval_node.py +17 -38
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/test_evaluator.py +9 -13
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/full.yaml +4 -7
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/3/config.yaml +1 -2
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/simple.yaml +1 -2
- autorag-0.2.9/autorag/VERSION +0 -1
- autorag-0.2.9/autorag/nodes/retrieval/hybrid_cc.py +0 -63
- autorag-0.2.9/autorag/nodes/retrieval/hybrid_dbsf.py +0 -70
- autorag-0.2.9/autorag/nodes/retrieval/hybrid_rsf.py +0 -96
- autorag-0.2.9/autorag/nodes/retrieval/run.py +0 -217
- autorag-0.2.9/docs/source/nodes/retrieval/hybrid_cc.md +0 -44
- autorag-0.2.9/docs/source/nodes/retrieval/hybrid_dbsf.md +0 -131
- autorag-0.2.9/docs/source/nodes/retrieval/hybrid_rrf.md +0 -40
- autorag-0.2.9/docs/source/nodes/retrieval/hybrid_rsf.md +0 -134
- autorag-0.2.9/docs/source/optimization/sample_full_config.yaml +0 -78
- autorag-0.2.9/sample_config/compact_openai.yaml +0 -65
- autorag-0.2.9/tests/autorag/nodes/retrieval/test_hybrid_cc.py +0 -33
- autorag-0.2.9/tests/autorag/nodes/retrieval/test_hybrid_dbsf.py +0 -20
- autorag-0.2.9/tests/autorag/nodes/retrieval/test_hybrid_rsf.py +0 -20
- {autorag-0.2.9 → autorag-0.2.10}/.github/FUNDING.yml +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/.github/dependabot.yml +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/.github/workflows/publish.yml +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/.github/workflows/sphinx.yml +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/.github/workflows/test.yml +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/.gitignore +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/AutoRAG.egg-info/dependency_links.txt +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/AutoRAG.egg-info/entry_points.txt +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/AutoRAG.egg-info/requires.txt +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/AutoRAG.egg-info/top_level.txt +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/CODE_OF_CONDUCT.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/CONTRIBUTING.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/LICENSE +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/README.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/__init__.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/cli.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/dashboard.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/data/__init__.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/data/corpus/__init__.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/data/corpus/langchain.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/data/corpus/llama_index.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/data/qacreation/__init__.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/data/qacreation/base.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/data/qacreation/llama_index.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/data/qacreation/llama_index_default_prompt.txt +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/data/qacreation/ragas.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/data/qacreation/simple.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/data/utils/__init__.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/data/utils/util.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/deploy.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/evaluation/__init__.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/evaluation/generation.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/evaluation/metric/__init__.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/evaluation/metric/g_eval_prompts/coh_detailed.txt +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/evaluation/metric/g_eval_prompts/con_detailed.txt +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/evaluation/metric/g_eval_prompts/flu_detailed.txt +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/evaluation/metric/g_eval_prompts/rel_detailed.txt +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/evaluation/metric/generation.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/evaluation/metric/retrieval.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/evaluation/metric/retrieval_contents.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/evaluation/metric/util.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/evaluation/retrieval.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/evaluation/retrieval_contents.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/evaluation/util.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/evaluator.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/node_line.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/__init__.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/generator/__init__.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/generator/base.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/generator/llama_index_llm.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/generator/openai_llm.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/generator/run.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/generator/vllm.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passageaugmenter/__init__.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passageaugmenter/base.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passageaugmenter/pass_passage_augmenter.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passageaugmenter/prev_next_augmenter.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passageaugmenter/run.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagecompressor/__init__.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagecompressor/base.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagecompressor/longllmlingua.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagecompressor/pass_compressor.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagecompressor/refine.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagecompressor/run.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagecompressor/tree_summarize.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagefilter/__init__.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagefilter/base.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagefilter/pass_passage_filter.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagefilter/percentile_cutoff.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagefilter/recency.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagefilter/run.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagefilter/similarity_percentile_cutoff.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagefilter/similarity_threshold_cutoff.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagefilter/threshold_cutoff.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagereranker/__init__.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagereranker/base.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagereranker/cohere.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagereranker/colbert.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagereranker/flag_embedding.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagereranker/flag_embedding_llm.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagereranker/jina.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagereranker/koreranker.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagereranker/monot5.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagereranker/pass_reranker.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagereranker/rankgpt.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagereranker/run.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagereranker/sentence_transformer.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagereranker/tart/__init__.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagereranker/tart/modeling_enc_t5.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagereranker/tart/tart.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagereranker/tart/tokenization_enc_t5.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagereranker/time_reranker.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagereranker/upr.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/promptmaker/__init__.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/promptmaker/base.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/promptmaker/fstring.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/promptmaker/long_context_reorder.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/promptmaker/run.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/promptmaker/window_replacement.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/queryexpansion/__init__.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/queryexpansion/base.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/queryexpansion/hyde.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/queryexpansion/multi_query_expansion.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/queryexpansion/pass_query_expansion.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/queryexpansion/query_decompose.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/queryexpansion/run.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/retrieval/bm25.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/retrieval/vectordb.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/schema/__init__.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/schema/module.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/schema/node.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/strategy.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/utils/__init__.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/utils/preprocess.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/utils/util.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/autorag/web.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/Makefile +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/make.bat +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/requirements.txt +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/CNAME +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/data_creation.png +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/data_folder.png +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/dcg.png +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/f1_score.png +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/map.png +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/mrr.png +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/ndcg.png +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/ndcg_formula.png +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/node_folder.png +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/node_line_folder.png +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/node_line_summary.png +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/node_lines.png +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/node_summary.png +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/normal_distribution.png +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/project_folder_example.png +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/project_folders.png +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/resources_folder.png +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/roadmap/RAG_paradigms.png +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/roadmap/advanced_RAG.png +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/roadmap/cycle.png +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/roadmap/merger.png +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/roadmap/node_line_modular.png +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/roadmap/policy.png +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/samsung_sundae.jpeg +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/score_fusion.png +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/trial_folder.png +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/trial_json.png +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/trial_summary.png +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/web_interface.png +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/api_spec/autorag.data.corpus.rst +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/api_spec/autorag.data.qacreation.rst +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/api_spec/autorag.data.rst +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/api_spec/autorag.data.utils.rst +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/api_spec/autorag.evaluation.metric.rst +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/api_spec/autorag.evaluation.rst +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/api_spec/autorag.nodes.generator.rst +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/api_spec/autorag.nodes.passageaugmenter.rst +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/api_spec/autorag.nodes.passagecompressor.rst +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/api_spec/autorag.nodes.passagefilter.rst +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/api_spec/autorag.nodes.passagereranker.rst +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/api_spec/autorag.nodes.passagereranker.tart.rst +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/api_spec/autorag.nodes.promptmaker.rst +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/api_spec/autorag.nodes.queryexpansion.rst +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/api_spec/autorag.nodes.rst +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/api_spec/autorag.rst +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/api_spec/autorag.schema.rst +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/api_spec/autorag.utils.rst +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/api_spec/modules.rst +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/conf.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/data_creation/data_format.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/data_creation/ragas.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/data_creation/tutorial.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/deploy/api_endpoint.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/deploy/web.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/evaluate_metrics/retrieval.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/evaluate_metrics/retrieval_contents.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/index.rst +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/install.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/local_model.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/generator/generator.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/generator/llama_index_llm.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/generator/openai_llm.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/generator/vllm.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/index.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/passage_augmenter/passage_augmenter.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/passage_augmenter/prev_next_augmenter.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/passage_compressor/longllmlingua.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/passage_compressor/passage_compressor.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/passage_compressor/refine.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/passage_compressor/tree_summarize.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/passage_filter/passage_filter.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/passage_filter/percentile_cutoff.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/passage_filter/recency_filter.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/passage_filter/similarity_percentile_cutoff.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/passage_filter/similarity_threshold_cutoff.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/passage_filter/threshold_cutoff.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/passage_reranker/cohere.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/passage_reranker/colbert.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/passage_reranker/flag_embedding_llm_reranker.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/passage_reranker/flag_embedding_reranker.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/passage_reranker/jina_reranker.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/passage_reranker/koreranker.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/passage_reranker/monot5.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/passage_reranker/passage_reranker.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/passage_reranker/rankgpt.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/passage_reranker/sentence_transformer_reranker.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/passage_reranker/tart.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/passage_reranker/time_reranker.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/passage_reranker/upr.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/prompt_maker/fstring.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/prompt_maker/long_context_reorder.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/prompt_maker/prompt_maker.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/prompt_maker/window_replacement.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/query_expansion/hyde.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/query_expansion/multi_query_expansion.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/query_expansion/query_decompose.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/query_expansion/query_expansion.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/retrieval/bm25.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/retrieval/vectordb.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/optimization/custom_config.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/optimization/folder_structure.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/optimization/optimization.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/optimization/strategies.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/roadmap/modular_rag.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/structure.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/troubleshooting.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/docs/source/tutorial.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/pyproject.toml +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/requirements.txt +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/sample_config/config_korean.yaml +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/sample_config/extracted_sample.yaml +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/sample_config/simple_local.yaml +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/sample_config/simple_openai.yaml +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/sample_dataset/README.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/sample_dataset/eli5/load_eli5_dataset.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/sample_dataset/hotpotqa/load_hotpotqa_dataset.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/sample_dataset/msmarco/load_msmarco_dataset.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/sample_dataset/triviaqa/load_triviaqa_dataset.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/setup.cfg +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/data/corpus/test_base.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/data/corpus/test_langchain.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/data/corpus/test_llama_index_corpus.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/data/qacreation/test_base_qacreation.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/data/qacreation/test_llama_index_qacreation.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/data/qacreation/test_ragas_qa_creation.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/data/qacreation/test_simple.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/evaluate/metric/test_generation_metric.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/evaluate/metric/test_retrieval_contents_metric.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/evaluate/metric/test_retrieval_metric.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/evaluate/test_evaluate_util.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/evaluate/test_generation_evaluate.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/evaluate/test_retrieval_contents_evaluate.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/evaluate/test_retrieval_evaluate.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/generator/test_generator_base.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/generator/test_llama_index_llm.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/generator/test_openai.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/generator/test_run_generator_node.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/generator/test_vllm.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passageaugmenter/test_base_passage_augmenter.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passageaugmenter/test_pass_passage_augmenter.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passageaugmenter/test_prev_next_augmenter.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passageaugmenter/test_run_passage_augmenter.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagecompressor/test_base_passage_compressor.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagecompressor/test_longllmlingua.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagecompressor/test_pass_compressor.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagecompressor/test_refine.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagecompressor/test_run_passage_compressor_node.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagecompressor/test_tree_summarize.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagefilter/test_pass_passage_filter.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagefilter/test_passage_filter_base.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagefilter/test_passage_filter_run.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagefilter/test_percentile_cutoff.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagefilter/test_recency_filter.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagefilter/test_similarity_percentile_cutoff.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagefilter/test_similarity_threshold_cutoff.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagefilter/test_threshold_cutoff.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagereranker/test_cohere_reranker.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagereranker/test_colbert_reranker.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagereranker/test_flag_embedding.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagereranker/test_flag_embedding_llm.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagereranker/test_jina_reranker.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagereranker/test_koreranker.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagereranker/test_monot5.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagereranker/test_pass_reranker.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagereranker/test_passage_reranker_base.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagereranker/test_passage_reranker_run.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagereranker/test_rankgpt.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagereranker/test_sentence_transformer.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagereranker/test_tart.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagereranker/test_time_reranker.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagereranker/test_upr.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/promptmaker/test_fstring.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/promptmaker/test_long_context_reorder.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/promptmaker/test_prompt_maker_base.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/promptmaker/test_prompt_maker_run.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/promptmaker/test_window_replacement.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/queryexpansion/test_hyde.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/queryexpansion/test_multi_query_expansion.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/queryexpansion/test_pass_query_expansion.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/queryexpansion/test_query_decompose.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/queryexpansion/test_query_expansion_base.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/queryexpansion/test_query_expansion_run.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/retrieval/test_bm25.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/retrieval/test_retrieval_base.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/retrieval/test_vectordb.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/schema/test_module_schema.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/schema/test_node_schema.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/test_cli.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/test_dashboard.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/test_deploy.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/test_strategy.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/test_support.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/test_web.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/utils/test_preprocess.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/utils/test_util.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/conftest.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/delete_tests.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/mock.py +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/requirements.txt +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/README.md +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/corpus_data_sample.parquet +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/data_creation/raw_dir/sample1.txt +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/data_creation/raw_dir/sample2.txt +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/data_creation/raw_dir/sample3.txt +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/qa_data_sample.parquet +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/qa_gen_prompts/prompt1.txt +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/qa_gen_prompts/prompt2.txt +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/qa_gen_prompts/prompt3.txt +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/qa_test_data_sample.parquet +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/config.yaml +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/post_retrieve_node_line/generator/0.parquet +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/post_retrieve_node_line/generator/1.parquet +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/post_retrieve_node_line/generator/2.parquet +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/post_retrieve_node_line/generator/3.parquet +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/post_retrieve_node_line/generator/4.parquet +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/post_retrieve_node_line/generator/5.parquet +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/post_retrieve_node_line/generator/best_5.parquet +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/post_retrieve_node_line/generator/summary.csv +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/post_retrieve_node_line/prompt_maker/0.parquet +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/post_retrieve_node_line/prompt_maker/1.parquet +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/post_retrieve_node_line/prompt_maker/best_0.parquet +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/post_retrieve_node_line/prompt_maker/summary.csv +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/post_retrieve_node_line/summary.csv +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/pre_retrieve_node_line/query_expansion/0.parquet +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/pre_retrieve_node_line/query_expansion/1.parquet +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/pre_retrieve_node_line/query_expansion/2.parquet +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/pre_retrieve_node_line/query_expansion/best_0.parquet +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/pre_retrieve_node_line/query_expansion/summary.csv +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/pre_retrieve_node_line/summary.csv +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/retrieve_node_line/passage_compressor/0.parquet +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/retrieve_node_line/passage_compressor/best_0.parquet +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/retrieve_node_line/passage_compressor/summary.csv +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/retrieve_node_line/passage_reranker/0.parquet +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/retrieve_node_line/passage_reranker/1.parquet +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/retrieve_node_line/passage_reranker/best_0.parquet +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/retrieve_node_line/passage_reranker/summary.csv +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/retrieve_node_line/retrieval/0.parquet +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/retrieve_node_line/retrieval/1.parquet +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/retrieve_node_line/retrieval/best_0.parquet +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/retrieve_node_line/retrieval/summary.csv +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/retrieve_node_line/summary.csv +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/summary.csv +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/1/config.yaml +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/1/pre_retrieve_node_line/query_expansion/0.parquet +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/1/pre_retrieve_node_line/query_expansion/1.parquet +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/1/pre_retrieve_node_line/query_expansion/2.parquet +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/1/pre_retrieve_node_line/query_expansion/best_0.parquet +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/1/pre_retrieve_node_line/query_expansion/summary.csv +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/1/pre_retrieve_node_line/summary.csv +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/1/retrieve_node_line/passage_filter/0.parquet +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/1/retrieve_node_line/passage_reranker/0.parquet +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/1/retrieve_node_line/passage_reranker/1.parquet +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/1/retrieve_node_line/passage_reranker/best_0.parquet +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/1/retrieve_node_line/passage_reranker/summary.csv +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/1/retrieve_node_line/retrieval/0.parquet +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/1/retrieve_node_line/retrieval/1.parquet +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/1/retrieve_node_line/retrieval/best_0.parquet +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/1/retrieve_node_line/retrieval/summary.csv +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/2/config.yaml +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/2/post_retrieve_node_line/prompt_maker/0.parquet +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/2/pre_retrieve_node_line/query_expansion/0.parquet +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/2/pre_retrieve_node_line/query_expansion/1.parquet +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/2/pre_retrieve_node_line/query_expansion/2.parquet +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/2/pre_retrieve_node_line/query_expansion/best_0.parquet +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/2/pre_retrieve_node_line/query_expansion/summary.csv +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/2/pre_retrieve_node_line/summary.csv +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/2/retrieve_node_line/passage_compressor/0.parquet +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/2/retrieve_node_line/passage_compressor/best_0.parquet +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/2/retrieve_node_line/passage_compressor/summary.csv +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/2/retrieve_node_line/passage_reranker/0.parquet +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/2/retrieve_node_line/passage_reranker/1.parquet +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/2/retrieve_node_line/passage_reranker/best_0.parquet +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/2/retrieve_node_line/passage_reranker/summary.csv +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/2/retrieve_node_line/retrieval/0.parquet +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/2/retrieve_node_line/retrieval/1.parquet +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/2/retrieve_node_line/retrieval/best_0.parquet +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/2/retrieve_node_line/retrieval/summary.csv +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/2/retrieve_node_line/summary.csv +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/best.yaml +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/data/corpus.parquet +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/data/qa.parquet +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/resources/bm25_porter_stemmer.pkl +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/resources/chroma/6595d304-5270-4d9f-a267-c191366804ce/data_level0.bin +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/resources/chroma/6595d304-5270-4d9f-a267-c191366804ce/header.bin +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/resources/chroma/6595d304-5270-4d9f-a267-c191366804ce/length.bin +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/resources/chroma/6595d304-5270-4d9f-a267-c191366804ce/link_lists.bin +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/resources/chroma/chroma.sqlite3 +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/trial.json +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/sample_contents_nqa.csv +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/sample_project/data/corpus.parquet +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/sample_project/data/qa.parquet +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/sample_project/resources/bm25_gpt2.pkl +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/sample_project/resources/bm25_porter_stemmer.pkl +0 -0
- {autorag-0.2.9 → autorag-0.2.10}/tests/resources/test_bm25_retrieval.pkl +0 -0
|
@@ -116,9 +116,7 @@ autorag/nodes/retrieval/__init__.py
|
|
|
116
116
|
autorag/nodes/retrieval/base.py
|
|
117
117
|
autorag/nodes/retrieval/bm25.py
|
|
118
118
|
autorag/nodes/retrieval/hybrid_cc.py
|
|
119
|
-
autorag/nodes/retrieval/hybrid_dbsf.py
|
|
120
119
|
autorag/nodes/retrieval/hybrid_rrf.py
|
|
121
|
-
autorag/nodes/retrieval/hybrid_rsf.py
|
|
122
120
|
autorag/nodes/retrieval/run.py
|
|
123
121
|
autorag/nodes/retrieval/vectordb.py
|
|
124
122
|
autorag/schema/__init__.py
|
|
@@ -235,15 +233,12 @@ docs/source/nodes/query_expansion/query_decompose.md
|
|
|
235
233
|
docs/source/nodes/query_expansion/query_expansion.md
|
|
236
234
|
docs/source/nodes/retrieval/bm25.md
|
|
237
235
|
docs/source/nodes/retrieval/hybrid_cc.md
|
|
238
|
-
docs/source/nodes/retrieval/hybrid_dbsf.md
|
|
239
236
|
docs/source/nodes/retrieval/hybrid_rrf.md
|
|
240
|
-
docs/source/nodes/retrieval/hybrid_rsf.md
|
|
241
237
|
docs/source/nodes/retrieval/retrieval.md
|
|
242
238
|
docs/source/nodes/retrieval/vectordb.md
|
|
243
239
|
docs/source/optimization/custom_config.md
|
|
244
240
|
docs/source/optimization/folder_structure.md
|
|
245
241
|
docs/source/optimization/optimization.md
|
|
246
|
-
docs/source/optimization/sample_full_config.yaml
|
|
247
242
|
docs/source/optimization/strategies.md
|
|
248
243
|
docs/source/roadmap/modular_rag.md
|
|
249
244
|
sample_config/compact_local.yaml
|
|
@@ -336,9 +331,7 @@ tests/autorag/nodes/queryexpansion/test_query_expansion_run.py
|
|
|
336
331
|
tests/autorag/nodes/retrieval/test_bm25.py
|
|
337
332
|
tests/autorag/nodes/retrieval/test_hybrid_base.py
|
|
338
333
|
tests/autorag/nodes/retrieval/test_hybrid_cc.py
|
|
339
|
-
tests/autorag/nodes/retrieval/test_hybrid_dbsf.py
|
|
340
334
|
tests/autorag/nodes/retrieval/test_hybrid_rrf.py
|
|
341
|
-
tests/autorag/nodes/retrieval/test_hybrid_rsf.py
|
|
342
335
|
tests/autorag/nodes/retrieval/test_retrieval_base.py
|
|
343
336
|
tests/autorag/nodes/retrieval/test_run_retrieval_node.py
|
|
344
337
|
tests/autorag/nodes/retrieval/test_vectordb.py
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
0.2.10
|
|
@@ -32,7 +32,6 @@ def retrieval_node(func):
|
|
|
32
32
|
project_dir: Union[str, Path],
|
|
33
33
|
previous_result: pd.DataFrame,
|
|
34
34
|
**kwargs) -> Tuple[List[List[str]], List[List[str]], List[List[float]]]:
|
|
35
|
-
logger.info(f"Running retrieval node - {func.__name__} module...")
|
|
36
35
|
validate_qa_dataset(previous_result)
|
|
37
36
|
resources_dir = os.path.join(project_dir, "resources")
|
|
38
37
|
data_dir = os.path.join(project_dir, "data")
|
|
@@ -75,10 +74,10 @@ def retrieval_node(func):
|
|
|
75
74
|
del embedding_model
|
|
76
75
|
if torch.cuda.is_available():
|
|
77
76
|
torch.cuda.empty_cache()
|
|
78
|
-
elif func.__name__ in ["hybrid_rrf", "hybrid_cc"
|
|
79
|
-
if 'ids' in kwargs and 'scores' in kwargs:
|
|
77
|
+
elif func.__name__ in ["hybrid_rrf", "hybrid_cc"]:
|
|
78
|
+
if 'ids' in kwargs and 'scores' in kwargs: # ordinary run_evaluate
|
|
80
79
|
ids, scores = func(**kwargs)
|
|
81
|
-
else:
|
|
80
|
+
else: # => for Runner.run
|
|
82
81
|
if not ('target_modules' in kwargs and 'target_module_params' in kwargs):
|
|
83
82
|
raise ValueError(
|
|
84
83
|
f"If there are no ids and scores specified, target_modules and target_module_params must be specified for using {func.__name__}.")
|
|
@@ -0,0 +1,137 @@
|
|
|
1
|
+
from typing import Tuple, List
|
|
2
|
+
|
|
3
|
+
import numpy as np
|
|
4
|
+
import pandas as pd
|
|
5
|
+
|
|
6
|
+
from autorag.nodes.retrieval import retrieval_node
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def normalize_mm(scores: List[str], fixed_min_value: float = 0):
|
|
10
|
+
arr = np.array(scores)
|
|
11
|
+
max_value = np.max(arr)
|
|
12
|
+
min_value = np.min(arr)
|
|
13
|
+
norm_score = (arr - min_value) / (max_value - min_value)
|
|
14
|
+
return norm_score
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def normalize_tmm(scores: List[str], fixed_min_value: float):
|
|
18
|
+
arr = np.array(scores)
|
|
19
|
+
max_value = np.max(arr)
|
|
20
|
+
norm_score = (arr - fixed_min_value) / (max_value - fixed_min_value)
|
|
21
|
+
return norm_score
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def normalize_z(scores: List[str], fixed_min_value: float = 0):
|
|
25
|
+
arr = np.array(scores)
|
|
26
|
+
mean_value = np.mean(arr)
|
|
27
|
+
std_value = np.std(arr)
|
|
28
|
+
norm_score = (arr - mean_value) / std_value
|
|
29
|
+
return norm_score
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def normalize_dbsf(scores: List[str], fixed_min_value: float = 0):
|
|
33
|
+
arr = np.array(scores)
|
|
34
|
+
mean_value = np.mean(arr)
|
|
35
|
+
std_value = np.std(arr)
|
|
36
|
+
min_value = mean_value - 3 * std_value
|
|
37
|
+
max_value = mean_value + 3 * std_value
|
|
38
|
+
norm_score = (arr - min_value) / (max_value - min_value)
|
|
39
|
+
return norm_score
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
normalize_method_dict = {
|
|
43
|
+
'mm': normalize_mm,
|
|
44
|
+
'tmm': normalize_tmm,
|
|
45
|
+
'z': normalize_z,
|
|
46
|
+
'dbsf': normalize_dbsf,
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
@retrieval_node
|
|
51
|
+
def hybrid_cc(
|
|
52
|
+
ids: Tuple,
|
|
53
|
+
scores: Tuple,
|
|
54
|
+
top_k: int,
|
|
55
|
+
weight: float,
|
|
56
|
+
normalize_method: str = 'mm',
|
|
57
|
+
semantic_theoretical_min_value: float = -1.0,
|
|
58
|
+
lexical_theoretical_min_value: float = 0.0,
|
|
59
|
+
) -> Tuple[List[List[str]], List[List[float]]]:
|
|
60
|
+
"""
|
|
61
|
+
Hybrid CC function.
|
|
62
|
+
CC (convex combination) is a method to fuse lexical and semantic retrieval results.
|
|
63
|
+
It is a method that first normalizes the scores of each retrieval result,
|
|
64
|
+
and then combines them with the given weights.
|
|
65
|
+
It is uniquer than other retrieval modules, because it does not really execute retrieval,
|
|
66
|
+
but just fuse the results of other retrieval functions.
|
|
67
|
+
So you have to run more than two retrieval modules before running this function.
|
|
68
|
+
And collect ids and scores result from each retrieval module.
|
|
69
|
+
Make it as tuple and input it to this function.
|
|
70
|
+
|
|
71
|
+
:param ids: The tuple of ids that you want to fuse.
|
|
72
|
+
The length of this must be the same as the length of scores.
|
|
73
|
+
The semantic retrieval ids must be the first index.
|
|
74
|
+
:param scores: The retrieve scores that you want to fuse.
|
|
75
|
+
The length of this must be the same as the length of ids.
|
|
76
|
+
The semantic retrieval scores must be the first index.
|
|
77
|
+
:param top_k: The number of passages to be retrieved.
|
|
78
|
+
:param normalize_method: The normalization method to use.
|
|
79
|
+
There are some normalization method that you can use at the hybrid cc method.
|
|
80
|
+
AutoRAG support following.
|
|
81
|
+
- `mm`: Min-max scaling
|
|
82
|
+
- `tmm`: Theoretical min-max scaling
|
|
83
|
+
- `z`: z-score normalization
|
|
84
|
+
- `dbsf`: 3-sigma normalization
|
|
85
|
+
:param weight: The weight value. If the weight is 1.0, it means the
|
|
86
|
+
weight to the semantic module will be 1.0 and weight to the lexical module will be 0.0.
|
|
87
|
+
:param semantic_theoretical_min_value: This value used by `tmm` normalization method. You can set the
|
|
88
|
+
theoretical minimum value by yourself. Default is -1.
|
|
89
|
+
:param lexical_theoretical_min_value: This value used by `tmm` normalization method. You can set the
|
|
90
|
+
theoretical minimum value by yourself. Default is 0.
|
|
91
|
+
:return: The tuple of ids and fused scores that fused by CC. Plus, the third element is selected weight value.
|
|
92
|
+
"""
|
|
93
|
+
assert len(ids) == len(scores), "The length of ids and scores must be the same."
|
|
94
|
+
assert len(ids) > 1, "You must input more than one retrieval results."
|
|
95
|
+
assert top_k > 0, "top_k must be greater than 0."
|
|
96
|
+
assert weight >= 0, "The weight must be greater than 0."
|
|
97
|
+
assert weight <= 1, "The weight must be less than 1."
|
|
98
|
+
|
|
99
|
+
df = pd.DataFrame({
|
|
100
|
+
'semantic_ids': ids[0],
|
|
101
|
+
'lexical_ids': ids[1],
|
|
102
|
+
'semantic_score': scores[0],
|
|
103
|
+
'lexical_score': scores[1],
|
|
104
|
+
})
|
|
105
|
+
|
|
106
|
+
def cc_pure_apply(row):
|
|
107
|
+
return fuse_per_query(row['semantic_ids'], row['lexical_ids'],
|
|
108
|
+
row['semantic_score'], row['lexical_score'],
|
|
109
|
+
normalize_method=normalize_method,
|
|
110
|
+
weight=weight, top_k=top_k,
|
|
111
|
+
semantic_theoretical_min_value=semantic_theoretical_min_value,
|
|
112
|
+
lexical_theoretical_min_value=lexical_theoretical_min_value)
|
|
113
|
+
|
|
114
|
+
# fixed weight
|
|
115
|
+
df[['cc_id', 'cc_score']] = df.apply(lambda row: cc_pure_apply(row), axis=1,
|
|
116
|
+
result_type='expand')
|
|
117
|
+
return df['cc_id'].tolist(), df['cc_score'].tolist()
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def fuse_per_query(semantic_ids: List[str], lexical_ids: List[str],
|
|
121
|
+
semantic_scores: List[float], lexical_scores: List[float],
|
|
122
|
+
normalize_method: str,
|
|
123
|
+
weight: float,
|
|
124
|
+
top_k: int,
|
|
125
|
+
semantic_theoretical_min_value: float,
|
|
126
|
+
lexical_theoretical_min_value: float):
|
|
127
|
+
normalize_func = normalize_method_dict[normalize_method]
|
|
128
|
+
norm_semantic_scores = normalize_func(semantic_scores, semantic_theoretical_min_value)
|
|
129
|
+
norm_lexical_scores = normalize_func(lexical_scores, lexical_theoretical_min_value)
|
|
130
|
+
ids = [semantic_ids, lexical_ids]
|
|
131
|
+
scores = [norm_semantic_scores, norm_lexical_scores]
|
|
132
|
+
df = pd.concat([pd.Series(dict(zip(_id, score))) for _id, score in zip(ids, scores)], axis=1)
|
|
133
|
+
df.columns = ['semantic', 'lexical']
|
|
134
|
+
df = df.fillna(0)
|
|
135
|
+
df['weighted_sum'] = df.mul((weight, 1.0 - weight)).sum(axis=1)
|
|
136
|
+
df = df.sort_values(by='weighted_sum', ascending=False)
|
|
137
|
+
return df.index.tolist()[:top_k], df['weighted_sum'][:top_k].tolist()
|
|
@@ -10,7 +10,8 @@ def hybrid_rrf(
|
|
|
10
10
|
ids: Tuple,
|
|
11
11
|
scores: Tuple,
|
|
12
12
|
top_k: int,
|
|
13
|
-
|
|
13
|
+
weight: int = 60,
|
|
14
|
+
rrf_k: int = -1, ) -> Tuple[List[List[str]], List[List[float]]]:
|
|
14
15
|
"""
|
|
15
16
|
Hybrid RRF function.
|
|
16
17
|
RRF (Rank Reciprocal Fusion) is a method to fuse multiple retrieval results.
|
|
@@ -27,15 +28,23 @@ def hybrid_rrf(
|
|
|
27
28
|
:param scores: The retrieve scores that you want to fuse.
|
|
28
29
|
The length of this must be the same as the length of ids.
|
|
29
30
|
:param top_k: The number of passages to be retrieved.
|
|
30
|
-
:param
|
|
31
|
+
:param weight: Hyperparameter for RRF.
|
|
32
|
+
It was originally rrf_k value.
|
|
31
33
|
Default is 60.
|
|
32
34
|
For more information, please visit our documentation.
|
|
35
|
+
:param rrf_k: (Deprecated) Hyperparameter for RRF.
|
|
36
|
+
It was originally rrf_k value. Will remove at further version.
|
|
33
37
|
:return: The tuple of ids and fused scores that fused by RRF.
|
|
34
38
|
"""
|
|
35
39
|
assert len(ids) == len(scores), "The length of ids and scores must be the same."
|
|
36
40
|
assert len(ids) > 1, "You must input more than one retrieval results."
|
|
37
41
|
assert top_k > 0, "top_k must be greater than 0."
|
|
38
|
-
assert
|
|
42
|
+
assert weight > 0, "rrf_k must be greater than 0."
|
|
43
|
+
|
|
44
|
+
if rrf_k != -1:
|
|
45
|
+
weight = int(rrf_k)
|
|
46
|
+
else:
|
|
47
|
+
weight = int(weight)
|
|
39
48
|
|
|
40
49
|
id_df = pd.DataFrame({f'id_{i}': id_list for i, id_list in enumerate(ids)})
|
|
41
50
|
score_df = pd.DataFrame({f'score_{i}': score_list for i, score_list in enumerate(scores)})
|
|
@@ -44,9 +53,9 @@ def hybrid_rrf(
|
|
|
44
53
|
def rrf_pure_apply(row):
|
|
45
54
|
ids_tuple = tuple(row[[f'id_{i}' for i in range(len(ids))]].values)
|
|
46
55
|
scores_tuple = tuple(row[[f'score_{i}' for i in range(len(scores))]].values)
|
|
47
|
-
return pd.Series(rrf_pure(ids_tuple, scores_tuple,
|
|
56
|
+
return pd.Series(rrf_pure(ids_tuple, scores_tuple, weight, top_k))
|
|
48
57
|
|
|
49
|
-
df[['rrf_id', 'rrf_score']] = df.
|
|
58
|
+
df[['rrf_id', 'rrf_score']] = df.apply(rrf_pure_apply, axis=1)
|
|
50
59
|
return df['rrf_id'].tolist(), df['rrf_score'].tolist()
|
|
51
60
|
|
|
52
61
|
|
|
@@ -0,0 +1,285 @@
|
|
|
1
|
+
import logging
|
|
2
|
+
import os
|
|
3
|
+
import pathlib
|
|
4
|
+
from typing import List, Callable, Dict, Tuple
|
|
5
|
+
|
|
6
|
+
import numpy as np
|
|
7
|
+
import pandas as pd
|
|
8
|
+
from tqdm import tqdm
|
|
9
|
+
|
|
10
|
+
from autorag.evaluation import evaluate_retrieval
|
|
11
|
+
from autorag.strategy import measure_speed, filter_by_threshold, select_best
|
|
12
|
+
|
|
13
|
+
logger = logging.getLogger("AutoRAG")
|
|
14
|
+
|
|
15
|
+
semantic_module_names = ['vectordb']
|
|
16
|
+
lexical_module_names = ['bm25']
|
|
17
|
+
hybrid_module_names = ['hybrid_rrf', 'hybrid_cc']
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def run_retrieval_node(modules: List[Callable],
|
|
21
|
+
module_params: List[Dict],
|
|
22
|
+
previous_result: pd.DataFrame,
|
|
23
|
+
node_line_dir: str,
|
|
24
|
+
strategies: Dict,
|
|
25
|
+
) -> pd.DataFrame:
|
|
26
|
+
"""
|
|
27
|
+
Run evaluation and select the best module among retrieval node results.
|
|
28
|
+
|
|
29
|
+
:param modules: Retrieval modules to run.
|
|
30
|
+
:param module_params: Retrieval module parameters.
|
|
31
|
+
:param previous_result: Previous result dataframe.
|
|
32
|
+
Could be query expansion's best result or qa data.
|
|
33
|
+
:param node_line_dir: This node line's directory.
|
|
34
|
+
:param strategies: Strategies for retrieval node.
|
|
35
|
+
:return: The best result dataframe.
|
|
36
|
+
It contains previous result columns and retrieval node's result columns.
|
|
37
|
+
"""
|
|
38
|
+
if not os.path.exists(node_line_dir):
|
|
39
|
+
os.makedirs(node_line_dir)
|
|
40
|
+
project_dir = pathlib.PurePath(node_line_dir).parent.parent
|
|
41
|
+
qa_df = pd.read_parquet(os.path.join(project_dir, "data", "qa.parquet"), engine='pyarrow')
|
|
42
|
+
retrieval_gt = qa_df['retrieval_gt'].tolist()
|
|
43
|
+
retrieval_gt = [[[str(uuid) for uuid in sub_array] if sub_array.size > 0 else [] for sub_array in inner_array]
|
|
44
|
+
for inner_array in retrieval_gt]
|
|
45
|
+
|
|
46
|
+
save_dir = os.path.join(node_line_dir, "retrieval") # node name
|
|
47
|
+
if not os.path.exists(save_dir):
|
|
48
|
+
os.makedirs(save_dir)
|
|
49
|
+
|
|
50
|
+
def run(input_modules, input_module_params) -> Tuple[List[pd.DataFrame], List]:
|
|
51
|
+
"""
|
|
52
|
+
Run input modules and parameters.
|
|
53
|
+
|
|
54
|
+
:param input_modules: Input modules
|
|
55
|
+
:param input_module_params: Input module parameters
|
|
56
|
+
:return: First, it returns list of result dataframe.
|
|
57
|
+
Second, it returns list of execution times.
|
|
58
|
+
"""
|
|
59
|
+
result, execution_times = zip(*map(lambda task: measure_speed(
|
|
60
|
+
task[0], project_dir=project_dir, previous_result=previous_result, **task[1]),
|
|
61
|
+
zip(input_modules, input_module_params)))
|
|
62
|
+
average_times = list(map(lambda x: x / len(result[0]), execution_times))
|
|
63
|
+
|
|
64
|
+
# run metrics before filtering
|
|
65
|
+
if strategies.get('metrics') is None:
|
|
66
|
+
raise ValueError("You must at least one metrics for retrieval evaluation.")
|
|
67
|
+
result = list(map(lambda x: evaluate_retrieval_node(x, retrieval_gt, strategies.get('metrics'),
|
|
68
|
+
qa_df['query'].tolist(),
|
|
69
|
+
qa_df['generation_gt'].tolist()), result))
|
|
70
|
+
|
|
71
|
+
return result, average_times
|
|
72
|
+
|
|
73
|
+
def save_and_summary(input_modules, input_module_params, result_list,
|
|
74
|
+
execution_time_list, filename_start: int):
|
|
75
|
+
"""
|
|
76
|
+
Save the result and make summary file
|
|
77
|
+
|
|
78
|
+
:param input_modules: Input modules
|
|
79
|
+
:param input_module_params: Input module parameters
|
|
80
|
+
:param result_list: Result list
|
|
81
|
+
:param execution_time_list: Execution times
|
|
82
|
+
:param filename_start: The first filename to use
|
|
83
|
+
:return: First, it returns list of result dataframe.
|
|
84
|
+
Second, it returns list of execution times.
|
|
85
|
+
"""
|
|
86
|
+
|
|
87
|
+
# save results to folder
|
|
88
|
+
filepaths = list(map(lambda x: os.path.join(save_dir, f'{x}.parquet'),
|
|
89
|
+
range(filename_start, filename_start + len(input_modules))))
|
|
90
|
+
list(map(lambda x: x[0].to_parquet(x[1], index=False), zip(result_list, filepaths))) # execute save to parquet
|
|
91
|
+
filename_list = list(map(lambda x: os.path.basename(x), filepaths))
|
|
92
|
+
|
|
93
|
+
summary_df = pd.DataFrame({
|
|
94
|
+
'filename': filename_list,
|
|
95
|
+
'module_name': list(map(lambda module: module.__name__, input_modules)),
|
|
96
|
+
'module_params': input_module_params,
|
|
97
|
+
'execution_time': execution_time_list,
|
|
98
|
+
**{metric: list(map(lambda result: result[metric].mean(), result_list)) for metric in
|
|
99
|
+
strategies.get('metrics')},
|
|
100
|
+
})
|
|
101
|
+
summary_df.to_csv(os.path.join(save_dir, 'summary.csv'), index=False)
|
|
102
|
+
return summary_df
|
|
103
|
+
|
|
104
|
+
def find_best(results, average_times, filenames):
|
|
105
|
+
# filter by strategies
|
|
106
|
+
if strategies.get('speed_threshold') is not None:
|
|
107
|
+
results, filenames = filter_by_threshold(results, average_times, strategies['speed_threshold'], filenames)
|
|
108
|
+
selected_result, selected_filename = select_best(results, strategies.get('metrics'), filenames,
|
|
109
|
+
strategies.get('strategy', 'mean'))
|
|
110
|
+
return selected_result, selected_filename
|
|
111
|
+
|
|
112
|
+
filename_first = 0
|
|
113
|
+
# run semantic modules
|
|
114
|
+
logger.info(f"Running retrieval node - semantic retrieval module...")
|
|
115
|
+
if any([module.__name__ in semantic_module_names for module in modules]):
|
|
116
|
+
semantic_modules, semantic_module_params = zip(*filter(lambda x: x[0].__name__ in semantic_module_names,
|
|
117
|
+
zip(modules, module_params)))
|
|
118
|
+
semantic_results, semantic_times = run(semantic_modules, semantic_module_params)
|
|
119
|
+
semantic_summary_df = save_and_summary(semantic_modules, semantic_module_params,
|
|
120
|
+
semantic_results, semantic_times, filename_first)
|
|
121
|
+
semantic_selected_result, semantic_selected_filename = find_best(semantic_results, semantic_times,
|
|
122
|
+
semantic_summary_df['filename'].tolist())
|
|
123
|
+
semantic_summary_df['is_best'] = semantic_summary_df['filename'] == semantic_selected_filename
|
|
124
|
+
filename_first += len(semantic_modules)
|
|
125
|
+
else:
|
|
126
|
+
semantic_selected_filename, semantic_summary_df, semantic_results, semantic_times = None, pd.DataFrame(), [], []
|
|
127
|
+
# run lexical modules
|
|
128
|
+
logger.info(f"Running retrieval node - lexical retrieval module...")
|
|
129
|
+
if any([module.__name__ in lexical_module_names for module in modules]):
|
|
130
|
+
lexical_modules, lexical_module_params = zip(*filter(lambda x: x[0].__name__ in lexical_module_names,
|
|
131
|
+
zip(modules, module_params)))
|
|
132
|
+
lexical_results, lexical_times = run(lexical_modules, lexical_module_params)
|
|
133
|
+
lexical_summary_df = save_and_summary(lexical_modules, lexical_module_params,
|
|
134
|
+
lexical_results, lexical_times, filename_first)
|
|
135
|
+
lexical_selected_result, lexical_selected_filename = find_best(lexical_results, lexical_times,
|
|
136
|
+
lexical_summary_df['filename'].tolist())
|
|
137
|
+
lexical_summary_df['is_best'] = lexical_summary_df['filename'] == lexical_selected_filename
|
|
138
|
+
filename_first += len(lexical_modules)
|
|
139
|
+
else:
|
|
140
|
+
lexical_selected_filename, lexical_summary_df, lexical_results, lexical_times = None, pd.DataFrame(), [], []
|
|
141
|
+
|
|
142
|
+
logger.info(f"Running retrieval node - hybrid retrieval module...")
|
|
143
|
+
# Next, run hybrid retrieval
|
|
144
|
+
if any([module.__name__ in hybrid_module_names for module in modules]):
|
|
145
|
+
hybrid_modules, hybrid_module_params = zip(*filter(lambda x: x[0].__name__ in hybrid_module_names,
|
|
146
|
+
zip(modules, module_params)))
|
|
147
|
+
if all(['target_module_params' in x for x in hybrid_module_params]): # for Runner.run
|
|
148
|
+
# If target_module_params are already given, run hybrid retrieval directly
|
|
149
|
+
hybrid_results, hybrid_times = run(hybrid_modules, hybrid_module_params)
|
|
150
|
+
hybrid_summary_df = save_and_summary(hybrid_modules, hybrid_module_params,
|
|
151
|
+
hybrid_results, hybrid_times, filename_first)
|
|
152
|
+
filename_first += len(hybrid_modules)
|
|
153
|
+
else: # for Evaluator
|
|
154
|
+
# get id and score
|
|
155
|
+
ids_scores = get_ids_and_scores(save_dir, [semantic_selected_filename, lexical_selected_filename])
|
|
156
|
+
hybrid_module_params = list(map(lambda x: {**x, **ids_scores}, hybrid_module_params))
|
|
157
|
+
|
|
158
|
+
# optimize each modules
|
|
159
|
+
real_hybrid_times = [get_hybrid_execution_times(semantic_summary_df, lexical_summary_df)
|
|
160
|
+
] * len(hybrid_module_params)
|
|
161
|
+
hybrid_times = real_hybrid_times.copy()
|
|
162
|
+
hybrid_results = []
|
|
163
|
+
for module, module_param in zip(hybrid_modules, hybrid_module_params):
|
|
164
|
+
module_result_df, module_best_weight = optimize_hybrid(module, module_param, strategies,
|
|
165
|
+
retrieval_gt, qa_df,
|
|
166
|
+
project_dir, previous_result)
|
|
167
|
+
module_param['weight'] = module_best_weight
|
|
168
|
+
hybrid_results.append(module_result_df)
|
|
169
|
+
|
|
170
|
+
hybrid_summary_df = save_and_summary(hybrid_modules, hybrid_module_params,
|
|
171
|
+
hybrid_results, hybrid_times, filename_first)
|
|
172
|
+
filename_first += len(hybrid_modules)
|
|
173
|
+
hybrid_summary_df['execution_time'] = hybrid_times
|
|
174
|
+
best_semantic_summary_row = semantic_summary_df.loc[semantic_summary_df['is_best'] == True].iloc[0]
|
|
175
|
+
best_lexical_summary_row = lexical_summary_df.loc[lexical_summary_df['is_best'] == True].iloc[0]
|
|
176
|
+
target_modules = (best_semantic_summary_row['module_name'], best_lexical_summary_row['module_name'])
|
|
177
|
+
target_module_params = (
|
|
178
|
+
best_semantic_summary_row['module_params'], best_lexical_summary_row['module_params'])
|
|
179
|
+
hybrid_summary_df = edit_summary_df_params(hybrid_summary_df, target_modules, target_module_params)
|
|
180
|
+
else:
|
|
181
|
+
if any([module.__name__ in hybrid_module_names for module in modules]):
|
|
182
|
+
logger.warning("You must at least one semantic module and lexical module for hybrid evaluation."
|
|
183
|
+
"Passing hybrid module.")
|
|
184
|
+
hybrid_selected_filename, hybrid_summary_df, hybrid_results, hybrid_times = None, pd.DataFrame(), [], []
|
|
185
|
+
|
|
186
|
+
summary = pd.concat([semantic_summary_df, lexical_summary_df, hybrid_summary_df], ignore_index=True)
|
|
187
|
+
results = semantic_results + lexical_results + hybrid_results
|
|
188
|
+
average_times = semantic_times + lexical_times + hybrid_times
|
|
189
|
+
filenames = summary['filename'].tolist()
|
|
190
|
+
|
|
191
|
+
# filter by strategies
|
|
192
|
+
selected_result, selected_filename = find_best(results, average_times, filenames)
|
|
193
|
+
best_result = pd.concat([previous_result, selected_result], axis=1)
|
|
194
|
+
|
|
195
|
+
# add summary.csv 'is_best' column
|
|
196
|
+
summary['is_best'] = summary['filename'] == selected_filename
|
|
197
|
+
|
|
198
|
+
# save the result files
|
|
199
|
+
best_result.to_parquet(os.path.join(save_dir, f'best_{os.path.splitext(selected_filename)[0]}.parquet'),
|
|
200
|
+
index=False)
|
|
201
|
+
summary.to_csv(os.path.join(save_dir, 'summary.csv'), index=False)
|
|
202
|
+
return best_result
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
def evaluate_retrieval_node(result_df: pd.DataFrame, retrieval_gt, metrics,
|
|
206
|
+
queries: List[str], generation_gt: List[List[str]]) -> pd.DataFrame:
|
|
207
|
+
"""
|
|
208
|
+
Evaluate retrieval node from retrieval node result dataframe.
|
|
209
|
+
|
|
210
|
+
:param result_df: The result dataframe from a retrieval node.
|
|
211
|
+
:param retrieval_gt: Ground truth for retrieval from qa dataset.
|
|
212
|
+
:param metrics: Metric list from input strategies.
|
|
213
|
+
:param queries: Query list from input strategies.
|
|
214
|
+
:param generation_gt: Ground truth for generation from qa dataset.
|
|
215
|
+
:return: Return result_df with metrics columns.
|
|
216
|
+
The columns will be 'retrieved_contents', 'retrieved_ids', 'retrieve_scores', and metric names.
|
|
217
|
+
"""
|
|
218
|
+
|
|
219
|
+
@evaluate_retrieval(retrieval_gt=retrieval_gt, metrics=metrics, queries=queries, generation_gt=generation_gt)
|
|
220
|
+
def evaluate_this_module(df: pd.DataFrame):
|
|
221
|
+
return df['retrieved_contents'].tolist(), df['retrieved_ids'].tolist(), df['retrieve_scores'].tolist()
|
|
222
|
+
|
|
223
|
+
return evaluate_this_module(result_df)
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
def edit_summary_df_params(summary_df: pd.DataFrame, target_modules, target_module_params) -> pd.DataFrame:
|
|
227
|
+
def delete_ids_scores(x):
|
|
228
|
+
del x['ids']
|
|
229
|
+
del x['scores']
|
|
230
|
+
return x
|
|
231
|
+
|
|
232
|
+
summary_df['module_params'] = summary_df['module_params'].apply(delete_ids_scores)
|
|
233
|
+
summary_df['new_params'] = [{'target_modules': target_modules,
|
|
234
|
+
'target_module_params': target_module_params}] * len(summary_df)
|
|
235
|
+
summary_df['module_params'] = summary_df.apply(lambda row: {**row['module_params'], **row['new_params']}, axis=1)
|
|
236
|
+
summary_df = summary_df.drop(columns=['new_params'])
|
|
237
|
+
return summary_df
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
def get_ids_and_scores(node_dir: str, filenames: List[str]) -> Dict:
|
|
241
|
+
best_results_df = list(
|
|
242
|
+
map(lambda filename: pd.read_parquet(os.path.join(node_dir, filename), engine='pyarrow'), filenames))
|
|
243
|
+
ids = tuple(map(lambda df: df['retrieved_ids'].apply(list).tolist(), best_results_df))
|
|
244
|
+
scores = tuple(map(lambda df: df['retrieve_scores'].apply(list).tolist(), best_results_df))
|
|
245
|
+
return {
|
|
246
|
+
'ids': ids,
|
|
247
|
+
'scores': scores,
|
|
248
|
+
}
|
|
249
|
+
|
|
250
|
+
|
|
251
|
+
def get_hybrid_execution_times(lexical_summary, semantic_summary) -> float:
|
|
252
|
+
lexical_execution_time = lexical_summary.loc[lexical_summary['is_best'] == True].iloc[0]['execution_time']
|
|
253
|
+
semantic_execution_time = semantic_summary.loc[semantic_summary['is_best'] == True].iloc[0]['execution_time']
|
|
254
|
+
return lexical_execution_time + semantic_execution_time
|
|
255
|
+
|
|
256
|
+
|
|
257
|
+
def optimize_hybrid(hybrid_module_func: Callable, hybrid_module_param: Dict,
|
|
258
|
+
strategy: Dict, retrieval_gt, qa_df: pd.DataFrame,
|
|
259
|
+
project_dir, previous_result):
|
|
260
|
+
if hybrid_module_func.__name__ == 'hybrid_rrf':
|
|
261
|
+
weight_range = hybrid_module_param.pop('weight_range', (4, 80))
|
|
262
|
+
test_weight_size = weight_range[1] - weight_range[0] + 1
|
|
263
|
+
else:
|
|
264
|
+
weight_range = hybrid_module_param.pop('weight_range', (0.0, 1.0))
|
|
265
|
+
test_weight_size = hybrid_module_param.pop('test_weight_size', 101)
|
|
266
|
+
|
|
267
|
+
weight_candidates = np.linspace(weight_range[0], weight_range[1], test_weight_size).tolist()
|
|
268
|
+
|
|
269
|
+
result_list = []
|
|
270
|
+
for weight_value in tqdm(weight_candidates):
|
|
271
|
+
result_df = hybrid_module_func(project_dir=project_dir, previous_result=previous_result,
|
|
272
|
+
weight=weight_value, **hybrid_module_param)
|
|
273
|
+
result_list.append(result_df)
|
|
274
|
+
|
|
275
|
+
# evaluate here
|
|
276
|
+
if strategy.get('metrics') is None:
|
|
277
|
+
raise ValueError("You must at least one metrics for retrieval evaluation.")
|
|
278
|
+
result_list = list(map(lambda x: evaluate_retrieval_node(x, retrieval_gt, strategy.get('metrics'),
|
|
279
|
+
qa_df['query'].tolist(),
|
|
280
|
+
qa_df['generation_gt'].tolist()), result_list))
|
|
281
|
+
|
|
282
|
+
# select best result
|
|
283
|
+
best_result_df, best_weight = select_best(result_list, strategy.get('metrics'), metadatas=weight_candidates,
|
|
284
|
+
strategy_name=strategy.get('strategy', 'normalize_mean'))
|
|
285
|
+
return best_result_df, best_weight
|
|
@@ -24,8 +24,6 @@ def get_support_modules(module_name: str) -> Callable:
|
|
|
24
24
|
'vectordb': ('autorag.nodes.retrieval', 'vectordb'),
|
|
25
25
|
'hybrid_rrf': ('autorag.nodes.retrieval', 'hybrid_rrf'),
|
|
26
26
|
'hybrid_cc': ('autorag.nodes.retrieval', 'hybrid_cc'),
|
|
27
|
-
'hybrid_rsf': ('autorag.nodes.retrieval', 'hybrid_rsf'),
|
|
28
|
-
'hybrid_dbsf': ('autorag.nodes.retrieval', 'hybrid_dbsf'),
|
|
29
27
|
# passage_augmenter
|
|
30
28
|
'prev_next_augmenter': ('autorag.nodes.passageaugmenter', 'prev_next_augmenter'),
|
|
31
29
|
'pass_passage_augmenter': ('autorag.nodes.passageaugmenter', 'pass_passage_augmenter'),
|
|
@@ -28,14 +28,6 @@ autorag.nodes.retrieval.hybrid\_cc module
|
|
|
28
28
|
:undoc-members:
|
|
29
29
|
:show-inheritance:
|
|
30
30
|
|
|
31
|
-
autorag.nodes.retrieval.hybrid\_dbsf module
|
|
32
|
-
-------------------------------------------
|
|
33
|
-
|
|
34
|
-
.. automodule:: autorag.nodes.retrieval.hybrid_dbsf
|
|
35
|
-
:members:
|
|
36
|
-
:undoc-members:
|
|
37
|
-
:show-inheritance:
|
|
38
|
-
|
|
39
31
|
autorag.nodes.retrieval.hybrid\_rrf module
|
|
40
32
|
------------------------------------------
|
|
41
33
|
|
|
@@ -44,14 +36,6 @@ autorag.nodes.retrieval.hybrid\_rrf module
|
|
|
44
36
|
:undoc-members:
|
|
45
37
|
:show-inheritance:
|
|
46
38
|
|
|
47
|
-
autorag.nodes.retrieval.hybrid\_rsf module
|
|
48
|
-
------------------------------------------
|
|
49
|
-
|
|
50
|
-
.. automodule:: autorag.nodes.retrieval.hybrid_rsf
|
|
51
|
-
:members:
|
|
52
|
-
:undoc-members:
|
|
53
|
-
:show-inheritance:
|
|
54
|
-
|
|
55
39
|
autorag.nodes.retrieval.run module
|
|
56
40
|
----------------------------------
|
|
57
41
|
|
|
@@ -63,7 +63,7 @@ at [here](https://medium.com/@autorag/sem-score-maybe-the-answer-to-rag-evaluati
|
|
|
63
63
|
|
|
64
64
|
## 5. G-Eval
|
|
65
65
|
|
|
66
|
-
### 📌Definition
|
|
66
|
+
### 📌 Definition
|
|
67
67
|
|
|
68
68
|
Here is the [link](https://arxiv.org/abs/2303.16634) that introduced ***G-Eval***
|
|
69
69
|
|
|
@@ -77,14 +77,14 @@ So, in AutoRAG, we use **G-Eval with GPT-4**
|
|
|
77
77
|
|
|
78
78
|
---
|
|
79
79
|
|
|
80
|
-
###
|
|
80
|
+
### 5-1. Coherence
|
|
81
81
|
|
|
82
82
|
- Evaluate whether the answer is logically consistent and flows naturally.
|
|
83
83
|
- Evaluate the connections between sentences and how they fit into the overall context.
|
|
84
84
|
|
|
85
85
|
---
|
|
86
86
|
|
|
87
|
-
###
|
|
87
|
+
### 5-2. Consistency
|
|
88
88
|
|
|
89
89
|
- Evaluate whether the answer is consistent with and does not contradict the question asked or the information
|
|
90
90
|
presented.
|
|
@@ -93,17 +93,28 @@ So, in AutoRAG, we use **G-Eval with GPT-4**
|
|
|
93
93
|
|
|
94
94
|
---
|
|
95
95
|
|
|
96
|
-
###
|
|
96
|
+
### 5-3. Fluency
|
|
97
97
|
|
|
98
98
|
- Evaluate answers for fluency
|
|
99
99
|
|
|
100
100
|
---
|
|
101
101
|
|
|
102
|
-
###
|
|
102
|
+
### 5-4. Relevance
|
|
103
103
|
|
|
104
104
|
- Evaluate how well the answer meets the question's requirements
|
|
105
105
|
- A highly relevant answer should be directly related to the question's core topic or keyword.
|
|
106
106
|
|
|
107
|
+
### ❗How to use specific G-Eval metrics
|
|
108
|
+
|
|
109
|
+
You can use specific G-Eval metrics to use `metrics` parameter.
|
|
110
|
+
|
|
111
|
+
Here is an example yaml file that uses **G-Eval consistency** metric.
|
|
112
|
+
|
|
113
|
+
```yaml
|
|
114
|
+
- metric_name: g_eval
|
|
115
|
+
metrics: [ consistency ]
|
|
116
|
+
```
|
|
117
|
+
|
|
107
118
|
## 6. Bert Score
|
|
108
119
|
|
|
109
120
|
### 📌Definition
|