AutoRAG 0.2.13__tar.gz → 0.2.14__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- autorag-0.2.14/.github/ISSUE_TEMPLATE/bug_report.md +35 -0
- autorag-0.2.14/.github/ISSUE_TEMPLATE/feature_request.md +20 -0
- {autorag-0.2.13 → autorag-0.2.14}/AutoRAG.egg-info/PKG-INFO +4 -4
- {autorag-0.2.13 → autorag-0.2.14}/AutoRAG.egg-info/SOURCES.txt +2 -0
- {autorag-0.2.13 → autorag-0.2.14}/AutoRAG.egg-info/requires.txt +3 -3
- {autorag-0.2.13 → autorag-0.2.14}/PKG-INFO +4 -4
- autorag-0.2.14/autorag/VERSION +1 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/__init__.py +3 -5
- autorag-0.2.14/autorag/data/qacreation/__init__.py +2 -0
- autorag-0.2.14/autorag/data/qacreation/base.py +168 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/data/qacreation/llama_index.py +41 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/deploy.py +6 -4
- {autorag-0.2.13 → autorag-0.2.14}/autorag/evaluation/metric/generation.py +2 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/passageaugmenter/base.py +2 -1
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/passagefilter/threshold_cutoff.py +10 -17
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/retrieval/base.py +1 -1
- {autorag-0.2.13 → autorag-0.2.14}/autorag/strategy.py +1 -1
- {autorag-0.2.13 → autorag-0.2.14}/autorag/utils/util.py +1 -1
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/data_creation/tutorial.md +46 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/local_model.md +1 -0
- {autorag-0.2.13 → autorag-0.2.14}/requirements.txt +3 -3
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/data/qacreation/test_base_qacreation.py +37 -1
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/data/qacreation/test_llama_index_qacreation.py +23 -1
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/test_cli.py +2 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/test_deploy.py +0 -2
- autorag-0.2.13/autorag/VERSION +0 -1
- autorag-0.2.13/autorag/data/qacreation/__init__.py +0 -2
- autorag-0.2.13/autorag/data/qacreation/base.py +0 -79
- {autorag-0.2.13 → autorag-0.2.14}/.github/FUNDING.yml +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/.github/dependabot.yml +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/.github/workflows/publish.yml +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/.github/workflows/sphinx.yml +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/.github/workflows/test.yml +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/.gitignore +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/AutoRAG.egg-info/dependency_links.txt +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/AutoRAG.egg-info/entry_points.txt +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/AutoRAG.egg-info/top_level.txt +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/CODE_OF_CONDUCT.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/CONTRIBUTING.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/LICENSE +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/README.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/cli.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/dashboard.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/data/__init__.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/data/corpus/__init__.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/data/corpus/langchain.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/data/corpus/llama_index.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/data/qacreation/llama_index_default_prompt.txt +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/data/qacreation/ragas.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/data/qacreation/simple.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/data/utils/__init__.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/data/utils/util.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/evaluation/__init__.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/evaluation/generation.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/evaluation/metric/__init__.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/evaluation/metric/g_eval_prompts/coh_detailed.txt +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/evaluation/metric/g_eval_prompts/con_detailed.txt +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/evaluation/metric/g_eval_prompts/flu_detailed.txt +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/evaluation/metric/g_eval_prompts/rel_detailed.txt +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/evaluation/metric/retrieval.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/evaluation/metric/retrieval_contents.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/evaluation/metric/util.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/evaluation/retrieval.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/evaluation/retrieval_contents.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/evaluation/util.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/evaluator.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/node_line.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/__init__.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/generator/__init__.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/generator/base.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/generator/llama_index_llm.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/generator/openai_llm.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/generator/run.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/generator/vllm.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/passageaugmenter/__init__.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/passageaugmenter/pass_passage_augmenter.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/passageaugmenter/prev_next_augmenter.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/passageaugmenter/run.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/passagecompressor/__init__.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/passagecompressor/base.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/passagecompressor/longllmlingua.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/passagecompressor/pass_compressor.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/passagecompressor/refine.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/passagecompressor/run.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/passagecompressor/tree_summarize.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/passagefilter/__init__.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/passagefilter/base.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/passagefilter/pass_passage_filter.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/passagefilter/percentile_cutoff.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/passagefilter/recency.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/passagefilter/run.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/passagefilter/similarity_percentile_cutoff.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/passagefilter/similarity_threshold_cutoff.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/passagereranker/__init__.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/passagereranker/base.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/passagereranker/cohere.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/passagereranker/colbert.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/passagereranker/flag_embedding.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/passagereranker/flag_embedding_llm.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/passagereranker/jina.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/passagereranker/koreranker.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/passagereranker/monot5.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/passagereranker/pass_reranker.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/passagereranker/rankgpt.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/passagereranker/run.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/passagereranker/sentence_transformer.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/passagereranker/tart/__init__.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/passagereranker/tart/modeling_enc_t5.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/passagereranker/tart/tart.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/passagereranker/tart/tokenization_enc_t5.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/passagereranker/time_reranker.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/passagereranker/upr.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/promptmaker/__init__.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/promptmaker/base.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/promptmaker/fstring.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/promptmaker/long_context_reorder.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/promptmaker/run.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/promptmaker/window_replacement.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/queryexpansion/__init__.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/queryexpansion/base.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/queryexpansion/hyde.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/queryexpansion/multi_query_expansion.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/queryexpansion/pass_query_expansion.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/queryexpansion/query_decompose.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/queryexpansion/run.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/retrieval/__init__.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/retrieval/bm25.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/retrieval/hybrid_cc.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/retrieval/hybrid_rrf.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/retrieval/run.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/nodes/retrieval/vectordb.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/schema/__init__.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/schema/module.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/schema/node.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/support.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/utils/__init__.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/utils/preprocess.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/validator.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/autorag/web.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/Makefile +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/make.bat +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/requirements.txt +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/CNAME +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/_static/data_creation.png +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/_static/data_folder.png +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/_static/dcg.png +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/_static/f1_score.png +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/_static/map.png +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/_static/mrr.png +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/_static/ndcg.png +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/_static/ndcg_formula.png +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/_static/node_folder.png +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/_static/node_line_folder.png +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/_static/node_line_summary.png +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/_static/node_lines.png +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/_static/node_summary.png +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/_static/normal_distribution.png +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/_static/project_folder_example.png +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/_static/project_folders.png +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/_static/resources_folder.png +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/_static/roadmap/RAG_paradigms.png +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/_static/roadmap/advanced_RAG.png +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/_static/roadmap/cycle.png +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/_static/roadmap/merger.png +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/_static/roadmap/node_line_modular.png +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/_static/roadmap/policy.png +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/_static/samsung_sundae.jpeg +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/_static/score_fusion.png +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/_static/trial_folder.png +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/_static/trial_json.png +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/_static/trial_summary.png +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/_static/web_interface.png +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/api_spec/autorag.data.corpus.rst +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/api_spec/autorag.data.qacreation.rst +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/api_spec/autorag.data.rst +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/api_spec/autorag.data.utils.rst +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/api_spec/autorag.evaluation.metric.rst +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/api_spec/autorag.evaluation.rst +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/api_spec/autorag.nodes.generator.rst +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/api_spec/autorag.nodes.passageaugmenter.rst +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/api_spec/autorag.nodes.passagecompressor.rst +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/api_spec/autorag.nodes.passagefilter.rst +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/api_spec/autorag.nodes.passagereranker.rst +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/api_spec/autorag.nodes.passagereranker.tart.rst +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/api_spec/autorag.nodes.promptmaker.rst +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/api_spec/autorag.nodes.queryexpansion.rst +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/api_spec/autorag.nodes.retrieval.rst +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/api_spec/autorag.nodes.rst +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/api_spec/autorag.rst +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/api_spec/autorag.schema.rst +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/api_spec/autorag.utils.rst +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/api_spec/modules.rst +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/conf.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/data_creation/data_format.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/data_creation/ragas.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/deploy/api_endpoint.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/deploy/web.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/evaluate_metrics/generation.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/evaluate_metrics/retrieval.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/evaluate_metrics/retrieval_contents.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/index.rst +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/install.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/nodes/generator/generator.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/nodes/generator/llama_index_llm.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/nodes/generator/openai_llm.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/nodes/generator/vllm.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/nodes/index.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/nodes/passage_augmenter/passage_augmenter.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/nodes/passage_augmenter/prev_next_augmenter.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/nodes/passage_compressor/longllmlingua.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/nodes/passage_compressor/passage_compressor.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/nodes/passage_compressor/refine.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/nodes/passage_compressor/tree_summarize.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/nodes/passage_filter/passage_filter.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/nodes/passage_filter/percentile_cutoff.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/nodes/passage_filter/recency_filter.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/nodes/passage_filter/similarity_percentile_cutoff.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/nodes/passage_filter/similarity_threshold_cutoff.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/nodes/passage_filter/threshold_cutoff.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/nodes/passage_reranker/cohere.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/nodes/passage_reranker/colbert.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/nodes/passage_reranker/flag_embedding_llm_reranker.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/nodes/passage_reranker/flag_embedding_reranker.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/nodes/passage_reranker/jina_reranker.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/nodes/passage_reranker/koreranker.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/nodes/passage_reranker/monot5.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/nodes/passage_reranker/passage_reranker.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/nodes/passage_reranker/rankgpt.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/nodes/passage_reranker/sentence_transformer_reranker.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/nodes/passage_reranker/tart.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/nodes/passage_reranker/time_reranker.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/nodes/passage_reranker/upr.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/nodes/prompt_maker/fstring.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/nodes/prompt_maker/long_context_reorder.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/nodes/prompt_maker/prompt_maker.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/nodes/prompt_maker/window_replacement.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/nodes/query_expansion/hyde.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/nodes/query_expansion/multi_query_expansion.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/nodes/query_expansion/query_decompose.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/nodes/query_expansion/query_expansion.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/nodes/retrieval/bm25.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/nodes/retrieval/hybrid_cc.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/nodes/retrieval/hybrid_rrf.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/nodes/retrieval/retrieval.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/nodes/retrieval/vectordb.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/optimization/custom_config.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/optimization/folder_structure.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/optimization/optimization.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/optimization/strategies.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/roadmap/modular_rag.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/structure.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/troubleshooting.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/docs/source/tutorial.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/pyproject.toml +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/sample_config/compact_local.yaml +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/sample_config/compact_openai.yaml +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/sample_config/config_korean.yaml +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/sample_config/extracted_sample.yaml +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/sample_config/full.yaml +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/sample_config/simple_local.yaml +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/sample_config/simple_ollama.yaml +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/sample_config/simple_openai.yaml +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/sample_dataset/README.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/sample_dataset/eli5/load_eli5_dataset.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/sample_dataset/hotpotqa/load_hotpotqa_dataset.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/sample_dataset/msmarco/load_msmarco_dataset.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/sample_dataset/triviaqa/load_triviaqa_dataset.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/setup.cfg +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/data/corpus/test_base.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/data/corpus/test_langchain.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/data/corpus/test_llama_index_corpus.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/data/qacreation/test_ragas_qa_creation.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/data/qacreation/test_simple.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/evaluate/metric/test_generation_metric.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/evaluate/metric/test_retrieval_contents_metric.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/evaluate/metric/test_retrieval_metric.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/evaluate/test_evaluate_util.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/evaluate/test_generation_evaluate.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/evaluate/test_retrieval_contents_evaluate.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/evaluate/test_retrieval_evaluate.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/generator/test_generator_base.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/generator/test_llama_index_llm.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/generator/test_openai.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/generator/test_run_generator_node.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/generator/test_vllm.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/passageaugmenter/test_base_passage_augmenter.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/passageaugmenter/test_pass_passage_augmenter.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/passageaugmenter/test_prev_next_augmenter.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/passageaugmenter/test_run_passage_augmenter.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/passagecompressor/test_base_passage_compressor.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/passagecompressor/test_longllmlingua.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/passagecompressor/test_pass_compressor.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/passagecompressor/test_refine.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/passagecompressor/test_run_passage_compressor_node.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/passagecompressor/test_tree_summarize.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/passagefilter/test_pass_passage_filter.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/passagefilter/test_passage_filter_base.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/passagefilter/test_passage_filter_run.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/passagefilter/test_percentile_cutoff.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/passagefilter/test_recency_filter.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/passagefilter/test_similarity_percentile_cutoff.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/passagefilter/test_similarity_threshold_cutoff.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/passagefilter/test_threshold_cutoff.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/passagereranker/test_cohere_reranker.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/passagereranker/test_colbert_reranker.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/passagereranker/test_flag_embedding.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/passagereranker/test_flag_embedding_llm.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/passagereranker/test_jina_reranker.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/passagereranker/test_koreranker.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/passagereranker/test_monot5.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/passagereranker/test_pass_reranker.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/passagereranker/test_passage_reranker_base.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/passagereranker/test_passage_reranker_run.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/passagereranker/test_rankgpt.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/passagereranker/test_sentence_transformer.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/passagereranker/test_tart.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/passagereranker/test_time_reranker.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/passagereranker/test_upr.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/promptmaker/test_fstring.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/promptmaker/test_long_context_reorder.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/promptmaker/test_prompt_maker_base.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/promptmaker/test_prompt_maker_run.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/promptmaker/test_window_replacement.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/queryexpansion/test_hyde.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/queryexpansion/test_multi_query_expansion.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/queryexpansion/test_pass_query_expansion.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/queryexpansion/test_query_decompose.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/queryexpansion/test_query_expansion_base.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/queryexpansion/test_query_expansion_run.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/retrieval/test_bm25.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/retrieval/test_hybrid_base.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/retrieval/test_hybrid_cc.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/retrieval/test_hybrid_rrf.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/retrieval/test_retrieval_base.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/retrieval/test_run_retrieval_node.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/nodes/retrieval/test_vectordb.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/schema/test_module_schema.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/schema/test_node_schema.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/test_dashboard.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/test_evaluator.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/test_strategy.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/test_support.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/test_validator.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/test_web.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/utils/test_preprocess.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/autorag/utils/test_util.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/conftest.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/delete_tests.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/mock.py +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/requirements.txt +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/README.md +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/corpus_data_sample.parquet +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/data_creation/raw_dir/sample1.txt +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/data_creation/raw_dir/sample2.txt +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/data_creation/raw_dir/sample3.txt +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/full.yaml +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/qa_data_sample.parquet +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/qa_gen_prompts/prompt1.txt +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/qa_gen_prompts/prompt2.txt +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/qa_gen_prompts/prompt3.txt +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/qa_test_data_sample.parquet +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/0/config.yaml +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/0/post_retrieve_node_line/generator/0.parquet +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/0/post_retrieve_node_line/generator/1.parquet +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/0/post_retrieve_node_line/generator/2.parquet +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/0/post_retrieve_node_line/generator/3.parquet +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/0/post_retrieve_node_line/generator/4.parquet +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/0/post_retrieve_node_line/generator/5.parquet +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/0/post_retrieve_node_line/generator/best_5.parquet +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/0/post_retrieve_node_line/generator/summary.csv +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/0/post_retrieve_node_line/prompt_maker/0.parquet +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/0/post_retrieve_node_line/prompt_maker/1.parquet +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/0/post_retrieve_node_line/prompt_maker/best_0.parquet +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/0/post_retrieve_node_line/prompt_maker/summary.csv +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/0/post_retrieve_node_line/summary.csv +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/0/pre_retrieve_node_line/query_expansion/0.parquet +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/0/pre_retrieve_node_line/query_expansion/1.parquet +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/0/pre_retrieve_node_line/query_expansion/2.parquet +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/0/pre_retrieve_node_line/query_expansion/best_0.parquet +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/0/pre_retrieve_node_line/query_expansion/summary.csv +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/0/pre_retrieve_node_line/summary.csv +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/0/retrieve_node_line/passage_compressor/0.parquet +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/0/retrieve_node_line/passage_compressor/best_0.parquet +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/0/retrieve_node_line/passage_compressor/summary.csv +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/0/retrieve_node_line/passage_reranker/0.parquet +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/0/retrieve_node_line/passage_reranker/1.parquet +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/0/retrieve_node_line/passage_reranker/best_0.parquet +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/0/retrieve_node_line/passage_reranker/summary.csv +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/0/retrieve_node_line/retrieval/0.parquet +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/0/retrieve_node_line/retrieval/1.parquet +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/0/retrieve_node_line/retrieval/best_0.parquet +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/0/retrieve_node_line/retrieval/summary.csv +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/0/retrieve_node_line/summary.csv +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/0/summary.csv +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/1/config.yaml +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/1/pre_retrieve_node_line/query_expansion/0.parquet +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/1/pre_retrieve_node_line/query_expansion/1.parquet +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/1/pre_retrieve_node_line/query_expansion/2.parquet +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/1/pre_retrieve_node_line/query_expansion/best_0.parquet +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/1/pre_retrieve_node_line/query_expansion/summary.csv +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/1/pre_retrieve_node_line/summary.csv +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/1/retrieve_node_line/passage_filter/0.parquet +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/1/retrieve_node_line/passage_reranker/0.parquet +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/1/retrieve_node_line/passage_reranker/1.parquet +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/1/retrieve_node_line/passage_reranker/best_0.parquet +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/1/retrieve_node_line/passage_reranker/summary.csv +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/1/retrieve_node_line/retrieval/0.parquet +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/1/retrieve_node_line/retrieval/1.parquet +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/1/retrieve_node_line/retrieval/best_0.parquet +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/1/retrieve_node_line/retrieval/summary.csv +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/2/config.yaml +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/2/post_retrieve_node_line/prompt_maker/0.parquet +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/2/pre_retrieve_node_line/query_expansion/0.parquet +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/2/pre_retrieve_node_line/query_expansion/1.parquet +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/2/pre_retrieve_node_line/query_expansion/2.parquet +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/2/pre_retrieve_node_line/query_expansion/best_0.parquet +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/2/pre_retrieve_node_line/query_expansion/summary.csv +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/2/pre_retrieve_node_line/summary.csv +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/2/retrieve_node_line/passage_compressor/0.parquet +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/2/retrieve_node_line/passage_compressor/best_0.parquet +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/2/retrieve_node_line/passage_compressor/summary.csv +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/2/retrieve_node_line/passage_reranker/0.parquet +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/2/retrieve_node_line/passage_reranker/1.parquet +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/2/retrieve_node_line/passage_reranker/best_0.parquet +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/2/retrieve_node_line/passage_reranker/summary.csv +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/2/retrieve_node_line/retrieval/0.parquet +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/2/retrieve_node_line/retrieval/1.parquet +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/2/retrieve_node_line/retrieval/best_0.parquet +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/2/retrieve_node_line/retrieval/summary.csv +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/2/retrieve_node_line/summary.csv +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/3/config.yaml +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/best.yaml +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/data/corpus.parquet +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/data/qa.parquet +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/resources/bm25_porter_stemmer.pkl +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/resources/chroma/6595d304-5270-4d9f-a267-c191366804ce/data_level0.bin +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/resources/chroma/6595d304-5270-4d9f-a267-c191366804ce/header.bin +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/resources/chroma/6595d304-5270-4d9f-a267-c191366804ce/length.bin +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/resources/chroma/6595d304-5270-4d9f-a267-c191366804ce/link_lists.bin +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/resources/chroma/chroma.sqlite3 +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/result_project/trial.json +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/sample_contents_nqa.csv +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/sample_project/data/corpus.parquet +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/sample_project/data/qa.parquet +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/sample_project/resources/bm25_gpt2.pkl +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/sample_project/resources/bm25_porter_stemmer.pkl +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/simple.yaml +0 -0
- {autorag-0.2.13 → autorag-0.2.14}/tests/resources/test_bm25_retrieval.pkl +0 -0
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: Bug report
|
|
3
|
+
about: Create a report to help us improve
|
|
4
|
+
title: "[BUG] "
|
|
5
|
+
labels: bug
|
|
6
|
+
assignees: ''
|
|
7
|
+
|
|
8
|
+
---
|
|
9
|
+
|
|
10
|
+
**Describe the bug**
|
|
11
|
+
A clear and concise description of what the bug is.
|
|
12
|
+
|
|
13
|
+
**To Reproduce**
|
|
14
|
+
Steps to reproduce the behavior:
|
|
15
|
+
1. Go to '...'
|
|
16
|
+
2. Click on '....'
|
|
17
|
+
3. Scroll down to '....'
|
|
18
|
+
4. See error
|
|
19
|
+
|
|
20
|
+
**Expected behavior**
|
|
21
|
+
A clear and concise description of what you expected to happen.
|
|
22
|
+
|
|
23
|
+
**Full Error log**
|
|
24
|
+
If applicable, add full error log to help explain your problem.
|
|
25
|
+
|
|
26
|
+
**Code that bug is happened**
|
|
27
|
+
If applicable, add the code that bug is happened.
|
|
28
|
+
(Especially, your AutoRAG YAML file or python codes that you wrote)
|
|
29
|
+
|
|
30
|
+
**Desktop (please complete the following information):**
|
|
31
|
+
- OS: [e.g. Windows, Linux, MacOS]
|
|
32
|
+
- Python version [e.g. 3.10]
|
|
33
|
+
|
|
34
|
+
**Additional context**
|
|
35
|
+
Add any other context about the problem here.
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: Feature request
|
|
3
|
+
about: Suggest an idea for this project
|
|
4
|
+
title: "[Feature Request]"
|
|
5
|
+
labels: enhancement
|
|
6
|
+
assignees: ''
|
|
7
|
+
|
|
8
|
+
---
|
|
9
|
+
|
|
10
|
+
**Is your feature request related to a problem? Please describe.**
|
|
11
|
+
A clear and concise description of what the problem is. Ex. I'm always frustrated when [...]
|
|
12
|
+
|
|
13
|
+
**Describe the solution you'd like**
|
|
14
|
+
A clear and concise description of what you want to happen.
|
|
15
|
+
|
|
16
|
+
**Describe alternatives you've considered**
|
|
17
|
+
A clear and concise description of any alternative solutions or features you've considered.
|
|
18
|
+
|
|
19
|
+
**Additional context**
|
|
20
|
+
Add any other context or screenshots about the feature request here.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: AutoRAG
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.14
|
|
4
4
|
Summary: Automatically Evaluate RAG pipelines with your own data. Find optimal structure for new RAG product.
|
|
5
5
|
Author-email: Marker-Inc <vkehfdl1@gmail.com>
|
|
6
6
|
License: Apache License
|
|
@@ -221,13 +221,12 @@ Requires-Python: >=3.8
|
|
|
221
221
|
Description-Content-Type: text/markdown
|
|
222
222
|
License-File: LICENSE
|
|
223
223
|
Requires-Dist: numpy<2.0.0
|
|
224
|
-
Requires-Dist: pandas
|
|
224
|
+
Requires-Dist: pandas>=2.1.0
|
|
225
225
|
Requires-Dist: tqdm
|
|
226
226
|
Requires-Dist: tiktoken>=0.7.0
|
|
227
227
|
Requires-Dist: openai>=1.0.0
|
|
228
228
|
Requires-Dist: rank_bm25
|
|
229
229
|
Requires-Dist: transformers
|
|
230
|
-
Requires-Dist: swifter
|
|
231
230
|
Requires-Dist: pyyaml
|
|
232
231
|
Requires-Dist: pyarrow
|
|
233
232
|
Requires-Dist: fastparquet
|
|
@@ -242,7 +241,7 @@ Requires-Dist: uvicorn
|
|
|
242
241
|
Requires-Dist: torch
|
|
243
242
|
Requires-Dist: sentencepiece
|
|
244
243
|
Requires-Dist: guidance
|
|
245
|
-
Requires-Dist: cohere
|
|
244
|
+
Requires-Dist: cohere>=5.8.0
|
|
246
245
|
Requires-Dist: tokenlog>=0.0.2
|
|
247
246
|
Requires-Dist: aiohttp
|
|
248
247
|
Requires-Dist: bert_score
|
|
@@ -250,6 +249,7 @@ Requires-Dist: sentence-transformers
|
|
|
250
249
|
Requires-Dist: FlagEmbedding
|
|
251
250
|
Requires-Dist: ragas
|
|
252
251
|
Requires-Dist: llmlingua
|
|
252
|
+
Requires-Dist: peft
|
|
253
253
|
Requires-Dist: llama-index>=0.10.1
|
|
254
254
|
Requires-Dist: llama-index-core>=0.10.1
|
|
255
255
|
Requires-Dist: llama-index-readers-file
|
|
@@ -1,11 +1,10 @@
|
|
|
1
1
|
numpy<2.0.0
|
|
2
|
-
pandas
|
|
2
|
+
pandas>=2.1.0
|
|
3
3
|
tqdm
|
|
4
4
|
tiktoken>=0.7.0
|
|
5
5
|
openai>=1.0.0
|
|
6
6
|
rank_bm25
|
|
7
7
|
transformers
|
|
8
|
-
swifter
|
|
9
8
|
pyyaml
|
|
10
9
|
pyarrow
|
|
11
10
|
fastparquet
|
|
@@ -20,7 +19,7 @@ uvicorn
|
|
|
20
19
|
torch
|
|
21
20
|
sentencepiece
|
|
22
21
|
guidance
|
|
23
|
-
cohere
|
|
22
|
+
cohere>=5.8.0
|
|
24
23
|
tokenlog>=0.0.2
|
|
25
24
|
aiohttp
|
|
26
25
|
bert_score
|
|
@@ -28,6 +27,7 @@ sentence-transformers
|
|
|
28
27
|
FlagEmbedding
|
|
29
28
|
ragas
|
|
30
29
|
llmlingua
|
|
30
|
+
peft
|
|
31
31
|
llama-index>=0.10.1
|
|
32
32
|
llama-index-core>=0.10.1
|
|
33
33
|
llama-index-readers-file
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: AutoRAG
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.14
|
|
4
4
|
Summary: Automatically Evaluate RAG pipelines with your own data. Find optimal structure for new RAG product.
|
|
5
5
|
Author-email: Marker-Inc <vkehfdl1@gmail.com>
|
|
6
6
|
License: Apache License
|
|
@@ -221,13 +221,12 @@ Requires-Python: >=3.8
|
|
|
221
221
|
Description-Content-Type: text/markdown
|
|
222
222
|
License-File: LICENSE
|
|
223
223
|
Requires-Dist: numpy<2.0.0
|
|
224
|
-
Requires-Dist: pandas
|
|
224
|
+
Requires-Dist: pandas>=2.1.0
|
|
225
225
|
Requires-Dist: tqdm
|
|
226
226
|
Requires-Dist: tiktoken>=0.7.0
|
|
227
227
|
Requires-Dist: openai>=1.0.0
|
|
228
228
|
Requires-Dist: rank_bm25
|
|
229
229
|
Requires-Dist: transformers
|
|
230
|
-
Requires-Dist: swifter
|
|
231
230
|
Requires-Dist: pyyaml
|
|
232
231
|
Requires-Dist: pyarrow
|
|
233
232
|
Requires-Dist: fastparquet
|
|
@@ -242,7 +241,7 @@ Requires-Dist: uvicorn
|
|
|
242
241
|
Requires-Dist: torch
|
|
243
242
|
Requires-Dist: sentencepiece
|
|
244
243
|
Requires-Dist: guidance
|
|
245
|
-
Requires-Dist: cohere
|
|
244
|
+
Requires-Dist: cohere>=5.8.0
|
|
246
245
|
Requires-Dist: tokenlog>=0.0.2
|
|
247
246
|
Requires-Dist: aiohttp
|
|
248
247
|
Requires-Dist: bert_score
|
|
@@ -250,6 +249,7 @@ Requires-Dist: sentence-transformers
|
|
|
250
249
|
Requires-Dist: FlagEmbedding
|
|
251
250
|
Requires-Dist: ragas
|
|
252
251
|
Requires-Dist: llmlingua
|
|
252
|
+
Requires-Dist: peft
|
|
253
253
|
Requires-Dist: llama-index>=0.10.1
|
|
254
254
|
Requires-Dist: llama-index-core>=0.10.1
|
|
255
255
|
Requires-Dist: llama-index-readers-file
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
0.2.14
|
|
@@ -9,19 +9,16 @@ from llama_index.embeddings.huggingface import HuggingFaceEmbedding
|
|
|
9
9
|
from llama_index.embeddings.openai import OpenAIEmbedding
|
|
10
10
|
from llama_index.embeddings.openai import OpenAIEmbeddingModelType
|
|
11
11
|
from llama_index.llms.huggingface import HuggingFaceLLM
|
|
12
|
-
from llama_index.llms.openai import OpenAI
|
|
13
12
|
from llama_index.llms.ollama import Ollama
|
|
13
|
+
from llama_index.llms.openai import OpenAI
|
|
14
14
|
from llama_index.llms.openai_like import OpenAILike
|
|
15
15
|
from rich.logging import RichHandler
|
|
16
|
-
from swifter import set_defaults
|
|
17
16
|
|
|
18
17
|
version_path = os.path.join(os.path.dirname(os.path.realpath(__file__)), 'VERSION')
|
|
19
18
|
|
|
20
19
|
with open(version_path, 'r') as f:
|
|
21
20
|
__version__ = f.read().strip()
|
|
22
21
|
|
|
23
|
-
set_defaults(allow_dask_on_strings=True)
|
|
24
|
-
|
|
25
22
|
|
|
26
23
|
class LazyInit:
|
|
27
24
|
def __init__(self, factory, *args, **kwargs):
|
|
@@ -50,7 +47,8 @@ embedding_models = {
|
|
|
50
47
|
'huggingface_cointegrated_rubert_tiny2': LazyInit(HuggingFaceEmbedding, model_name="cointegrated/rubert-tiny2"),
|
|
51
48
|
'huggingface_all_mpnet_base_v2': LazyInit(HuggingFaceEmbedding,
|
|
52
49
|
model_name="sentence-transformers/all-mpnet-base-v2",
|
|
53
|
-
max_length=512, )
|
|
50
|
+
max_length=512, ),
|
|
51
|
+
'huggingface_bge_m3': LazyInit(HuggingFaceEmbedding, model_name="BAAI/bge-m3"),
|
|
54
52
|
}
|
|
55
53
|
|
|
56
54
|
generator_models = {
|
|
@@ -0,0 +1,168 @@
|
|
|
1
|
+
import logging
|
|
2
|
+
import uuid
|
|
3
|
+
from typing import Callable, Optional, List
|
|
4
|
+
|
|
5
|
+
import chromadb
|
|
6
|
+
import pandas as pd
|
|
7
|
+
from tqdm import tqdm
|
|
8
|
+
|
|
9
|
+
import autorag
|
|
10
|
+
from autorag.nodes.retrieval.vectordb import vectordb_ingest, vectordb
|
|
11
|
+
from autorag.utils.util import save_parquet_safe, fetch_contents
|
|
12
|
+
|
|
13
|
+
logger = logging.getLogger("AutoRAG")
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def make_single_content_qa(corpus_df: pd.DataFrame,
|
|
17
|
+
content_size: int,
|
|
18
|
+
qa_creation_func: Callable,
|
|
19
|
+
output_filepath: Optional[str] = None,
|
|
20
|
+
upsert: bool = False,
|
|
21
|
+
random_state: int = 42,
|
|
22
|
+
cache_batch: int = 32,
|
|
23
|
+
**kwargs) -> pd.DataFrame:
|
|
24
|
+
"""
|
|
25
|
+
Make single content (single-hop, single-document) QA dataset using given qa_creation_func.
|
|
26
|
+
It generates a single content QA dataset, which means its retrieval ground truth will be only one.
|
|
27
|
+
It is the most basic form of QA dataset.
|
|
28
|
+
|
|
29
|
+
:param corpus_df: The corpus dataframe to make QA dataset from.
|
|
30
|
+
:param content_size: This function will generate QA dataset for the given number of contents.
|
|
31
|
+
:param qa_creation_func: The function to create QA pairs.
|
|
32
|
+
You can use like `generate_qa_llama_index` or `generate_qa_llama_index_by_ratio`.
|
|
33
|
+
The input func must have `contents` parameter for the list of content string.
|
|
34
|
+
:param output_filepath: Optional filepath to save the parquet file.
|
|
35
|
+
If None, the function will return the processed_data as pd.DataFrame, but do not save as parquet.
|
|
36
|
+
File directory must exist. File extension must be .parquet
|
|
37
|
+
:param upsert: If true, the function will overwrite the existing file if it exists.
|
|
38
|
+
Default is False.
|
|
39
|
+
:param random_state: The random state for sampling corpus from the given corpus_df.
|
|
40
|
+
:param cache_batch: The number of batches to use for caching the generated QA dataset.
|
|
41
|
+
When the cache_batch size data is generated, the dataset will save to the designated output_filepath.
|
|
42
|
+
If the cache_batch size is too small, the process time will be longer.
|
|
43
|
+
:param kwargs: The keyword arguments for qa_creation_func.
|
|
44
|
+
:return: QA dataset dataframe.
|
|
45
|
+
You can save this as parquet file to use at AutoRAG.
|
|
46
|
+
"""
|
|
47
|
+
assert content_size > 0, "content_size must be greater than 0."
|
|
48
|
+
if content_size > len(corpus_df):
|
|
49
|
+
logger.warning(f"content_size {content_size} is larger than the corpus size {len(corpus_df)}. "
|
|
50
|
+
"Setting content_size to the corpus size.")
|
|
51
|
+
content_size = len(corpus_df)
|
|
52
|
+
sampled_corpus = corpus_df.sample(n=content_size, random_state=random_state)
|
|
53
|
+
sampled_corpus = sampled_corpus.reset_index(drop=True)
|
|
54
|
+
|
|
55
|
+
def make_query_generation_gt(row):
|
|
56
|
+
return row['qa']['query'], row['qa']['generation_gt']
|
|
57
|
+
|
|
58
|
+
qa_data = pd.DataFrame()
|
|
59
|
+
for idx, i in tqdm(enumerate(range(0, len(sampled_corpus), cache_batch))):
|
|
60
|
+
qa = qa_creation_func(contents=sampled_corpus['contents'].tolist()[i:i + cache_batch], **kwargs)
|
|
61
|
+
|
|
62
|
+
temp_qa_data = pd.DataFrame({
|
|
63
|
+
'qa': qa,
|
|
64
|
+
'retrieval_gt': sampled_corpus['doc_id'].tolist()[i:i + cache_batch],
|
|
65
|
+
})
|
|
66
|
+
temp_qa_data = temp_qa_data.explode('qa', ignore_index=True)
|
|
67
|
+
temp_qa_data['qid'] = [str(uuid.uuid4()) for _ in range(len(temp_qa_data))]
|
|
68
|
+
temp_qa_data[['query', 'generation_gt']] = temp_qa_data.apply(make_query_generation_gt, axis=1,
|
|
69
|
+
result_type='expand')
|
|
70
|
+
temp_qa_data = temp_qa_data.drop(columns=['qa'])
|
|
71
|
+
|
|
72
|
+
temp_qa_data['retrieval_gt'] = temp_qa_data['retrieval_gt'].apply(lambda x: [[x]])
|
|
73
|
+
temp_qa_data['generation_gt'] = temp_qa_data['generation_gt'].apply(lambda x: [x])
|
|
74
|
+
|
|
75
|
+
if idx == 0:
|
|
76
|
+
qa_data = temp_qa_data
|
|
77
|
+
else:
|
|
78
|
+
qa_data = pd.concat([qa_data, temp_qa_data], ignore_index=True)
|
|
79
|
+
if output_filepath is not None:
|
|
80
|
+
save_parquet_safe(qa_data, output_filepath, upsert=upsert)
|
|
81
|
+
|
|
82
|
+
return qa_data
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def make_qa_with_existing_queries(
|
|
86
|
+
corpus_df: pd.DataFrame,
|
|
87
|
+
existing_query_df: pd.DataFrame,
|
|
88
|
+
content_size: int,
|
|
89
|
+
answer_creation_func: Callable,
|
|
90
|
+
output_filepath: Optional[str] = None,
|
|
91
|
+
embedding_model: str = 'openai_embed_3_large',
|
|
92
|
+
collection: Optional[chromadb.Collection] = None,
|
|
93
|
+
upsert: bool = False,
|
|
94
|
+
random_state: int = 42,
|
|
95
|
+
cache_batch: int = 32,
|
|
96
|
+
top_k: int = 3,
|
|
97
|
+
**kwargs
|
|
98
|
+
) -> pd.DataFrame:
|
|
99
|
+
"""
|
|
100
|
+
Make single-hop QA dataset using given qa_creation_func and existing queries.
|
|
101
|
+
|
|
102
|
+
:param corpus_df: The corpus dataframe to make QA dataset from.
|
|
103
|
+
:param existing_query_df: Dataframe containing existing queries to use for QA pair creation.
|
|
104
|
+
:param content_size: This function will generate QA dataset for the given number of contents.
|
|
105
|
+
:param answer_creation_func: The function to create answer with input query.
|
|
106
|
+
:param output_filepath: Optional filepath to save the parquet file.
|
|
107
|
+
:param embedding_model: The embedding model to use for vectorization.
|
|
108
|
+
You can add your own embedding model in the autorag.embedding_models.
|
|
109
|
+
Please refer to how to add an embedding model in this doc: https://docs.auto-rag.com/local_model.html
|
|
110
|
+
The default is 'openai_embed_3_large'.
|
|
111
|
+
:param collection: The chromadb collection to use for vector DB.
|
|
112
|
+
You can make any chromadb collection and use it here.
|
|
113
|
+
If you already ingested the corpus_df to the collection, the embedding process will not be repeated.
|
|
114
|
+
The default is None. If None, it makes a temporary collection.
|
|
115
|
+
:param upsert: If true, the function will overwrite the existing file if it exists.
|
|
116
|
+
:param random_state: The random state for sampling corpus from the given corpus_df.
|
|
117
|
+
:param cache_batch: The number of batches to use for caching the generated QA dataset.
|
|
118
|
+
:param top_k: The number of sources to refer by model.
|
|
119
|
+
Default is 3.
|
|
120
|
+
:param kwargs: The keyword arguments for qa_creation_func.
|
|
121
|
+
:return: QA dataset dataframe.
|
|
122
|
+
"""
|
|
123
|
+
assert 'query' in existing_query_df.columns, "existing_query_df must have 'query' column."
|
|
124
|
+
assert content_size > 0, "content_size must be greater than 0."
|
|
125
|
+
if content_size > len(corpus_df):
|
|
126
|
+
logger.warning(f"content_size {content_size} is larger than the corpus size {len(corpus_df)}. "
|
|
127
|
+
"Setting content_size to the corpus size.")
|
|
128
|
+
content_size = len(corpus_df)
|
|
129
|
+
|
|
130
|
+
logger.info("Loading local embedding model...")
|
|
131
|
+
embeddings = autorag.embedding_models[embedding_model]
|
|
132
|
+
|
|
133
|
+
# Vector DB creation
|
|
134
|
+
if collection is None:
|
|
135
|
+
chroma_client = chromadb.Client()
|
|
136
|
+
collection_name = "auto-rag"
|
|
137
|
+
collection = chroma_client.get_or_create_collection(collection_name)
|
|
138
|
+
|
|
139
|
+
# embed corpus_df
|
|
140
|
+
vectordb_ingest(collection, corpus_df, embeddings)
|
|
141
|
+
vectordb_func = vectordb.__wrapped__
|
|
142
|
+
retrieved_ids, retrieve_scores = vectordb_func(existing_query_df['query'].tolist(), top_k, collection, embeddings)
|
|
143
|
+
|
|
144
|
+
retrieved_contents: List[List[str]] = fetch_contents(corpus_df, retrieved_ids)
|
|
145
|
+
input_passage_strs: List[str] = list(map(
|
|
146
|
+
lambda x: '\n'.join([f"Document {i + 1}\n{content}" for i, content in enumerate(x)]),
|
|
147
|
+
retrieved_contents))
|
|
148
|
+
retrieved_qa_df = pd.DataFrame({
|
|
149
|
+
'qid': [str(uuid.uuid4()) for _ in range(len(existing_query_df))],
|
|
150
|
+
'query': existing_query_df['query'].tolist(),
|
|
151
|
+
'retrieval_gt': list(map(lambda x: [x], retrieved_ids)),
|
|
152
|
+
'input_passage_str': input_passage_strs,
|
|
153
|
+
})
|
|
154
|
+
|
|
155
|
+
sample_qa_df = retrieved_qa_df.sample(n=min(content_size, len(retrieved_qa_df)), random_state=random_state)
|
|
156
|
+
|
|
157
|
+
generation_gt = answer_creation_func(contents=sample_qa_df['input_passage_str'].tolist(),
|
|
158
|
+
queries=sample_qa_df['query'].tolist(),
|
|
159
|
+
batch=cache_batch,
|
|
160
|
+
**kwargs)
|
|
161
|
+
qa_df = sample_qa_df.copy(deep=True)
|
|
162
|
+
qa_df.drop(columns=['input_passage_str'], inplace=True)
|
|
163
|
+
qa_df['generation_gt'] = generation_gt
|
|
164
|
+
|
|
165
|
+
if output_filepath is not None:
|
|
166
|
+
save_parquet_safe(qa_df, output_filepath, upsert=upsert)
|
|
167
|
+
|
|
168
|
+
return qa_df
|
|
@@ -4,6 +4,7 @@ import random
|
|
|
4
4
|
from typing import Optional, List, Dict, Any
|
|
5
5
|
|
|
6
6
|
import pandas as pd
|
|
7
|
+
from llama_index.core.base.llms.types import ChatMessage, MessageRole
|
|
7
8
|
from llama_index.core.service_context_elements.llm_predictor import LLMPredictorType
|
|
8
9
|
|
|
9
10
|
from autorag.utils.util import process_batch
|
|
@@ -53,6 +54,31 @@ def generate_qa_llama_index(
|
|
|
53
54
|
return results
|
|
54
55
|
|
|
55
56
|
|
|
57
|
+
def generate_answers(
|
|
58
|
+
llm: LLMPredictorType,
|
|
59
|
+
contents: List[str],
|
|
60
|
+
queries: List[str],
|
|
61
|
+
batch: int = 4,
|
|
62
|
+
) -> List[List[Dict]]:
|
|
63
|
+
"""
|
|
64
|
+
Generate qa sets from the list of contents using existing queries.
|
|
65
|
+
|
|
66
|
+
:param llm: Llama index model
|
|
67
|
+
:param contents: List of content strings.
|
|
68
|
+
:param queries: List of existing queries.
|
|
69
|
+
:param batch: The batch size to process asynchronously.
|
|
70
|
+
:return: 2-d list of dictionaries containing the query and generation_gt.
|
|
71
|
+
"""
|
|
72
|
+
|
|
73
|
+
tasks = [
|
|
74
|
+
generate_basic_answer(llm, content, query)
|
|
75
|
+
for content, query in zip(contents, queries)
|
|
76
|
+
]
|
|
77
|
+
loops = asyncio.get_event_loop()
|
|
78
|
+
results = loops.run_until_complete(process_batch(tasks, batch))
|
|
79
|
+
return results
|
|
80
|
+
|
|
81
|
+
|
|
56
82
|
def generate_qa_llama_index_by_ratio(
|
|
57
83
|
llm: LLMPredictorType,
|
|
58
84
|
contents: List[str],
|
|
@@ -147,6 +173,21 @@ async def async_qa_gen_llama_index(
|
|
|
147
173
|
return await generate(content, llm)
|
|
148
174
|
|
|
149
175
|
|
|
176
|
+
async def generate_basic_answer(llm: LLMPredictorType, passage_str: str, query: str) -> str:
|
|
177
|
+
basic_answer_system_prompt = """You are an AI assistant to answer the given question in the provide evidence text.
|
|
178
|
+
You can find the evidence from the given text about question, and you have to write a proper answer to the given question.
|
|
179
|
+
You have to preserve the question's language at the answer.
|
|
180
|
+
For example, if the input question is Korean, the output answer must be in Korean.
|
|
181
|
+
"""
|
|
182
|
+
user_prompt = f"Text:\n<|text_start|>\n{passage_str}\n<|text_end|>\n\nQuestion:\n{query}\n\nAnswer:"
|
|
183
|
+
|
|
184
|
+
response = await llm.achat(messages=[
|
|
185
|
+
ChatMessage(role=MessageRole.SYSTEM, content=basic_answer_system_prompt),
|
|
186
|
+
ChatMessage(role=MessageRole.USER, content=user_prompt)
|
|
187
|
+
], temperature=1.0)
|
|
188
|
+
return response.message.content
|
|
189
|
+
|
|
190
|
+
|
|
150
191
|
def validate_llama_index_prompt(prompt: str) -> bool:
|
|
151
192
|
"""
|
|
152
193
|
Validate the prompt for the llama index model.
|
|
@@ -4,6 +4,7 @@ import uuid
|
|
|
4
4
|
from copy import deepcopy
|
|
5
5
|
from typing import Optional, Dict, List
|
|
6
6
|
|
|
7
|
+
import nest_asyncio
|
|
7
8
|
import pandas as pd
|
|
8
9
|
import uvicorn
|
|
9
10
|
import yaml
|
|
@@ -65,7 +66,7 @@ def summary_df_to_yaml(summary_df: pd.DataFrame, config_dict: Dict) -> Dict:
|
|
|
65
66
|
summary_df['categorical_node_line_name'] = pd.Categorical(summary_df['node_line_name'], categories=node_line_names,
|
|
66
67
|
ordered=True)
|
|
67
68
|
summary_df = summary_df.sort_values(by='categorical_node_line_name')
|
|
68
|
-
grouped = summary_df.groupby('categorical_node_line_name')
|
|
69
|
+
grouped = summary_df.groupby('categorical_node_line_name', observed=False)
|
|
69
70
|
|
|
70
71
|
node_lines = [
|
|
71
72
|
{
|
|
@@ -207,7 +208,7 @@ class Runner:
|
|
|
207
208
|
|
|
208
209
|
{
|
|
209
210
|
"Query": "your query",
|
|
210
|
-
"result_column": "
|
|
211
|
+
"result_column": "generated_texts"
|
|
211
212
|
}
|
|
212
213
|
|
|
213
214
|
And it returns json response like below:
|
|
@@ -222,10 +223,11 @@ class Runner:
|
|
|
222
223
|
:param port: The port of the api server.
|
|
223
224
|
:param kwargs: Other arguments for uvicorn.run.
|
|
224
225
|
"""
|
|
226
|
+
nest_asyncio.apply()
|
|
225
227
|
logger.info(f"Run api server at {host}:{port}")
|
|
226
|
-
uvicorn.run(self.app, host=host, port=port, **kwargs)
|
|
228
|
+
uvicorn.run(self.app, host=host, port=port, loop="asyncio", **kwargs)
|
|
227
229
|
|
|
228
230
|
|
|
229
231
|
class RunnerInput(BaseModel):
|
|
230
232
|
query: str
|
|
231
|
-
result_column: str = "
|
|
233
|
+
result_column: str = "generated_texts"
|
|
@@ -5,6 +5,7 @@ import os
|
|
|
5
5
|
from typing import List, Optional
|
|
6
6
|
|
|
7
7
|
import evaluate
|
|
8
|
+
import nltk
|
|
8
9
|
import pandas as pd
|
|
9
10
|
import torch
|
|
10
11
|
from llama_index.core.embeddings import BaseEmbedding
|
|
@@ -109,6 +110,7 @@ def meteor(generation_gt: List[List[str]], generations: List[str],
|
|
|
109
110
|
Default is 0.5.
|
|
110
111
|
:return: A list of computed metric scores.
|
|
111
112
|
"""
|
|
113
|
+
nltk.download('punkt_tab')
|
|
112
114
|
meteor_instance = evaluate.load("meteor")
|
|
113
115
|
result = huggingface_evaluate(meteor_instance, 'meteor', generation_gt, generations,
|
|
114
116
|
alpha=alpha, beta=beta, gamma=gamma)
|
|
@@ -44,7 +44,8 @@ def passage_augmenter_node(func):
|
|
|
44
44
|
if func.__name__ == 'prev_next_augmenter':
|
|
45
45
|
corpus_df = cast_corpus_dataset(corpus_df)
|
|
46
46
|
slim_corpus_df = corpus_df[["doc_id", "metadata"]]
|
|
47
|
-
slim_corpus_df['metadata'] = slim_corpus_df['metadata'].apply(
|
|
47
|
+
slim_corpus_df.loc[:, 'metadata'] = slim_corpus_df['metadata'].apply(
|
|
48
|
+
filter_dict_keys, keys=['prev_id', 'next_id'])
|
|
48
49
|
|
|
49
50
|
mode = kwargs.pop("mode", 'both')
|
|
50
51
|
num_passages = kwargs.pop("num_passages", 1)
|
|
@@ -1,8 +1,7 @@
|
|
|
1
1
|
from typing import List, Tuple
|
|
2
2
|
|
|
3
|
-
import numpy as np
|
|
4
|
-
|
|
5
3
|
from autorag.nodes.passagefilter.base import passage_filter_node
|
|
4
|
+
from autorag.utils.util import convert_inputs_to_list
|
|
6
5
|
|
|
7
6
|
|
|
8
7
|
@passage_filter_node
|
|
@@ -32,6 +31,7 @@ def threshold_cutoff(queries: List[str], contents_list: List[List[str]],
|
|
|
32
31
|
return remain_content_list, remain_ids_list, remain_scores_list
|
|
33
32
|
|
|
34
33
|
|
|
34
|
+
@convert_inputs_to_list
|
|
35
35
|
def threshold_cutoff_pure(scores_list: List[float],
|
|
36
36
|
threshold: float,
|
|
37
37
|
reverse: bool = False) -> List[int]:
|
|
@@ -45,20 +45,13 @@ def threshold_cutoff_pure(scores_list: List[float],
|
|
|
45
45
|
Default is False.
|
|
46
46
|
:return: Indices to remain at the contents
|
|
47
47
|
"""
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
remain_indices = np.where(scores_list >= threshold)[0]
|
|
54
|
-
default_index = np.argmax(scores_list)
|
|
48
|
+
assert isinstance(scores_list, list), "scores_list must be a list."
|
|
49
|
+
|
|
50
|
+
if reverse:
|
|
51
|
+
remain_indices = [i for i, score in enumerate(scores_list) if score <= threshold]
|
|
52
|
+
default_index = scores_list.index(min(scores_list))
|
|
55
53
|
else:
|
|
56
|
-
if
|
|
57
|
-
|
|
58
|
-
default_index = scores_list.index(min(scores_list))
|
|
59
|
-
else:
|
|
60
|
-
remain_indices = [i for i, score in enumerate(scores_list) if score >= threshold]
|
|
61
|
-
default_index = scores_list.index(max(scores_list))
|
|
54
|
+
remain_indices = [i for i, score in enumerate(scores_list) if score >= threshold]
|
|
55
|
+
default_index = scores_list.index(max(scores_list))
|
|
62
56
|
|
|
63
|
-
return remain_indices
|
|
64
|
-
remain_indices if remain_indices else [default_index]
|
|
57
|
+
return remain_indices if remain_indices else [default_index]
|
|
@@ -55,7 +55,7 @@ def retrieval_node(func):
|
|
|
55
55
|
assert "query" in previous_result.columns, "previous_result must have query column."
|
|
56
56
|
if "queries" not in previous_result.columns:
|
|
57
57
|
previous_result["queries"] = previous_result["query"]
|
|
58
|
-
previous_result["queries"] = previous_result["queries"].apply(cast_queries)
|
|
58
|
+
previous_result.loc[:, "queries"] = previous_result["queries"].apply(cast_queries)
|
|
59
59
|
queries = previous_result["queries"].tolist()
|
|
60
60
|
|
|
61
61
|
# run retrieval function
|
|
@@ -127,7 +127,7 @@ def select_best_rr(results: List[pd.DataFrame], columns: Iterable[str],
|
|
|
127
127
|
results, columns, metadatas = validate_strategy_inputs(results, columns, metadatas)
|
|
128
128
|
each_average_df = pd.DataFrame([df[columns].mean(axis=0).to_dict() for df in results])
|
|
129
129
|
rank_df = each_average_df.rank(ascending=False)
|
|
130
|
-
rr_df = rank_df.
|
|
130
|
+
rr_df = rank_df.map(lambda x: 1 / x)
|
|
131
131
|
best_index = np.array(rr_df.sum(axis=1)).argmax()
|
|
132
132
|
return results[best_index], metadatas[best_index]
|
|
133
133
|
|
|
@@ -96,7 +96,7 @@ def load_summary_file(summary_path: str,
|
|
|
96
96
|
raise ValueError(f"Malformed dict received : {elem}\nCan't convert to dict properly")
|
|
97
97
|
return {'threshold': date_object}
|
|
98
98
|
|
|
99
|
-
summary_df[dict_columns] = summary_df[dict_columns].
|
|
99
|
+
summary_df[dict_columns] = summary_df[dict_columns].map(convert_dict)
|
|
100
100
|
return summary_df
|
|
101
101
|
|
|
102
102
|
|
|
@@ -14,6 +14,7 @@ myst:
|
|
|
14
14
|
3. [Corpus data to QA data](#make-qa-data-from-corpus-data)
|
|
15
15
|
4. [Use custom prompt](#use-custom-prompt)
|
|
16
16
|
5. [Use multiple prompts](#use-multiple-prompts)
|
|
17
|
+
6. [If there are existing queries](#when-you-have-existing-queries)
|
|
17
18
|
|
|
18
19
|
|
|
19
20
|
## Overview
|
|
@@ -167,3 +168,48 @@ qa_df = make_single_content_qa(corpus_df, content_size=50, qa_creation_func=gene
|
|
|
167
168
|
```{warning}
|
|
168
169
|
Remeber all prompts must have the placeholders `{{text}}` and `{{num_questions}}`.
|
|
169
170
|
```
|
|
171
|
+
|
|
172
|
+
## When you have existing queries
|
|
173
|
+
|
|
174
|
+
When you have existing queries, you can use it for AutoRAG.
|
|
175
|
+
The real user's question is valuable data, so it is always great to use it prior to generating synthetic data.
|
|
176
|
+
|
|
177
|
+
But you have to make retrieval_gt for existing queries from your corpus data.
|
|
178
|
+
The process to find the retrieval_gt at the corpus is hard, but must be accurate.
|
|
179
|
+
For making it less hard, we use an embedding model and vectordb for finding relevant passages.
|
|
180
|
+
After that, you have to clarify the retrieval_gt is right.
|
|
181
|
+
If retrieval_gt is not relevant, you have to remove it on the dataset.
|
|
182
|
+
|
|
183
|
+
```python
|
|
184
|
+
import pandas as pd
|
|
185
|
+
from llama_index.llms.openai import OpenAI
|
|
186
|
+
from autorag.data.qacreation import make_qa_with_existing_queries, generate_answers
|
|
187
|
+
|
|
188
|
+
corpus_df = pd.read_parquet('path/to/corpus.parquet')
|
|
189
|
+
existing_qa_df = pd.read_parquet('path/to/existing_qa.parquet') # It have to contain 'query' column
|
|
190
|
+
llm = OpenAI(model='gpt-3.5-turbo', temperature=1.0)
|
|
191
|
+
qa_df = make_qa_with_existing_queries(corpus_df, existing_qa_df, content_size=50,
|
|
192
|
+
answer_creation_func=generate_answers,
|
|
193
|
+
llm=llm, output_filepath='path/to/qa.parquet', cache_batch=64,
|
|
194
|
+
embedding_model='openai_embed_3_large', top_k=5)
|
|
195
|
+
```
|
|
196
|
+
|
|
197
|
+
You can use `PersistentClient` for saving corpus embeddings locally as well.
|
|
198
|
+
|
|
199
|
+
```python
|
|
200
|
+
import pandas as pd
|
|
201
|
+
import chromadb
|
|
202
|
+
from llama_index.llms.openai import OpenAI
|
|
203
|
+
from autorag.data.qacreation import make_qa_with_existing_queries, generate_answers
|
|
204
|
+
|
|
205
|
+
client = chromadb.PersistentClient('path/to/chromadb')
|
|
206
|
+
collection = client.get_or_create_collection('auto-rag')
|
|
207
|
+
|
|
208
|
+
corpus_df = pd.read_parquet('path/to/corpus.parquet')
|
|
209
|
+
existing_qa_df = pd.read_parquet('path/to/existing_qa.parquet') # It have to contain 'query' column
|
|
210
|
+
llm = OpenAI(model='gpt-3.5-turbo', temperature=1.0)
|
|
211
|
+
qa_df = make_qa_with_existing_queries(corpus_df, existing_qa_df, content_size=50,
|
|
212
|
+
answer_creation_func=generate_answers, collection=collection,
|
|
213
|
+
llm=llm, output_filepath='path/to/qa.parquet', cache_batch=64,
|
|
214
|
+
embedding_model='openai_embed_3_large', top_k=5)
|
|
215
|
+
```
|
|
@@ -117,6 +117,7 @@ To change the embedding model, you can change the `embedding_model` parameter to
|
|
|
117
117
|
| [BAAI/bge-small-en-v1.5](https://huggingface.co/BAAI/bge-small-en-v1.5) | huggingface_baai_bge_small |
|
|
118
118
|
| [cointegrated/rubert-tiny2](https://huggingface.co/cointegrated/rubert-tiny2) | huggingface_cointegrated_rubert_tiny2 |
|
|
119
119
|
| [sentence-transformers/all-mpnet-base-v2](https://huggingface.co/sentence-transformers/all-mpnet-base-v2) | huggingface_all_mpnet_base_v2 |
|
|
120
|
+
| [BAAI/bge-m3](https://huggingface.co/BAAI/bge-m3) | huggingface_bge_m3 |
|
|
120
121
|
|
|
121
122
|
For example, if you want to use OpenAI text embedding large model, you can set `embedding_model` parameter
|
|
122
123
|
to `openai_embed_3_large`.
|