AutoRAG 0.2.12__tar.gz → 0.2.13__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {autorag-0.2.12 → autorag-0.2.13}/.github/workflows/test.yml +5 -1
- {autorag-0.2.12 → autorag-0.2.13}/AutoRAG.egg-info/PKG-INFO +7 -4
- {autorag-0.2.12 → autorag-0.2.13}/AutoRAG.egg-info/SOURCES.txt +2 -0
- {autorag-0.2.12 → autorag-0.2.13}/AutoRAG.egg-info/requires.txt +7 -1
- {autorag-0.2.12 → autorag-0.2.13}/PKG-INFO +7 -4
- {autorag-0.2.12 → autorag-0.2.13}/README.md +1 -2
- autorag-0.2.13/autorag/VERSION +1 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/cli.py +15 -2
- {autorag-0.2.12 → autorag-0.2.13}/autorag/evaluator.py +4 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/retrieval/bm25.py +32 -1
- autorag-0.2.13/autorag/validator.py +66 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/api_spec/autorag.rst +8 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/data_creation/data_format.md +10 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/install.md +14 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/local_model.md +1 -1
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/nodes/generator/llama_index_llm.md +3 -1
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/nodes/prompt_maker/prompt_maker.md +18 -6
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/nodes/retrieval/bm25.md +24 -8
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/optimization/optimization.md +7 -1
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/tutorial.md +47 -17
- {autorag-0.2.12 → autorag-0.2.13}/pyproject.toml +4 -0
- {autorag-0.2.12 → autorag-0.2.13}/requirements.txt +0 -2
- {autorag-0.2.12 → autorag-0.2.13}/sample_config/config_korean.yaml +1 -1
- {autorag-0.2.12 → autorag-0.2.13}/sample_config/full.yaml +1 -1
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/retrieval/test_bm25.py +24 -8
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/test_cli.py +8 -0
- autorag-0.2.13/tests/autorag/test_validator.py +16 -0
- autorag-0.2.12/autorag/VERSION +0 -1
- {autorag-0.2.12 → autorag-0.2.13}/.github/FUNDING.yml +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/.github/dependabot.yml +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/.github/workflows/publish.yml +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/.github/workflows/sphinx.yml +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/.gitignore +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/AutoRAG.egg-info/dependency_links.txt +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/AutoRAG.egg-info/entry_points.txt +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/AutoRAG.egg-info/top_level.txt +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/CODE_OF_CONDUCT.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/CONTRIBUTING.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/LICENSE +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/__init__.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/dashboard.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/data/__init__.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/data/corpus/__init__.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/data/corpus/langchain.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/data/corpus/llama_index.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/data/qacreation/__init__.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/data/qacreation/base.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/data/qacreation/llama_index.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/data/qacreation/llama_index_default_prompt.txt +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/data/qacreation/ragas.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/data/qacreation/simple.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/data/utils/__init__.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/data/utils/util.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/deploy.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/evaluation/__init__.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/evaluation/generation.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/evaluation/metric/__init__.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/evaluation/metric/g_eval_prompts/coh_detailed.txt +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/evaluation/metric/g_eval_prompts/con_detailed.txt +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/evaluation/metric/g_eval_prompts/flu_detailed.txt +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/evaluation/metric/g_eval_prompts/rel_detailed.txt +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/evaluation/metric/generation.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/evaluation/metric/retrieval.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/evaluation/metric/retrieval_contents.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/evaluation/metric/util.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/evaluation/retrieval.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/evaluation/retrieval_contents.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/evaluation/util.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/node_line.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/__init__.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/generator/__init__.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/generator/base.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/generator/llama_index_llm.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/generator/openai_llm.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/generator/run.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/generator/vllm.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/passageaugmenter/__init__.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/passageaugmenter/base.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/passageaugmenter/pass_passage_augmenter.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/passageaugmenter/prev_next_augmenter.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/passageaugmenter/run.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/passagecompressor/__init__.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/passagecompressor/base.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/passagecompressor/longllmlingua.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/passagecompressor/pass_compressor.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/passagecompressor/refine.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/passagecompressor/run.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/passagecompressor/tree_summarize.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/passagefilter/__init__.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/passagefilter/base.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/passagefilter/pass_passage_filter.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/passagefilter/percentile_cutoff.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/passagefilter/recency.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/passagefilter/run.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/passagefilter/similarity_percentile_cutoff.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/passagefilter/similarity_threshold_cutoff.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/passagefilter/threshold_cutoff.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/passagereranker/__init__.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/passagereranker/base.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/passagereranker/cohere.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/passagereranker/colbert.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/passagereranker/flag_embedding.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/passagereranker/flag_embedding_llm.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/passagereranker/jina.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/passagereranker/koreranker.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/passagereranker/monot5.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/passagereranker/pass_reranker.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/passagereranker/rankgpt.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/passagereranker/run.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/passagereranker/sentence_transformer.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/passagereranker/tart/__init__.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/passagereranker/tart/modeling_enc_t5.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/passagereranker/tart/tart.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/passagereranker/tart/tokenization_enc_t5.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/passagereranker/time_reranker.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/passagereranker/upr.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/promptmaker/__init__.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/promptmaker/base.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/promptmaker/fstring.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/promptmaker/long_context_reorder.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/promptmaker/run.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/promptmaker/window_replacement.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/queryexpansion/__init__.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/queryexpansion/base.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/queryexpansion/hyde.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/queryexpansion/multi_query_expansion.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/queryexpansion/pass_query_expansion.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/queryexpansion/query_decompose.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/queryexpansion/run.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/retrieval/__init__.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/retrieval/base.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/retrieval/hybrid_cc.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/retrieval/hybrid_rrf.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/retrieval/run.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/nodes/retrieval/vectordb.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/schema/__init__.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/schema/module.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/schema/node.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/strategy.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/support.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/utils/__init__.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/utils/preprocess.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/utils/util.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/autorag/web.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/Makefile +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/make.bat +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/requirements.txt +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/CNAME +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/_static/data_creation.png +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/_static/data_folder.png +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/_static/dcg.png +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/_static/f1_score.png +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/_static/map.png +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/_static/mrr.png +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/_static/ndcg.png +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/_static/ndcg_formula.png +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/_static/node_folder.png +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/_static/node_line_folder.png +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/_static/node_line_summary.png +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/_static/node_lines.png +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/_static/node_summary.png +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/_static/normal_distribution.png +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/_static/project_folder_example.png +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/_static/project_folders.png +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/_static/resources_folder.png +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/_static/roadmap/RAG_paradigms.png +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/_static/roadmap/advanced_RAG.png +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/_static/roadmap/cycle.png +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/_static/roadmap/merger.png +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/_static/roadmap/node_line_modular.png +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/_static/roadmap/policy.png +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/_static/samsung_sundae.jpeg +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/_static/score_fusion.png +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/_static/trial_folder.png +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/_static/trial_json.png +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/_static/trial_summary.png +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/_static/web_interface.png +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/api_spec/autorag.data.corpus.rst +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/api_spec/autorag.data.qacreation.rst +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/api_spec/autorag.data.rst +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/api_spec/autorag.data.utils.rst +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/api_spec/autorag.evaluation.metric.rst +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/api_spec/autorag.evaluation.rst +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/api_spec/autorag.nodes.generator.rst +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/api_spec/autorag.nodes.passageaugmenter.rst +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/api_spec/autorag.nodes.passagecompressor.rst +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/api_spec/autorag.nodes.passagefilter.rst +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/api_spec/autorag.nodes.passagereranker.rst +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/api_spec/autorag.nodes.passagereranker.tart.rst +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/api_spec/autorag.nodes.promptmaker.rst +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/api_spec/autorag.nodes.queryexpansion.rst +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/api_spec/autorag.nodes.retrieval.rst +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/api_spec/autorag.nodes.rst +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/api_spec/autorag.schema.rst +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/api_spec/autorag.utils.rst +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/api_spec/modules.rst +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/conf.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/data_creation/ragas.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/data_creation/tutorial.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/deploy/api_endpoint.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/deploy/web.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/evaluate_metrics/generation.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/evaluate_metrics/retrieval.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/evaluate_metrics/retrieval_contents.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/index.rst +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/nodes/generator/generator.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/nodes/generator/openai_llm.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/nodes/generator/vllm.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/nodes/index.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/nodes/passage_augmenter/passage_augmenter.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/nodes/passage_augmenter/prev_next_augmenter.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/nodes/passage_compressor/longllmlingua.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/nodes/passage_compressor/passage_compressor.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/nodes/passage_compressor/refine.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/nodes/passage_compressor/tree_summarize.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/nodes/passage_filter/passage_filter.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/nodes/passage_filter/percentile_cutoff.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/nodes/passage_filter/recency_filter.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/nodes/passage_filter/similarity_percentile_cutoff.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/nodes/passage_filter/similarity_threshold_cutoff.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/nodes/passage_filter/threshold_cutoff.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/nodes/passage_reranker/cohere.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/nodes/passage_reranker/colbert.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/nodes/passage_reranker/flag_embedding_llm_reranker.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/nodes/passage_reranker/flag_embedding_reranker.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/nodes/passage_reranker/jina_reranker.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/nodes/passage_reranker/koreranker.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/nodes/passage_reranker/monot5.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/nodes/passage_reranker/passage_reranker.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/nodes/passage_reranker/rankgpt.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/nodes/passage_reranker/sentence_transformer_reranker.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/nodes/passage_reranker/tart.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/nodes/passage_reranker/time_reranker.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/nodes/passage_reranker/upr.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/nodes/prompt_maker/fstring.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/nodes/prompt_maker/long_context_reorder.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/nodes/prompt_maker/window_replacement.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/nodes/query_expansion/hyde.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/nodes/query_expansion/multi_query_expansion.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/nodes/query_expansion/query_decompose.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/nodes/query_expansion/query_expansion.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/nodes/retrieval/hybrid_cc.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/nodes/retrieval/hybrid_rrf.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/nodes/retrieval/retrieval.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/nodes/retrieval/vectordb.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/optimization/custom_config.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/optimization/folder_structure.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/optimization/strategies.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/roadmap/modular_rag.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/structure.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/docs/source/troubleshooting.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/sample_config/compact_local.yaml +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/sample_config/compact_openai.yaml +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/sample_config/extracted_sample.yaml +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/sample_config/simple_local.yaml +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/sample_config/simple_ollama.yaml +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/sample_config/simple_openai.yaml +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/sample_dataset/README.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/sample_dataset/eli5/load_eli5_dataset.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/sample_dataset/hotpotqa/load_hotpotqa_dataset.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/sample_dataset/msmarco/load_msmarco_dataset.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/sample_dataset/triviaqa/load_triviaqa_dataset.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/setup.cfg +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/data/corpus/test_base.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/data/corpus/test_langchain.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/data/corpus/test_llama_index_corpus.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/data/qacreation/test_base_qacreation.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/data/qacreation/test_llama_index_qacreation.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/data/qacreation/test_ragas_qa_creation.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/data/qacreation/test_simple.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/evaluate/metric/test_generation_metric.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/evaluate/metric/test_retrieval_contents_metric.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/evaluate/metric/test_retrieval_metric.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/evaluate/test_evaluate_util.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/evaluate/test_generation_evaluate.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/evaluate/test_retrieval_contents_evaluate.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/evaluate/test_retrieval_evaluate.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/generator/test_generator_base.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/generator/test_llama_index_llm.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/generator/test_openai.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/generator/test_run_generator_node.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/generator/test_vllm.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/passageaugmenter/test_base_passage_augmenter.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/passageaugmenter/test_pass_passage_augmenter.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/passageaugmenter/test_prev_next_augmenter.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/passageaugmenter/test_run_passage_augmenter.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/passagecompressor/test_base_passage_compressor.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/passagecompressor/test_longllmlingua.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/passagecompressor/test_pass_compressor.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/passagecompressor/test_refine.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/passagecompressor/test_run_passage_compressor_node.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/passagecompressor/test_tree_summarize.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/passagefilter/test_pass_passage_filter.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/passagefilter/test_passage_filter_base.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/passagefilter/test_passage_filter_run.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/passagefilter/test_percentile_cutoff.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/passagefilter/test_recency_filter.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/passagefilter/test_similarity_percentile_cutoff.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/passagefilter/test_similarity_threshold_cutoff.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/passagefilter/test_threshold_cutoff.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/passagereranker/test_cohere_reranker.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/passagereranker/test_colbert_reranker.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/passagereranker/test_flag_embedding.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/passagereranker/test_flag_embedding_llm.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/passagereranker/test_jina_reranker.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/passagereranker/test_koreranker.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/passagereranker/test_monot5.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/passagereranker/test_pass_reranker.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/passagereranker/test_passage_reranker_base.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/passagereranker/test_passage_reranker_run.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/passagereranker/test_rankgpt.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/passagereranker/test_sentence_transformer.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/passagereranker/test_tart.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/passagereranker/test_time_reranker.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/passagereranker/test_upr.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/promptmaker/test_fstring.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/promptmaker/test_long_context_reorder.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/promptmaker/test_prompt_maker_base.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/promptmaker/test_prompt_maker_run.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/promptmaker/test_window_replacement.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/queryexpansion/test_hyde.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/queryexpansion/test_multi_query_expansion.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/queryexpansion/test_pass_query_expansion.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/queryexpansion/test_query_decompose.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/queryexpansion/test_query_expansion_base.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/queryexpansion/test_query_expansion_run.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/retrieval/test_hybrid_base.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/retrieval/test_hybrid_cc.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/retrieval/test_hybrid_rrf.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/retrieval/test_retrieval_base.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/retrieval/test_run_retrieval_node.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/nodes/retrieval/test_vectordb.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/schema/test_module_schema.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/schema/test_node_schema.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/test_dashboard.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/test_deploy.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/test_evaluator.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/test_strategy.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/test_support.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/test_web.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/utils/test_preprocess.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/autorag/utils/test_util.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/conftest.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/delete_tests.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/mock.py +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/requirements.txt +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/README.md +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/corpus_data_sample.parquet +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/data_creation/raw_dir/sample1.txt +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/data_creation/raw_dir/sample2.txt +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/data_creation/raw_dir/sample3.txt +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/full.yaml +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/qa_data_sample.parquet +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/qa_gen_prompts/prompt1.txt +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/qa_gen_prompts/prompt2.txt +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/qa_gen_prompts/prompt3.txt +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/qa_test_data_sample.parquet +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/0/config.yaml +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/0/post_retrieve_node_line/generator/0.parquet +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/0/post_retrieve_node_line/generator/1.parquet +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/0/post_retrieve_node_line/generator/2.parquet +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/0/post_retrieve_node_line/generator/3.parquet +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/0/post_retrieve_node_line/generator/4.parquet +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/0/post_retrieve_node_line/generator/5.parquet +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/0/post_retrieve_node_line/generator/best_5.parquet +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/0/post_retrieve_node_line/generator/summary.csv +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/0/post_retrieve_node_line/prompt_maker/0.parquet +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/0/post_retrieve_node_line/prompt_maker/1.parquet +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/0/post_retrieve_node_line/prompt_maker/best_0.parquet +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/0/post_retrieve_node_line/prompt_maker/summary.csv +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/0/post_retrieve_node_line/summary.csv +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/0/pre_retrieve_node_line/query_expansion/0.parquet +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/0/pre_retrieve_node_line/query_expansion/1.parquet +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/0/pre_retrieve_node_line/query_expansion/2.parquet +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/0/pre_retrieve_node_line/query_expansion/best_0.parquet +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/0/pre_retrieve_node_line/query_expansion/summary.csv +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/0/pre_retrieve_node_line/summary.csv +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/0/retrieve_node_line/passage_compressor/0.parquet +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/0/retrieve_node_line/passage_compressor/best_0.parquet +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/0/retrieve_node_line/passage_compressor/summary.csv +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/0/retrieve_node_line/passage_reranker/0.parquet +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/0/retrieve_node_line/passage_reranker/1.parquet +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/0/retrieve_node_line/passage_reranker/best_0.parquet +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/0/retrieve_node_line/passage_reranker/summary.csv +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/0/retrieve_node_line/retrieval/0.parquet +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/0/retrieve_node_line/retrieval/1.parquet +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/0/retrieve_node_line/retrieval/best_0.parquet +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/0/retrieve_node_line/retrieval/summary.csv +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/0/retrieve_node_line/summary.csv +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/0/summary.csv +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/1/config.yaml +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/1/pre_retrieve_node_line/query_expansion/0.parquet +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/1/pre_retrieve_node_line/query_expansion/1.parquet +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/1/pre_retrieve_node_line/query_expansion/2.parquet +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/1/pre_retrieve_node_line/query_expansion/best_0.parquet +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/1/pre_retrieve_node_line/query_expansion/summary.csv +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/1/pre_retrieve_node_line/summary.csv +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/1/retrieve_node_line/passage_filter/0.parquet +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/1/retrieve_node_line/passage_reranker/0.parquet +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/1/retrieve_node_line/passage_reranker/1.parquet +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/1/retrieve_node_line/passage_reranker/best_0.parquet +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/1/retrieve_node_line/passage_reranker/summary.csv +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/1/retrieve_node_line/retrieval/0.parquet +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/1/retrieve_node_line/retrieval/1.parquet +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/1/retrieve_node_line/retrieval/best_0.parquet +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/1/retrieve_node_line/retrieval/summary.csv +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/2/config.yaml +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/2/post_retrieve_node_line/prompt_maker/0.parquet +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/2/pre_retrieve_node_line/query_expansion/0.parquet +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/2/pre_retrieve_node_line/query_expansion/1.parquet +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/2/pre_retrieve_node_line/query_expansion/2.parquet +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/2/pre_retrieve_node_line/query_expansion/best_0.parquet +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/2/pre_retrieve_node_line/query_expansion/summary.csv +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/2/pre_retrieve_node_line/summary.csv +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/2/retrieve_node_line/passage_compressor/0.parquet +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/2/retrieve_node_line/passage_compressor/best_0.parquet +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/2/retrieve_node_line/passage_compressor/summary.csv +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/2/retrieve_node_line/passage_reranker/0.parquet +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/2/retrieve_node_line/passage_reranker/1.parquet +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/2/retrieve_node_line/passage_reranker/best_0.parquet +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/2/retrieve_node_line/passage_reranker/summary.csv +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/2/retrieve_node_line/retrieval/0.parquet +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/2/retrieve_node_line/retrieval/1.parquet +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/2/retrieve_node_line/retrieval/best_0.parquet +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/2/retrieve_node_line/retrieval/summary.csv +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/2/retrieve_node_line/summary.csv +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/3/config.yaml +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/best.yaml +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/data/corpus.parquet +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/data/qa.parquet +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/resources/bm25_porter_stemmer.pkl +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/resources/chroma/6595d304-5270-4d9f-a267-c191366804ce/data_level0.bin +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/resources/chroma/6595d304-5270-4d9f-a267-c191366804ce/header.bin +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/resources/chroma/6595d304-5270-4d9f-a267-c191366804ce/length.bin +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/resources/chroma/6595d304-5270-4d9f-a267-c191366804ce/link_lists.bin +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/resources/chroma/chroma.sqlite3 +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/result_project/trial.json +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/sample_contents_nqa.csv +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/sample_project/data/corpus.parquet +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/sample_project/data/qa.parquet +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/sample_project/resources/bm25_gpt2.pkl +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/sample_project/resources/bm25_porter_stemmer.pkl +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/simple.yaml +0 -0
- {autorag-0.2.12 → autorag-0.2.13}/tests/resources/test_bm25_retrieval.pkl +0 -0
|
@@ -16,6 +16,10 @@ jobs:
|
|
|
16
16
|
runs-on: ubuntu-latest
|
|
17
17
|
steps:
|
|
18
18
|
- uses: actions/checkout@v4
|
|
19
|
+
- uses: actions/setup-java@v4
|
|
20
|
+
with:
|
|
21
|
+
distribution: 'zulu'
|
|
22
|
+
java-version: '17'
|
|
19
23
|
- name: Upgrade pip
|
|
20
24
|
run: |
|
|
21
25
|
python3 -m pip install --upgrade pip
|
|
@@ -24,7 +28,7 @@ jobs:
|
|
|
24
28
|
sudo apt-get install gcc
|
|
25
29
|
- name: Install AutoRAG
|
|
26
30
|
run: |
|
|
27
|
-
pip install -e .
|
|
31
|
+
pip install -e '.[all]'
|
|
28
32
|
- name: Install dependencies
|
|
29
33
|
run: |
|
|
30
34
|
pip install -r tests/requirements.txt
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: AutoRAG
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.13
|
|
4
4
|
Summary: Automatically Evaluate RAG pipelines with your own data. Find optimal structure for new RAG product.
|
|
5
5
|
Author-email: Marker-Inc <vkehfdl1@gmail.com>
|
|
6
6
|
License: Apache License
|
|
@@ -249,7 +249,6 @@ Requires-Dist: bert_score
|
|
|
249
249
|
Requires-Dist: sentence-transformers
|
|
250
250
|
Requires-Dist: FlagEmbedding
|
|
251
251
|
Requires-Dist: ragas
|
|
252
|
-
Requires-Dist: kiwipiepy
|
|
253
252
|
Requires-Dist: llmlingua
|
|
254
253
|
Requires-Dist: llama-index>=0.10.1
|
|
255
254
|
Requires-Dist: llama-index-core>=0.10.1
|
|
@@ -268,6 +267,11 @@ Requires-Dist: seaborn
|
|
|
268
267
|
Requires-Dist: ipykernel
|
|
269
268
|
Requires-Dist: ipywidgets
|
|
270
269
|
Requires-Dist: ipywidgets_bokeh
|
|
270
|
+
Provides-Extra: ko
|
|
271
|
+
Requires-Dist: kiwipiepy; extra == "ko"
|
|
272
|
+
Requires-Dist: konlpy; extra == "ko"
|
|
273
|
+
Provides-Extra: all
|
|
274
|
+
Requires-Dist: AutoRAG[ko]; extra == "all"
|
|
271
275
|
|
|
272
276
|
# AutoRAG
|
|
273
277
|
|
|
@@ -282,7 +286,6 @@ Plus, join our 📞 [Discord](https://discord.gg/P4DYXfmSAs) Community.
|
|
|
282
286
|
### 💪 Colab Tutorial
|
|
283
287
|
|
|
284
288
|
- [Step 1: Basic of AutoRAG | Optimizing your RAG pipeline](https://colab.research.google.com/drive/19OEQXO_pHN6gnn2WdfPd4hjnS-4GurVd?usp=sharing)
|
|
285
|
-
- [Step 2: Create evaluation dataset](https://colab.research.google.com/drive/1HXjVHCLTaX7mkmZp3IKlEPt0B3jVeHvP#scrollTo=cgFUCuaUZvTr)
|
|
286
289
|
|
|
287
290
|
---
|
|
288
291
|
|
|
@@ -330,7 +333,7 @@ Try now and find the best RAG pipeline for your own use-case.
|
|
|
330
333
|
|
|
331
334
|
## ❗Supporting Nodes & modules
|
|
332
335
|
|
|
333
|
-

|
|
334
337
|

|
|
335
338
|
You can check our all supporting Nodes & modules
|
|
336
339
|
at [here](https://edai.notion.site/Supporting-Nodes-modules-0ebc7810649f4e41aead472a92976be4?pvs=4)
|
|
@@ -25,6 +25,7 @@ autorag/evaluator.py
|
|
|
25
25
|
autorag/node_line.py
|
|
26
26
|
autorag/strategy.py
|
|
27
27
|
autorag/support.py
|
|
28
|
+
autorag/validator.py
|
|
28
29
|
autorag/web.py
|
|
29
30
|
autorag/data/__init__.py
|
|
30
31
|
autorag/data/corpus/__init__.py
|
|
@@ -264,6 +265,7 @@ tests/autorag/test_deploy.py
|
|
|
264
265
|
tests/autorag/test_evaluator.py
|
|
265
266
|
tests/autorag/test_strategy.py
|
|
266
267
|
tests/autorag/test_support.py
|
|
268
|
+
tests/autorag/test_validator.py
|
|
267
269
|
tests/autorag/test_web.py
|
|
268
270
|
tests/autorag/data/corpus/test_base.py
|
|
269
271
|
tests/autorag/data/corpus/test_langchain.py
|
|
@@ -27,7 +27,6 @@ bert_score
|
|
|
27
27
|
sentence-transformers
|
|
28
28
|
FlagEmbedding
|
|
29
29
|
ragas
|
|
30
|
-
kiwipiepy
|
|
31
30
|
llmlingua
|
|
32
31
|
llama-index>=0.10.1
|
|
33
32
|
llama-index-core>=0.10.1
|
|
@@ -46,3 +45,10 @@ seaborn
|
|
|
46
45
|
ipykernel
|
|
47
46
|
ipywidgets
|
|
48
47
|
ipywidgets_bokeh
|
|
48
|
+
|
|
49
|
+
[all]
|
|
50
|
+
AutoRAG[ko]
|
|
51
|
+
|
|
52
|
+
[ko]
|
|
53
|
+
kiwipiepy
|
|
54
|
+
konlpy
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: AutoRAG
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.13
|
|
4
4
|
Summary: Automatically Evaluate RAG pipelines with your own data. Find optimal structure for new RAG product.
|
|
5
5
|
Author-email: Marker-Inc <vkehfdl1@gmail.com>
|
|
6
6
|
License: Apache License
|
|
@@ -249,7 +249,6 @@ Requires-Dist: bert_score
|
|
|
249
249
|
Requires-Dist: sentence-transformers
|
|
250
250
|
Requires-Dist: FlagEmbedding
|
|
251
251
|
Requires-Dist: ragas
|
|
252
|
-
Requires-Dist: kiwipiepy
|
|
253
252
|
Requires-Dist: llmlingua
|
|
254
253
|
Requires-Dist: llama-index>=0.10.1
|
|
255
254
|
Requires-Dist: llama-index-core>=0.10.1
|
|
@@ -268,6 +267,11 @@ Requires-Dist: seaborn
|
|
|
268
267
|
Requires-Dist: ipykernel
|
|
269
268
|
Requires-Dist: ipywidgets
|
|
270
269
|
Requires-Dist: ipywidgets_bokeh
|
|
270
|
+
Provides-Extra: ko
|
|
271
|
+
Requires-Dist: kiwipiepy; extra == "ko"
|
|
272
|
+
Requires-Dist: konlpy; extra == "ko"
|
|
273
|
+
Provides-Extra: all
|
|
274
|
+
Requires-Dist: AutoRAG[ko]; extra == "all"
|
|
271
275
|
|
|
272
276
|
# AutoRAG
|
|
273
277
|
|
|
@@ -282,7 +286,6 @@ Plus, join our 📞 [Discord](https://discord.gg/P4DYXfmSAs) Community.
|
|
|
282
286
|
### 💪 Colab Tutorial
|
|
283
287
|
|
|
284
288
|
- [Step 1: Basic of AutoRAG | Optimizing your RAG pipeline](https://colab.research.google.com/drive/19OEQXO_pHN6gnn2WdfPd4hjnS-4GurVd?usp=sharing)
|
|
285
|
-
- [Step 2: Create evaluation dataset](https://colab.research.google.com/drive/1HXjVHCLTaX7mkmZp3IKlEPt0B3jVeHvP#scrollTo=cgFUCuaUZvTr)
|
|
286
289
|
|
|
287
290
|
---
|
|
288
291
|
|
|
@@ -330,7 +333,7 @@ Try now and find the best RAG pipeline for your own use-case.
|
|
|
330
333
|
|
|
331
334
|
## ❗Supporting Nodes & modules
|
|
332
335
|
|
|
333
|
-

|
|
334
337
|

|
|
335
338
|
You can check our all supporting Nodes & modules
|
|
336
339
|
at [here](https://edai.notion.site/Supporting-Nodes-modules-0ebc7810649f4e41aead472a92976be4?pvs=4)
|
|
@@ -11,7 +11,6 @@ Plus, join our 📞 [Discord](https://discord.gg/P4DYXfmSAs) Community.
|
|
|
11
11
|
### 💪 Colab Tutorial
|
|
12
12
|
|
|
13
13
|
- [Step 1: Basic of AutoRAG | Optimizing your RAG pipeline](https://colab.research.google.com/drive/19OEQXO_pHN6gnn2WdfPd4hjnS-4GurVd?usp=sharing)
|
|
14
|
-
- [Step 2: Create evaluation dataset](https://colab.research.google.com/drive/1HXjVHCLTaX7mkmZp3IKlEPt0B3jVeHvP#scrollTo=cgFUCuaUZvTr)
|
|
15
14
|
|
|
16
15
|
---
|
|
17
16
|
|
|
@@ -59,7 +58,7 @@ Try now and find the best RAG pipeline for your own use-case.
|
|
|
59
58
|
|
|
60
59
|
## ❗Supporting Nodes & modules
|
|
61
60
|
|
|
62
|
-

|
|
63
62
|

|
|
64
63
|
You can check our all supporting Nodes & modules
|
|
65
64
|
at [here](https://edai.notion.site/Supporting-Nodes-modules-0ebc7810649f4e41aead472a92976be4?pvs=4)
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
0.2.13
|
|
@@ -12,6 +12,7 @@ from autorag import dashboard
|
|
|
12
12
|
from autorag.deploy import Runner
|
|
13
13
|
from autorag.deploy import extract_best_config as original_extract_best_config
|
|
14
14
|
from autorag.evaluator import Evaluator
|
|
15
|
+
from autorag.validator import Validator
|
|
15
16
|
|
|
16
17
|
logger = logging.getLogger("AutoRAG")
|
|
17
18
|
|
|
@@ -33,7 +34,6 @@ def evaluate(config, qa_data_path, corpus_data_path, project_dir):
|
|
|
33
34
|
raise ValueError(f"Config file {config} does not exist.")
|
|
34
35
|
evaluator = Evaluator(qa_data_path, corpus_data_path, project_dir=project_dir)
|
|
35
36
|
evaluator.start_trial(config)
|
|
36
|
-
logger.info('Evaluation complete.')
|
|
37
37
|
|
|
38
38
|
|
|
39
39
|
@click.command()
|
|
@@ -95,7 +95,19 @@ def restart_evaluate(trial_path):
|
|
|
95
95
|
corpus_data_path = os.path.join(project_dir, 'data', 'corpus.parquet')
|
|
96
96
|
evaluator = Evaluator(qa_data_path, corpus_data_path, project_dir)
|
|
97
97
|
evaluator.restart_trial(trial_path)
|
|
98
|
-
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
@click.command()
|
|
101
|
+
@click.option('--config', '-c', help='Path to config yaml file. Must be yaml or yml file.', type=str)
|
|
102
|
+
@click.option('--qa_data_path', help='Path to QA dataset. Must be parquet file.', type=str)
|
|
103
|
+
@click.option('--corpus_data_path', help='Path to corpus dataset. Must be parquet file.', type=str)
|
|
104
|
+
def validate(config, qa_data_path, corpus_data_path):
|
|
105
|
+
if not config.endswith('.yaml') and not config.endswith('.yml'):
|
|
106
|
+
raise ValueError(f"Config file {config} is not a parquet file.")
|
|
107
|
+
if not os.path.exists(config):
|
|
108
|
+
raise ValueError(f"Config file {config} does not exist.")
|
|
109
|
+
validator = Validator(qa_data_path=qa_data_path, corpus_data_path=corpus_data_path)
|
|
110
|
+
validator.validate(config)
|
|
99
111
|
|
|
100
112
|
|
|
101
113
|
cli.add_command(evaluate, 'evaluate')
|
|
@@ -104,6 +116,7 @@ cli.add_command(run_web, 'run_web')
|
|
|
104
116
|
cli.add_command(run_dashboard, 'dashboard')
|
|
105
117
|
cli.add_command(extract_best_config, 'extract_best_config')
|
|
106
118
|
cli.add_command(restart_evaluate, 'restart_evaluate')
|
|
119
|
+
cli.add_command(validate, 'validate')
|
|
107
120
|
|
|
108
121
|
if __name__ == '__main__':
|
|
109
122
|
cli()
|
|
@@ -101,6 +101,8 @@ class Evaluator:
|
|
|
101
101
|
|
|
102
102
|
trial_summary_df.to_csv(os.path.join(self.project_dir, trial_name, 'summary.csv'), index=False)
|
|
103
103
|
|
|
104
|
+
logger.info('Evaluation complete.')
|
|
105
|
+
|
|
104
106
|
def __embed(self, node_lines: Dict[str, List[Node]]):
|
|
105
107
|
if any(list(map(lambda nodes: module_type_exists(nodes, 'bm25'), node_lines.values()))):
|
|
106
108
|
# ingest BM25 corpus
|
|
@@ -265,6 +267,8 @@ class Evaluator:
|
|
|
265
267
|
trial_summary_df = self._append_node_line_summary(node_line_name, node_line_dir, trial_summary_df)
|
|
266
268
|
trial_summary_df.to_csv(os.path.join(trial_path, 'summary.csv'), index=False)
|
|
267
269
|
|
|
270
|
+
logger.info('Evaluation complete.')
|
|
271
|
+
|
|
268
272
|
def __find_conflict_point(self, trial_path: str, node_line_names: List[str],
|
|
269
273
|
node_lines: Dict[str, List[Node]]) -> tuple[str, str]:
|
|
270
274
|
for node_line_name in node_line_names:
|
|
@@ -6,7 +6,6 @@ from typing import List, Dict, Tuple, Callable, Union, Iterable, Optional
|
|
|
6
6
|
|
|
7
7
|
import numpy as np
|
|
8
8
|
import pandas as pd
|
|
9
|
-
from kiwipiepy import Kiwi, Token
|
|
10
9
|
from llama_index.core.indices.keyword_table.utils import simple_extract_keywords
|
|
11
10
|
from nltk import PorterStemmer
|
|
12
11
|
from rank_bm25 import BM25Okapi
|
|
@@ -18,12 +17,42 @@ from autorag.utils.util import normalize_string
|
|
|
18
17
|
|
|
19
18
|
|
|
20
19
|
def tokenize_ko_kiwi(texts: List[str]) -> List[List[str]]:
|
|
20
|
+
try:
|
|
21
|
+
from kiwipiepy import Kiwi, Token
|
|
22
|
+
except ImportError:
|
|
23
|
+
raise ImportError("You need to install kiwipiepy to use 'ko_kiwi' tokenizer. "
|
|
24
|
+
"Please install kiwipiepy by running 'pip install kiwipiepy'. "
|
|
25
|
+
"Or install Korean version of AutoRAG by running 'pip install AutoRAG[ko]'.")
|
|
21
26
|
texts = list(map(lambda x: x.strip().lower(), texts))
|
|
22
27
|
kiwi = Kiwi()
|
|
23
28
|
tokenized_list: Iterable[List[Token]] = kiwi.tokenize(texts)
|
|
24
29
|
return [list(map(lambda x: x.form, token_list)) for token_list in tokenized_list]
|
|
25
30
|
|
|
26
31
|
|
|
32
|
+
def tokenize_ko_kkma(texts: List[str]) -> List[List[str]]:
|
|
33
|
+
try:
|
|
34
|
+
from konlpy.tag import Kkma
|
|
35
|
+
except ImportError:
|
|
36
|
+
raise ImportError("You need to install konlpy to use 'ko_kkma' tokenizer. "
|
|
37
|
+
"Please install konlpy by running 'pip install konlpy'. "
|
|
38
|
+
"Or install Korean version of AutoRAG by running 'pip install AutoRAG[ko]'.")
|
|
39
|
+
tokenizer = Kkma()
|
|
40
|
+
tokenized_list: List[List[str]] = list(map(lambda x: tokenizer.morphs(x), texts))
|
|
41
|
+
return tokenized_list
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def tokenize_ko_okt(texts: List[str]) -> List[List[str]]:
|
|
45
|
+
try:
|
|
46
|
+
from konlpy.tag import Okt
|
|
47
|
+
except ImportError:
|
|
48
|
+
raise ImportError("You need to install konlpy to use 'ko_kkma' tokenizer. "
|
|
49
|
+
"Please install konlpy by running 'pip install konlpy'. "
|
|
50
|
+
"Or install Korean version of AutoRAG by running 'pip install AutoRAG[ko]'.")
|
|
51
|
+
tokenizer = Okt()
|
|
52
|
+
tokenized_list: List[List[str]] = list(map(lambda x: tokenizer.morphs(x), texts))
|
|
53
|
+
return tokenized_list
|
|
54
|
+
|
|
55
|
+
|
|
27
56
|
def tokenize_porter_stemmer(texts: List[str]) -> List[List[str]]:
|
|
28
57
|
def tokenize_remove_stopword(text: str, stemmer) -> List[str]:
|
|
29
58
|
text = text.lower()
|
|
@@ -47,6 +76,8 @@ BM25_TOKENIZER = {
|
|
|
47
76
|
'porter_stemmer': tokenize_porter_stemmer,
|
|
48
77
|
'ko_kiwi': tokenize_ko_kiwi,
|
|
49
78
|
'space': tokenize_space,
|
|
79
|
+
'ko_kkma': tokenize_ko_kkma,
|
|
80
|
+
'ko_okt': tokenize_ko_okt,
|
|
50
81
|
}
|
|
51
82
|
|
|
52
83
|
|
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
import itertools
|
|
2
|
+
import logging
|
|
3
|
+
import os
|
|
4
|
+
import tempfile
|
|
5
|
+
|
|
6
|
+
import pandas as pd
|
|
7
|
+
|
|
8
|
+
from autorag.evaluator import Evaluator
|
|
9
|
+
from autorag.utils import cast_qa_dataset, cast_corpus_dataset, validate_qa_from_corpus_dataset
|
|
10
|
+
|
|
11
|
+
logger = logging.getLogger("AutoRAG")
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class Validator:
|
|
15
|
+
def __init__(self, qa_data_path: str, corpus_data_path: str):
|
|
16
|
+
"""
|
|
17
|
+
Initialize a Validator object.
|
|
18
|
+
|
|
19
|
+
:param qa_data_path: The path to the QA dataset.
|
|
20
|
+
Must be parquet file.
|
|
21
|
+
:param corpus_data_path: The path to the corpus dataset.
|
|
22
|
+
Must be parquet file.
|
|
23
|
+
"""
|
|
24
|
+
# validate data paths
|
|
25
|
+
if not os.path.exists(qa_data_path):
|
|
26
|
+
raise ValueError(f"QA data path {qa_data_path} does not exist.")
|
|
27
|
+
if not os.path.exists(corpus_data_path):
|
|
28
|
+
raise ValueError(f"Corpus data path {corpus_data_path} does not exist.")
|
|
29
|
+
if not qa_data_path.endswith('.parquet'):
|
|
30
|
+
raise ValueError(f"QA data path {qa_data_path} is not a parquet file.")
|
|
31
|
+
if not corpus_data_path.endswith('.parquet'):
|
|
32
|
+
raise ValueError(f"Corpus data path {corpus_data_path} is not a parquet file.")
|
|
33
|
+
self.qa_data = pd.read_parquet(qa_data_path, engine='pyarrow')
|
|
34
|
+
self.corpus_data = pd.read_parquet(corpus_data_path, engine='pyarrow')
|
|
35
|
+
self.qa_data = cast_qa_dataset(self.qa_data)
|
|
36
|
+
self.corpus_data = cast_corpus_dataset(self.corpus_data)
|
|
37
|
+
|
|
38
|
+
def validate(self, yaml_path: str, qa_cnt: int = 5, random_state: int = 42):
|
|
39
|
+
# sample QA data
|
|
40
|
+
sample_qa_df = self.qa_data.sample(qa_cnt, random_state=random_state)
|
|
41
|
+
sample_qa_df.reset_index(drop=True, inplace=True)
|
|
42
|
+
|
|
43
|
+
# get doc_id
|
|
44
|
+
temp_qa_df = sample_qa_df.copy(deep=True)
|
|
45
|
+
flatten_retrieval_gts = temp_qa_df['retrieval_gt'].apply(
|
|
46
|
+
lambda x: list(itertools.chain.from_iterable(x))).tolist()
|
|
47
|
+
target_doc_ids = list(itertools.chain.from_iterable(flatten_retrieval_gts))
|
|
48
|
+
|
|
49
|
+
# make sample corpus data
|
|
50
|
+
sample_corpus_df = self.corpus_data.loc[self.corpus_data['doc_id'].isin(target_doc_ids)]
|
|
51
|
+
sample_corpus_df.reset_index(drop=True, inplace=True)
|
|
52
|
+
|
|
53
|
+
validate_qa_from_corpus_dataset(sample_qa_df, sample_corpus_df)
|
|
54
|
+
|
|
55
|
+
# start Evaluate at temp project directory
|
|
56
|
+
with tempfile.NamedTemporaryFile(suffix='.parquet') as qa_path, \
|
|
57
|
+
tempfile.NamedTemporaryFile(suffix='.parquet') as corpus_path, \
|
|
58
|
+
tempfile.TemporaryDirectory() as temp_project_dir:
|
|
59
|
+
sample_qa_df.to_parquet(qa_path.name, index=False)
|
|
60
|
+
sample_corpus_df.to_parquet(corpus_path.name, index=False)
|
|
61
|
+
|
|
62
|
+
evaluator = Evaluator(qa_data_path=qa_path.name, corpus_data_path=corpus_path.name,
|
|
63
|
+
project_dir=temp_project_dir)
|
|
64
|
+
evaluator.start_trial(yaml_path)
|
|
65
|
+
|
|
66
|
+
logger.info("Validation complete.")
|
|
@@ -72,6 +72,14 @@ autorag.support module
|
|
|
72
72
|
:undoc-members:
|
|
73
73
|
:show-inheritance:
|
|
74
74
|
|
|
75
|
+
autorag.validator module
|
|
76
|
+
------------------------
|
|
77
|
+
|
|
78
|
+
.. automodule:: autorag.validator
|
|
79
|
+
:members:
|
|
80
|
+
:undoc-members:
|
|
81
|
+
:show-inheritance:
|
|
82
|
+
|
|
75
83
|
autorag.web module
|
|
76
84
|
------------------
|
|
77
85
|
|
|
@@ -76,6 +76,14 @@ it is okay to save it as 1-d list or just string.
|
|
|
76
76
|
If you save it as 1-d list, it treats as 'and' operation.
|
|
77
77
|
```
|
|
78
78
|
|
|
79
|
+
```{attention}
|
|
80
|
+
The ids of retrieval_gt must be included in the corpus dataset as `doc_id`.
|
|
81
|
+
|
|
82
|
+
When AutoRAG starts to evaluate, it checks the existence of the retrieval ids in the corpus dataset.
|
|
83
|
+
|
|
84
|
+
You MUST match the retrieval_gt ids with the corpus dataset.
|
|
85
|
+
```
|
|
86
|
+
|
|
79
87
|
This column is crucial because AutoRAG evaluate retrieval performance with this column.
|
|
80
88
|
It can affect hugely to optimization performance or nodes like retrieval, query expansion or passage reranker.
|
|
81
89
|
|
|
@@ -111,6 +119,8 @@ A unique identifier for each passage. Its type is `string`.
|
|
|
111
119
|
|
|
112
120
|
```{warning}
|
|
113
121
|
Do not make a duplicate doc id. It can occur unexpected behavior.
|
|
122
|
+
|
|
123
|
+
Plus, we suggest you to double-check that retrieval_gt ids in the QA dataset are included in the corpus dataset as `doc_id`.
|
|
114
124
|
```
|
|
115
125
|
|
|
116
126
|
### contents
|
|
@@ -18,6 +18,20 @@ Do you have any trouble with installation?
|
|
|
18
18
|
First, you can check out the [troubleshooting](troubleshooting.md) page.
|
|
19
19
|
```
|
|
20
20
|
|
|
21
|
+
### Installation for Korean 🇰🇷
|
|
22
|
+
|
|
23
|
+
You can install optional dependencies for Korean language.
|
|
24
|
+
|
|
25
|
+
```bash
|
|
26
|
+
pip install AutoRAG[ko]
|
|
27
|
+
```
|
|
28
|
+
|
|
29
|
+
And after that, you have to install **jdk 17** for using `konlpy`.
|
|
30
|
+
Plus, don't forget to set environment PATH for jdk. (JAVA_HOME and PATH)
|
|
31
|
+
|
|
32
|
+
The instruction for mac users
|
|
33
|
+
is [here](https://velog.io/@yoonsy/M1%EC%B9%A9-Mac%EC%97%90-konlpy-%EC%84%A4%EC%B9%98%ED%95%98%EA%B8%B0).
|
|
34
|
+
|
|
21
35
|
## Setup OPENAI API KEY
|
|
22
36
|
To use LLM and embedding models, it is common to use OpenAI models.
|
|
23
37
|
If you want to use other models, check out [here](local_model.md)
|
|
@@ -66,7 +66,7 @@ This is the parameter for the LLM model.
|
|
|
66
66
|
You can set the model parameter for LlamaIndex LLM initialization.
|
|
67
67
|
The most frequently used parameters are `model`, `max_token`, and `temperature`.
|
|
68
68
|
Please check what you can set for the model parameter
|
|
69
|
-
at [LlamaIndex LLM](https://docs.llamaindex.ai/en/
|
|
69
|
+
at [LlamaIndex LLM](https://docs.llamaindex.ai/en/stable/module_guides/models/llms/).
|
|
70
70
|
|
|
71
71
|
### Add more LLM models
|
|
72
72
|
|
|
@@ -7,7 +7,9 @@ myst:
|
|
|
7
7
|
---
|
|
8
8
|
# llama_index LLM
|
|
9
9
|
|
|
10
|
-
The `llama_index_llm` module is generator based
|
|
10
|
+
The `llama_index_llm` module is generator based
|
|
11
|
+
on [llama_index](https://docs.llamaindex.ai/en/stable/module_guides/models/llms/). It gets the LLM instance from llama
|
|
12
|
+
index, and returns generated text by the input prompt.
|
|
11
13
|
It does not generate log probs.
|
|
12
14
|
|
|
13
15
|
## **Module Parameters**
|
|
@@ -25,18 +25,30 @@ Please refer to the parameter of [Generator Node](../generator/generator.md) for
|
|
|
25
25
|
|
|
26
26
|
#### **Strategy Parameters**:
|
|
27
27
|
|
|
28
|
-
1. **Metrics**: Metrics such as `bleu`,`meteor`, and `rouge` are used to evaluate the performance of the
|
|
29
|
-
|
|
30
|
-
|
|
28
|
+
1. **Metrics**: (Essential) Metrics such as `bleu`,`meteor`, and `rouge` are used to evaluate the performance of the
|
|
29
|
+
prompt maker process through its impact on generator (llm) outcomes.
|
|
30
|
+
2. **Speed Threshold**: (Optional) `speed_threshold` is applied across all nodes, ensuring that any method exceeding the
|
|
31
|
+
average processing time for a query is not utilized.
|
|
32
|
+
3. **Token Threshold**: (Optional) `token_threshold` ensuring that output prompt average token length does not exceed
|
|
33
|
+
the
|
|
31
34
|
threshold.
|
|
32
|
-
4. **tokenizer**: Since you don't know what LLM model you will use in the next nodes, you can specify the
|
|
35
|
+
4. **tokenizer**: (Optional) Since you don't know what LLM model you will use in the next nodes, you can specify the
|
|
36
|
+
tokenizer name
|
|
33
37
|
to use in `token_threshold` strategy.
|
|
34
38
|
You can use OpenAI model names or Huggingface model names that support `AutoTokenizer`.
|
|
35
39
|
It will automatically find the tokenizer for the model name you specify.
|
|
36
40
|
Default is 'gpt2'.
|
|
37
|
-
5. **Generator Modules**: The prompt maker node can use all modules and module
|
|
41
|
+
5. **Generator Modules**: (Optional, but recommended to set) The prompt maker node can use all modules and module
|
|
42
|
+
parameters from the generator node,
|
|
38
43
|
including:
|
|
39
44
|
- [llama_index_llm](../generator/llama_index_llm.md): with `llm` and additional llm parameters
|
|
45
|
+
- [openai_llm](../generator/openai_llm.md): with `llm` and additional openai `AsyncOpenAI` parameters
|
|
46
|
+
- [vllm](../generator/vllm.md): with `llm` and additional vllm parameters
|
|
47
|
+
|
|
48
|
+
And the default model of generator module for evaluating prompt maker is openai gpt-3.5-turbo model.
|
|
49
|
+
|
|
50
|
+
Plus, the evaluation of prompt maker will skip when there are only one combination of prompt maker modules and
|
|
51
|
+
options.
|
|
40
52
|
|
|
41
53
|
### Example config.yaml file
|
|
42
54
|
```yaml
|
|
@@ -69,4 +81,4 @@ maxdepth: 1
|
|
|
69
81
|
fstring.md
|
|
70
82
|
long_context_reorder.md
|
|
71
83
|
window_replacement.md
|
|
72
|
-
```
|
|
84
|
+
```
|
|
@@ -13,7 +13,8 @@ The `BM25` is the most popular TF-IDF method for retrieval, which reflects how i
|
|
|
13
13
|
|
|
14
14
|
- **bm25_tokenizer**: You can select which tokenize method you use for bm25.
|
|
15
15
|
The default method is 'porter_stemmer'.
|
|
16
|
-
And you can choose between '
|
|
16
|
+
And you can choose between 'space', and huggingface AutoTokenizer name.
|
|
17
|
+
Plus, you can choose Korean tokenizer such as 'ko_kiwi', 'ko_kkma', and 'ko_okt'.
|
|
17
18
|
|
|
18
19
|
### porter_stemmer
|
|
19
20
|
|
|
@@ -21,12 +22,6 @@ The default method to tokenize. It is optimized for English. It divides sentence
|
|
|
21
22
|
|
|
22
23
|
It means, stemmer can change 'studying', 'studies' to 'study'.
|
|
23
24
|
|
|
24
|
-
### ko_kiwi
|
|
25
|
-
|
|
26
|
-
It uses kiwi tokenizer for Korean language.
|
|
27
|
-
We highly recommend to use it for Korean documents.
|
|
28
|
-
You can check more information about kiwi at [here](https://github.com/bab2min/Kiwi).
|
|
29
|
-
|
|
30
25
|
### space
|
|
31
26
|
|
|
32
27
|
It is simple method to divide words into just space.
|
|
@@ -37,9 +32,30 @@ It is simple, but it can be a great choice for multilingual documents.
|
|
|
37
32
|
You can use any `AutoTokenizer` from huggingface, like gpt2 or mistralai/Mistral-7B-Instruct-v0.2.
|
|
38
33
|
Just type huggingface repo path, and you can use the tokenizer.
|
|
39
34
|
|
|
35
|
+
### ko_kiwi (For Korean 🇰🇷)
|
|
36
|
+
|
|
37
|
+
It uses kiwi tokenizer for Korean language.
|
|
38
|
+
We highly recommend to use it for Korean documents.
|
|
39
|
+
You can check more information about kiwi at [here](https://github.com/bab2min/Kiwi).
|
|
40
|
+
|
|
41
|
+
### ko_kkma (For Korean 🇰🇷)
|
|
42
|
+
|
|
43
|
+
It uses okt tokenizer for Korean. You have to install `konlpy` to use this tokenizer.
|
|
44
|
+
You can check more information about kkma at [here](https://konlpy.org/ko/latest/api/konlpy.tag/#konlpy.tag._kkma.Kkma).
|
|
45
|
+
|
|
46
|
+
### ko_okt (For Korean 🇰🇷)
|
|
47
|
+
|
|
48
|
+
It uses okt tokenizer for Korean. You have to install `konlpy` to use this tokenizer.
|
|
49
|
+
You can check more information about okt at [here](https://konlpy.org/ko/latest/api/konlpy.tag/#konlpy.tag._okt.Okt).
|
|
50
|
+
|
|
51
|
+
```{admonition} Any trouble to use Korean tokenizer?
|
|
52
|
+
You need to install extra dependencies to properly use Korean tokenizer.
|
|
53
|
+
Please go to [here](https://docs.auto-rag.com/install.html) to look at the installation guide.
|
|
54
|
+
```
|
|
55
|
+
|
|
40
56
|
## **Example config.yaml**
|
|
41
57
|
```yaml
|
|
42
58
|
modules:
|
|
43
59
|
- module_type: bm25
|
|
44
|
-
bm25_tokenizer: [ porter_stemmer, ko_kiwi, space, gpt2 ]
|
|
60
|
+
bm25_tokenizer: [ porter_stemmer, ko_kiwi, space, gpt2, ko_kkma, ko_okt ]
|
|
45
61
|
```
|
|
@@ -12,11 +12,17 @@ In this documentation, you can learn about how AutoRAG works under the hood.
|
|
|
12
12
|
|
|
13
13
|
## Swapping modules in Node
|
|
14
14
|
|
|
15
|
-

|
|
16
16
|
|
|
17
17
|
Here is the diagram of the overall AutoRAG pipeline.
|
|
18
18
|
Each node represents a node, and each node's result passed to the next node.
|
|
19
19
|
|
|
20
|
+
```{admonition} Do I need to use all nodes?
|
|
21
|
+
No. The essential node for the 'working' RAG pipeline is `retrieval`, `prompt maker` and `generator`.
|
|
22
|
+
|
|
23
|
+
The other nodes are optional, so you can add it for the better performance.
|
|
24
|
+
```
|
|
25
|
+
|
|
20
26
|
But remember, you can set multiple modules and multiple parameters in each node.
|
|
21
27
|
And you get the best result among them.
|
|
22
28
|
|