AutoRAG 0.0.3__tar.gz → 0.0.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/AutoRAG.egg-info/PKG-INFO +2 -2
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/AutoRAG.egg-info/SOURCES.txt +11 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/PKG-INFO +2 -2
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/README.md +1 -1
- AutoRAG-0.0.4/autorag/VERSION +1 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/deploy.py +21 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/nodes/passagecompressor/tree_summarize.py +1 -1
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/nodes/retrieval/hybrid_cc.py +2 -1
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/nodes/retrieval/run.py +28 -19
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/docs/source/conf.py +1 -1
- AutoRAG-0.0.4/docs/source/data_creation/tutorial.md +1 -0
- AutoRAG-0.0.4/docs/source/index.rst +45 -0
- AutoRAG-0.0.4/docs/source/install.md +1 -0
- AutoRAG-0.0.4/docs/source/modules/index.md +1 -0
- AutoRAG-0.0.4/docs/source/optimization/optimization.md +3 -0
- AutoRAG-0.0.4/docs/source/resources/samsung_sundae.jpeg +0 -0
- AutoRAG-0.0.4/docs/source/tutorial.md +1 -0
- AutoRAG-0.0.4/sample_dataset/README.md +25 -0
- AutoRAG-0.0.4/sample_dataset/eli5/load_eli5_dataset.py +30 -0
- AutoRAG-0.0.4/sample_dataset/msmarco/load_msmarco_dataset.py +30 -0
- AutoRAG-0.0.4/sample_dataset/triviaqa/load_triviaqa_dataset.py +30 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/autorag/nodes/retrieval/test_hybrid_cc.py +11 -1
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/autorag/nodes/retrieval/test_run_retrieval_node.py +46 -1
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/autorag/test_deploy.py +24 -1
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/autorag/test_evaluator.py +46 -1
- AutoRAG-0.0.4/tests/resources/qa_test_data_sample.parquet +0 -0
- AutoRAG-0.0.3/autorag/VERSION +0 -1
- AutoRAG-0.0.3/docs/source/index.rst +0 -22
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/.github/dependabot.yml +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/.github/workflows/sphinx.yml +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/.github/workflows/test.yml +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/.gitignore +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/AutoRAG.egg-info/dependency_links.txt +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/AutoRAG.egg-info/entry_points.txt +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/AutoRAG.egg-info/requires.txt +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/AutoRAG.egg-info/top_level.txt +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/LICENSE +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/__init__.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/data/__init__.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/evaluate/__init__.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/evaluate/generation.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/evaluate/metric/__init__.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/evaluate/metric/generation.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/evaluate/metric/retrieval.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/evaluate/metric/retrieval_contents.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/evaluate/retrieval.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/evaluate/retrieval_contents.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/evaluator.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/node_line.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/nodes/__init__.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/nodes/generator/__init__.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/nodes/generator/base.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/nodes/generator/llama_index_llm.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/nodes/generator/run.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/nodes/passagecompressor/__init__.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/nodes/passagecompressor/base.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/nodes/passagecompressor/run.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/nodes/passagereranker/__init__.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/nodes/passagereranker/base.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/nodes/passagereranker/monot5.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/nodes/passagereranker/run.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/nodes/passagereranker/tart/__init__.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/nodes/passagereranker/tart/modeling_enc_t5.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/nodes/passagereranker/tart/tart.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/nodes/passagereranker/tart/tokenization_enc_t5.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/nodes/passagereranker/upr.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/nodes/promptmaker/__init__.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/nodes/promptmaker/base.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/nodes/promptmaker/fstring.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/nodes/promptmaker/run.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/nodes/queryexpansion/__init__.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/nodes/queryexpansion/base.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/nodes/queryexpansion/hyde.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/nodes/queryexpansion/query_decompose.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/nodes/queryexpansion/run.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/nodes/retrieval/__init__.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/nodes/retrieval/base.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/nodes/retrieval/bm25.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/nodes/retrieval/hybrid_rrf.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/nodes/retrieval/vectordb.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/schema/__init__.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/schema/module.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/schema/node.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/strategy.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/support.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/utils/__init__.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/utils/preprocess.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/autorag/utils/util.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/dev_requirements.txt +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/docs/Makefile +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/docs/make.bat +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/docs/requirements.txt +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/docs/source/api_spec/autorag.data.rst +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/docs/source/api_spec/autorag.evaluate.metric.rst +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/docs/source/api_spec/autorag.evaluate.rst +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/docs/source/api_spec/autorag.nodes.generator.rst +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/docs/source/api_spec/autorag.nodes.passagecompressor.rst +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/docs/source/api_spec/autorag.nodes.passagereranker.rst +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/docs/source/api_spec/autorag.nodes.passagereranker.tart.rst +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/docs/source/api_spec/autorag.nodes.promptmaker.rst +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/docs/source/api_spec/autorag.nodes.queryexpansion.rst +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/docs/source/api_spec/autorag.nodes.retrieval.rst +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/docs/source/api_spec/autorag.nodes.rst +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/docs/source/api_spec/autorag.rst +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/docs/source/api_spec/autorag.schema.rst +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/docs/source/api_spec/autorag.utils.rst +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/docs/source/api_spec/modules.rst +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/pyproject.toml +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/requirements.txt +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/setup.cfg +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/autorag/evaluate/metric/test_generation_metric.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/autorag/evaluate/metric/test_retrieval_contents_metric.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/autorag/evaluate/metric/test_retrieval_metric.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/autorag/evaluate/test_generation_evaluate.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/autorag/evaluate/test_retrieval_contents_evaluate.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/autorag/evaluate/test_retrieval_evaluate.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/autorag/nodes/generator/test_llama_index_llm.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/autorag/nodes/generator/test_run_generator_node.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/autorag/nodes/passagecompressor/test_run_passage_compressor_node.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/autorag/nodes/passagecompressor/test_tree_summarize.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/autorag/nodes/passagereranker/test_monot5.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/autorag/nodes/passagereranker/test_passage_reranker_base.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/autorag/nodes/passagereranker/test_passage_reranker_run.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/autorag/nodes/passagereranker/test_tart.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/autorag/nodes/passagereranker/test_upr.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/autorag/nodes/promptmaker/test_fstring.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/autorag/nodes/promptmaker/test_prompt_maker_run.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/autorag/nodes/queryexpansion/test_hyde.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/autorag/nodes/queryexpansion/test_query_decompose.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/autorag/nodes/queryexpansion/test_query_expansion_base.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/autorag/nodes/queryexpansion/test_query_expansion_run.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/autorag/nodes/retrieval/test_bm25.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/autorag/nodes/retrieval/test_hybrid_rrf.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/autorag/nodes/retrieval/test_retrieval_base.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/autorag/nodes/retrieval/test_vectordb.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/autorag/schema/test_module_schema.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/autorag/schema/test_node_schema.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/autorag/test_strategy.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/autorag/test_support.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/autorag/utils/test_preprocess.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/autorag/utils/test_util.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/conftest.py +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/resources/corpus_data_sample.parquet +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/resources/full.yaml +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/resources/qa_data_sample.parquet +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/resources/result_project/0/config.yaml +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/resources/result_project/0/post_retrieve_node_line/generator/0.parquet +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/resources/result_project/0/post_retrieve_node_line/generator/1.parquet +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/resources/result_project/0/post_retrieve_node_line/generator/2.parquet +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/resources/result_project/0/post_retrieve_node_line/generator/3.parquet +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/resources/result_project/0/post_retrieve_node_line/generator/4.parquet +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/resources/result_project/0/post_retrieve_node_line/generator/5.parquet +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/resources/result_project/0/post_retrieve_node_line/generator/best_5.parquet +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/resources/result_project/0/post_retrieve_node_line/generator/summary.csv +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/resources/result_project/0/post_retrieve_node_line/prompt_maker/0.parquet +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/resources/result_project/0/post_retrieve_node_line/prompt_maker/1.parquet +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/resources/result_project/0/post_retrieve_node_line/prompt_maker/best_0.parquet +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/resources/result_project/0/post_retrieve_node_line/prompt_maker/summary.csv +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/resources/result_project/0/post_retrieve_node_line/summary.csv +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/resources/result_project/0/pre_retrieve_node_line/query_expansion/0.parquet +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/resources/result_project/0/pre_retrieve_node_line/query_expansion/1.parquet +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/resources/result_project/0/pre_retrieve_node_line/query_expansion/2.parquet +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/resources/result_project/0/pre_retrieve_node_line/query_expansion/best_0.parquet +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/resources/result_project/0/pre_retrieve_node_line/query_expansion/summary.csv +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/resources/result_project/0/pre_retrieve_node_line/summary.csv +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/resources/result_project/0/retrieve_node_line/passage_compressor/0.parquet +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/resources/result_project/0/retrieve_node_line/passage_compressor/best_0.parquet +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/resources/result_project/0/retrieve_node_line/passage_compressor/summary.csv +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/resources/result_project/0/retrieve_node_line/passage_reranker/0.parquet +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/resources/result_project/0/retrieve_node_line/passage_reranker/1.parquet +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/resources/result_project/0/retrieve_node_line/passage_reranker/best_0.parquet +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/resources/result_project/0/retrieve_node_line/passage_reranker/summary.csv +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/resources/result_project/0/retrieve_node_line/retrieval/0.parquet +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/resources/result_project/0/retrieve_node_line/retrieval/1.parquet +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/resources/result_project/0/retrieve_node_line/retrieval/best_0.parquet +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/resources/result_project/0/retrieve_node_line/retrieval/summary.csv +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/resources/result_project/0/retrieve_node_line/summary.csv +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/resources/result_project/0/summary.csv +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/resources/result_project/data/corpus.parquet +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/resources/result_project/data/qa.parquet +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/resources/result_project/resources/bm25.pkl +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/resources/result_project/resources/chroma/6595d304-5270-4d9f-a267-c191366804ce/data_level0.bin +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/resources/result_project/resources/chroma/6595d304-5270-4d9f-a267-c191366804ce/header.bin +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/resources/result_project/resources/chroma/6595d304-5270-4d9f-a267-c191366804ce/length.bin +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/resources/result_project/resources/chroma/6595d304-5270-4d9f-a267-c191366804ce/link_lists.bin +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/resources/result_project/resources/chroma/chroma.sqlite3 +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/resources/result_project/trial.json +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/resources/sample_project/data/corpus.parquet +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/resources/sample_project/data/qa.parquet +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/resources/sample_project/resources/bm25.pkl +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/resources/simple.yaml +0 -0
- {AutoRAG-0.0.3 → AutoRAG-0.0.4}/tests/resources/test_bm25_retrieval.pkl +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: AutoRAG
|
|
3
|
-
Version: 0.0.
|
|
3
|
+
Version: 0.0.4
|
|
4
4
|
Summary: Automatically Evaluate RAG pipelines with your own data. Find optimal structure for new RAG product.
|
|
5
5
|
Author-email: Marker-Inc <vkehfdl1@gmail.com>
|
|
6
6
|
License: Apache License
|
|
@@ -310,7 +310,7 @@ For evaluation, you need to prepare just three files.
|
|
|
310
310
|
|
|
311
311
|
There is a template for your evaluation data for using AutoRAG.
|
|
312
312
|
Check out the evaluation data rule at [here]().
|
|
313
|
-
Plus, you can get example datasets for testing AutoRAG at [here]().
|
|
313
|
+
Plus, you can get example datasets for testing AutoRAG at [here](./sample_dataset).
|
|
314
314
|
|
|
315
315
|
### Evaluate your data to various RAG modules
|
|
316
316
|
|
|
@@ -74,6 +74,8 @@ docs/make.bat
|
|
|
74
74
|
docs/requirements.txt
|
|
75
75
|
docs/source/conf.py
|
|
76
76
|
docs/source/index.rst
|
|
77
|
+
docs/source/install.md
|
|
78
|
+
docs/source/tutorial.md
|
|
77
79
|
docs/source/api_spec/autorag.data.rst
|
|
78
80
|
docs/source/api_spec/autorag.evaluate.metric.rst
|
|
79
81
|
docs/source/api_spec/autorag.evaluate.rst
|
|
@@ -89,6 +91,14 @@ docs/source/api_spec/autorag.rst
|
|
|
89
91
|
docs/source/api_spec/autorag.schema.rst
|
|
90
92
|
docs/source/api_spec/autorag.utils.rst
|
|
91
93
|
docs/source/api_spec/modules.rst
|
|
94
|
+
docs/source/data_creation/tutorial.md
|
|
95
|
+
docs/source/modules/index.md
|
|
96
|
+
docs/source/optimization/optimization.md
|
|
97
|
+
docs/source/resources/samsung_sundae.jpeg
|
|
98
|
+
sample_dataset/README.md
|
|
99
|
+
sample_dataset/eli5/load_eli5_dataset.py
|
|
100
|
+
sample_dataset/msmarco/load_msmarco_dataset.py
|
|
101
|
+
sample_dataset/triviaqa/load_triviaqa_dataset.py
|
|
92
102
|
tests/conftest.py
|
|
93
103
|
tests/autorag/test_deploy.py
|
|
94
104
|
tests/autorag/test_evaluator.py
|
|
@@ -128,6 +138,7 @@ tests/autorag/utils/test_util.py
|
|
|
128
138
|
tests/resources/corpus_data_sample.parquet
|
|
129
139
|
tests/resources/full.yaml
|
|
130
140
|
tests/resources/qa_data_sample.parquet
|
|
141
|
+
tests/resources/qa_test_data_sample.parquet
|
|
131
142
|
tests/resources/simple.yaml
|
|
132
143
|
tests/resources/test_bm25_retrieval.pkl
|
|
133
144
|
tests/resources/result_project/trial.json
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: AutoRAG
|
|
3
|
-
Version: 0.0.
|
|
3
|
+
Version: 0.0.4
|
|
4
4
|
Summary: Automatically Evaluate RAG pipelines with your own data. Find optimal structure for new RAG product.
|
|
5
5
|
Author-email: Marker-Inc <vkehfdl1@gmail.com>
|
|
6
6
|
License: Apache License
|
|
@@ -310,7 +310,7 @@ For evaluation, you need to prepare just three files.
|
|
|
310
310
|
|
|
311
311
|
There is a template for your evaluation data for using AutoRAG.
|
|
312
312
|
Check out the evaluation data rule at [here]().
|
|
313
|
-
Plus, you can get example datasets for testing AutoRAG at [here]().
|
|
313
|
+
Plus, you can get example datasets for testing AutoRAG at [here](./sample_dataset).
|
|
314
314
|
|
|
315
315
|
### Evaluate your data to various RAG modules
|
|
316
316
|
|
|
@@ -66,7 +66,7 @@ For evaluation, you need to prepare just three files.
|
|
|
66
66
|
|
|
67
67
|
There is a template for your evaluation data for using AutoRAG.
|
|
68
68
|
Check out the evaluation data rule at [here]().
|
|
69
|
-
Plus, you can get example datasets for testing AutoRAG at [here]().
|
|
69
|
+
Plus, you can get example datasets for testing AutoRAG at [here](./sample_dataset).
|
|
70
70
|
|
|
71
71
|
### Evaluate your data to various RAG modules
|
|
72
72
|
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
0.0.4
|
|
@@ -27,6 +27,20 @@ def extract_node_line_names(config_dict: Dict) -> List[str]:
|
|
|
27
27
|
return [node_line['node_line_name'] for node_line in config_dict['node_lines']]
|
|
28
28
|
|
|
29
29
|
|
|
30
|
+
def extract_node_strategy(config_dict: Dict) -> Dict:
|
|
31
|
+
"""
|
|
32
|
+
Extract node strategies with the given config dictionary.
|
|
33
|
+
The return value is a dictionary of node type and its strategy.
|
|
34
|
+
|
|
35
|
+
:param config_dict: The yaml configuration dict for the pipeline.
|
|
36
|
+
You can load this to access trail_folder/config.yaml.
|
|
37
|
+
:return: Key is node_type and value is strategy dict.
|
|
38
|
+
"""
|
|
39
|
+
return {node['node_type']: node.get('strategy', {})
|
|
40
|
+
for node_line in config_dict['node_lines']
|
|
41
|
+
for node in node_line['nodes']}
|
|
42
|
+
|
|
43
|
+
|
|
30
44
|
def summary_df_to_yaml(summary_df: pd.DataFrame, config_dict: Dict) -> Dict:
|
|
31
45
|
"""
|
|
32
46
|
Convert trial summary dataframe to config yaml file.
|
|
@@ -41,6 +55,12 @@ def summary_df_to_yaml(summary_df: pd.DataFrame, config_dict: Dict) -> Dict:
|
|
|
41
55
|
# summary_df columns : 'node_line_name', 'node_type', 'best_module_filename',
|
|
42
56
|
# 'best_module_name', 'best_module_params', 'best_execution_time'
|
|
43
57
|
node_line_names = extract_node_line_names(config_dict)
|
|
58
|
+
node_strategies = extract_node_strategy(config_dict)
|
|
59
|
+
strategy_df = pd.DataFrame({
|
|
60
|
+
'node_type': list(node_strategies.keys()),
|
|
61
|
+
'strategy': list(node_strategies.values())
|
|
62
|
+
})
|
|
63
|
+
summary_df = summary_df.merge(strategy_df, on='node_type', how='left')
|
|
44
64
|
summary_df['categorical_node_line_name'] = pd.Categorical(summary_df['node_line_name'], categories=node_line_names,
|
|
45
65
|
ordered=True)
|
|
46
66
|
summary_df = summary_df.sort_values(by='categorical_node_line_name')
|
|
@@ -52,6 +72,7 @@ def summary_df_to_yaml(summary_df: pd.DataFrame, config_dict: Dict) -> Dict:
|
|
|
52
72
|
'nodes': [
|
|
53
73
|
{
|
|
54
74
|
'node_type': row['node_type'],
|
|
75
|
+
'strategy': row['strategy'],
|
|
55
76
|
'modules': [{
|
|
56
77
|
'module_type': row['best_module_name'],
|
|
57
78
|
**row['best_module_params']
|
|
@@ -26,7 +26,7 @@ def tree_summarize(queries: List[str],
|
|
|
26
26
|
"""
|
|
27
27
|
Recursively merge retrieved texts and summarizes them in a bottom-up fashion.
|
|
28
28
|
This function is a wrapper for llama_index.response_synthesizers.TreeSummarize.
|
|
29
|
-
For more information, visit https://docs.llamaindex.ai/en/
|
|
29
|
+
For more information, visit https://docs.llamaindex.ai/en/latest/examples/response_synthesizers/tree_summarize.html.
|
|
30
30
|
|
|
31
31
|
:param queries: The queries for retrieved passages.
|
|
32
32
|
:param contents: The contents of retrieved passages.
|
|
@@ -55,8 +55,9 @@ def hybrid_cc(
|
|
|
55
55
|
|
|
56
56
|
def cc_pure(ids: Tuple, scores: Tuple, weights: Tuple, top_k: int) -> Tuple[
|
|
57
57
|
List[str], List[float]]:
|
|
58
|
-
df = pd.concat([pd.Series(dict(zip(_id, score))) for _id, score in zip(ids, scores)], axis=1
|
|
58
|
+
df = pd.concat([pd.Series(dict(zip(_id, score))) for _id, score in zip(ids, scores)], axis=1)
|
|
59
59
|
normalized_scores = (df - df.min()) / (df.max() - df.min())
|
|
60
|
+
normalized_scores = normalized_scores.fillna(0)
|
|
60
61
|
normalized_scores['weighted_sum'] = normalized_scores.mul(weights).sum(axis=1)
|
|
61
62
|
normalized_scores = normalized_scores.sort_values(by='weighted_sum', ascending=False)
|
|
62
63
|
return normalized_scores.index.tolist()[:top_k], normalized_scores['weighted_sum'][:top_k].tolist()
|
|
@@ -68,28 +68,37 @@ def run_retrieval_node(modules: List[Callable],
|
|
|
68
68
|
|
|
69
69
|
# run retrieval modules except hybrid
|
|
70
70
|
hybrid_module_names = ['hybrid_rrf', 'hybrid_cc']
|
|
71
|
-
non_hybrid_modules, non_hybrid_module_params = zip(*filter(lambda x: x[0].__name__ not in hybrid_module_names,
|
|
72
|
-
zip(modules, module_params)))
|
|
73
71
|
filename_first = 0
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
72
|
+
if any([module.__name__ not in hybrid_module_names for module in modules]):
|
|
73
|
+
non_hybrid_modules, non_hybrid_module_params = zip(*filter(lambda x: x[0].__name__ not in hybrid_module_names,
|
|
74
|
+
zip(modules, module_params)))
|
|
75
|
+
non_hybrid_results, non_hybrid_times, non_hybrid_summary_df = run_and_save(non_hybrid_modules,
|
|
76
|
+
non_hybrid_module_params, filename_first)
|
|
77
|
+
filename_first += len(non_hybrid_modules)
|
|
78
|
+
else:
|
|
79
|
+
non_hybrid_results, non_hybrid_times, non_hybrid_summary_df = [], [], pd.DataFrame()
|
|
77
80
|
|
|
78
81
|
if any([module.__name__ in hybrid_module_names for module in modules]):
|
|
79
82
|
hybrid_modules, hybrid_module_params = zip(*filter(lambda x: x[0].__name__ in hybrid_module_names,
|
|
80
83
|
zip(modules, module_params)))
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
84
|
+
if all(['target_module_params' in x for x in hybrid_module_params]):
|
|
85
|
+
# If target_module_params are already given, run hybrid retrieval directly
|
|
86
|
+
hybrid_results, hybrid_times, hybrid_summary_df = run_and_save(hybrid_modules, hybrid_module_params,
|
|
87
|
+
filename_first)
|
|
88
|
+
filename_first += len(hybrid_modules)
|
|
89
|
+
else:
|
|
90
|
+
target_modules = list(map(lambda x: x.pop('target_modules'), hybrid_module_params))
|
|
91
|
+
target_filenames = list(map(lambda x: select_result_for_hybrid(save_dir, x), target_modules))
|
|
92
|
+
ids_scores = list(map(lambda x: get_ids_and_scores(save_dir, x), target_filenames))
|
|
93
|
+
target_module_params = list(map(lambda x: get_module_params(save_dir, x), target_filenames))
|
|
94
|
+
hybrid_module_params = list(map(lambda x: {**x[0], **x[1]}, zip(hybrid_module_params, ids_scores)))
|
|
95
|
+
real_hybrid_times = list(map(lambda filename: get_hybrid_execution_times(save_dir, filename), target_filenames))
|
|
96
|
+
hybrid_results, hybrid_times, hybrid_summary_df = run_and_save(hybrid_modules, hybrid_module_params,
|
|
97
|
+
filename_first)
|
|
98
|
+
filename_first += len(hybrid_modules)
|
|
99
|
+
hybrid_times = real_hybrid_times.copy()
|
|
100
|
+
hybrid_summary_df['execution_time'] = hybrid_times
|
|
101
|
+
hybrid_summary_df = edit_summary_df_params(hybrid_summary_df, target_modules, target_module_params)
|
|
93
102
|
else:
|
|
94
103
|
hybrid_results, hybrid_times, hybrid_summary_df = [], [], pd.DataFrame()
|
|
95
104
|
|
|
@@ -161,11 +170,11 @@ def select_result_for_hybrid(node_dir: str, target_modules: Tuple) -> List[str]:
|
|
|
161
170
|
return best_filenames
|
|
162
171
|
|
|
163
172
|
|
|
164
|
-
def get_module_params(node_dir: str, filenames: List[str]) ->
|
|
173
|
+
def get_module_params(node_dir: str, filenames: List[str]) -> Tuple[Dict]:
|
|
165
174
|
summary_df = load_summary_file(os.path.join(node_dir, "summary.csv"))
|
|
166
175
|
best_results = summary_df[summary_df['filename'].isin(filenames)]
|
|
167
176
|
module_params = best_results['module_params'].tolist()
|
|
168
|
-
return module_params
|
|
177
|
+
return tuple(module_params)
|
|
169
178
|
|
|
170
179
|
|
|
171
180
|
def edit_summary_df_params(summary_df: pd.DataFrame, target_modules, target_module_params) -> pd.DataFrame:
|
|
@@ -10,7 +10,7 @@ project = 'AutoRAG'
|
|
|
10
10
|
copyright = '2024, Marker-Inc'
|
|
11
11
|
author = 'Marker-Inc'
|
|
12
12
|
|
|
13
|
-
with open('../../VERSION') as f:
|
|
13
|
+
with open('../../autorag/VERSION') as f:
|
|
14
14
|
version = f.read().strip()
|
|
15
15
|
|
|
16
16
|
# -- General configuration ---------------------------------------------------
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
# Start creating your own evaluation data
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
.. AutoRAG documentation master file, created by
|
|
2
|
+
sphinx-quickstart on Wed Jan 17 20:55:21 2024.
|
|
3
|
+
You can adapt this file completely to your liking, but it should at least
|
|
4
|
+
contain the root `toctree` directive.
|
|
5
|
+
|
|
6
|
+
AutoRAG documentation
|
|
7
|
+
===================================
|
|
8
|
+
|
|
9
|
+
.. toctree::
|
|
10
|
+
:maxdepth: 1
|
|
11
|
+
:caption: Getting Started
|
|
12
|
+
:hidden:
|
|
13
|
+
|
|
14
|
+
install.md
|
|
15
|
+
tutorial.md
|
|
16
|
+
|
|
17
|
+
.. toctree::
|
|
18
|
+
:maxdepth: 2
|
|
19
|
+
:caption: Data Creation
|
|
20
|
+
:hidden:
|
|
21
|
+
|
|
22
|
+
data_creation/tutorial.md
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
.. toctree::
|
|
26
|
+
:maxdepth: 2
|
|
27
|
+
:caption: Optimization
|
|
28
|
+
:hidden:
|
|
29
|
+
|
|
30
|
+
optimization/optimization.md
|
|
31
|
+
|
|
32
|
+
.. toctree::
|
|
33
|
+
:maxdepth: 2
|
|
34
|
+
:caption: Available Modules
|
|
35
|
+
:hidden:
|
|
36
|
+
|
|
37
|
+
modules/index.md
|
|
38
|
+
|
|
39
|
+
.. toctree::
|
|
40
|
+
:maxdepth: 1
|
|
41
|
+
:caption: API Reference
|
|
42
|
+
:hidden:
|
|
43
|
+
|
|
44
|
+
api_spec/modules
|
|
45
|
+
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
# Installation
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
# Available Module List
|
|
Binary file
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
# Tutorial
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
# sample_dataset handling
|
|
2
|
+
|
|
3
|
+
The sample_dataset folder does not includes a `qa.parquet`, `corpus.parquet` file that is significantly large and cannot be uploaded directly to Git due to size limitations.
|
|
4
|
+
|
|
5
|
+
To prepare and use datasets available in the sample_dataset folder, specifically `triviaqa`, `msmarco` and `eli5`, you can follow the outlined methods below.
|
|
6
|
+
|
|
7
|
+
## Usage
|
|
8
|
+
|
|
9
|
+
The example provided uses `triviaqa`, but the same approach applies to `msmarco` and `eli5`.
|
|
10
|
+
|
|
11
|
+
### 1. Run with a specified save path
|
|
12
|
+
To execute the Python script from the terminal and save the dataset to a specified path, use the command:
|
|
13
|
+
|
|
14
|
+
```bash
|
|
15
|
+
python ./sample_dataset/triviaqa/load_triviaqa_dataset.py --save_path /path/to/save/dataset
|
|
16
|
+
```
|
|
17
|
+
This runs the `load_triviaqa_dataset.py` script located in the `./sample_dataset/triviaqa/` directory,
|
|
18
|
+
using the `--save_path` argument to specify the dataset's save location.
|
|
19
|
+
|
|
20
|
+
### 2. Run without specifying a save path
|
|
21
|
+
If you run the script without the `--save_path` argument, the dataset will be saved to a default location, which is the directory containing the `load_triviaqa_dataset.py` file, essentially `./sample_dataset/triviaqa/`:
|
|
22
|
+
```bash
|
|
23
|
+
python ./sample_dataset/triviaqa/load_triviaqa_dataset.py
|
|
24
|
+
```
|
|
25
|
+
This behavior allows for a straightforward execution without needing to specify a path, making it convenient for quick tests or when working directly within the target directory.
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
import os
|
|
2
|
+
import pathlib
|
|
3
|
+
import click
|
|
4
|
+
|
|
5
|
+
from datasets import load_dataset
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
@click.command()
|
|
9
|
+
@click.option('--save_path', type=str, default=pathlib.PurePath(__file__).parent, help='Path to save sample eli5 dataset.')
|
|
10
|
+
def load_eli5_dataset(save_path):
|
|
11
|
+
# set file path
|
|
12
|
+
file_path = "MarkrAI/eli5_sample_autorag"
|
|
13
|
+
|
|
14
|
+
# load dataset
|
|
15
|
+
corpus_dataset = load_dataset(file_path, "corpus")['train'].to_pandas()
|
|
16
|
+
qa_train_dataset = load_dataset(file_path, "qa")['train'].to_pandas()
|
|
17
|
+
qa_test_dataset = load_dataset(file_path, "qa")['test'].to_pandas()
|
|
18
|
+
|
|
19
|
+
# save data
|
|
20
|
+
if os.path.exists(os.path.join(save_path, "corpus.parquet")) is True:
|
|
21
|
+
raise ValueError("corpus.parquet already exists")
|
|
22
|
+
if os.path.exists(os.path.join(save_path, "qa.parquet")) is True:
|
|
23
|
+
raise ValueError("qa.parquet already exists")
|
|
24
|
+
corpus_dataset.to_parquet(os.path.join(save_path, "corpus.parquet"))
|
|
25
|
+
qa_train_dataset.to_parquet(os.path.join(save_path, "qa_train.parquet"))
|
|
26
|
+
qa_test_dataset.to_parquet(os.path.join(save_path, "qa_test.parquet"))
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
if __name__ == '__main__':
|
|
30
|
+
load_eli5_dataset()
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
import os
|
|
2
|
+
import pathlib
|
|
3
|
+
import click
|
|
4
|
+
|
|
5
|
+
from datasets import load_dataset
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
@click.command()
|
|
9
|
+
@click.option('--save_path', type=str, default=pathlib.PurePath(__file__).parent, help='Path to save sample msmarco dataset.')
|
|
10
|
+
def load_msmarco_dataset(save_path):
|
|
11
|
+
# set file path
|
|
12
|
+
file_path = "MarkrAI/msmarco_sample_autorag"
|
|
13
|
+
|
|
14
|
+
# load dataset
|
|
15
|
+
corpus_dataset = load_dataset(file_path, "corpus")['train'].to_pandas()
|
|
16
|
+
qa_train_dataset = load_dataset(file_path, "qa")['train'].to_pandas()
|
|
17
|
+
qa_test_dataset = load_dataset(file_path, "qa")['test'].to_pandas()
|
|
18
|
+
|
|
19
|
+
# save corpus data
|
|
20
|
+
if os.path.exists(os.path.join(save_path, "corpus.parquet")) is True:
|
|
21
|
+
raise ValueError("corpus.parquet already exists")
|
|
22
|
+
if os.path.exists(os.path.join(save_path, "qa.parquet")) is True:
|
|
23
|
+
raise ValueError("qa.parquet already exists")
|
|
24
|
+
corpus_dataset.to_parquet(os.path.join(save_path, "corpus.parquet"), index=False)
|
|
25
|
+
qa_train_dataset.to_parquet(os.path.join(save_path, "qa_train.parquet"), index=False)
|
|
26
|
+
qa_test_dataset.to_parquet(os.path.join(save_path, "qa_test.parquet"), index=False)
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
if __name__ == '__main__':
|
|
30
|
+
load_msmarco_dataset()
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
import os
|
|
2
|
+
import pathlib
|
|
3
|
+
|
|
4
|
+
import click
|
|
5
|
+
from datasets import load_dataset
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
@click.command()
|
|
9
|
+
@click.option('--save_path', type=str, default=pathlib.PurePath(__file__).parent, help='Path to save sample triviaqa dataset.')
|
|
10
|
+
def load_triviaqa_dataset(save_path):
|
|
11
|
+
# set file path
|
|
12
|
+
file_path = "MarkrAI/triviaqa_sample_autorag"
|
|
13
|
+
|
|
14
|
+
# load dataset
|
|
15
|
+
corpus_dataset = load_dataset(file_path, "corpus")['train'].to_pandas()
|
|
16
|
+
qa_train_dataset = load_dataset(file_path, "qa")['train'].to_pandas()
|
|
17
|
+
qa_test_dataset = load_dataset(file_path, "qa")['test'].to_pandas()
|
|
18
|
+
|
|
19
|
+
# save corpus data
|
|
20
|
+
if os.path.exists(os.path.join(save_path, "corpus.parquet")) is True:
|
|
21
|
+
raise ValueError("corpus.parquet already exists")
|
|
22
|
+
if os.path.exists(os.path.join(save_path, "qa.parquet")) is True:
|
|
23
|
+
raise ValueError("qa.parquet already exists")
|
|
24
|
+
corpus_dataset.to_parquet(os.path.join(save_path, "corpus.parquet"), index=False)
|
|
25
|
+
qa_train_dataset.to_parquet(os.path.join(save_path, "qa_train.parquet"), index=False)
|
|
26
|
+
qa_test_dataset.to_parquet(os.path.join(save_path, "qa_test.parquet"), index=False)
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
if __name__ == '__main__':
|
|
30
|
+
load_triviaqa_dataset()
|
|
@@ -16,6 +16,16 @@ def test_cc_pure():
|
|
|
16
16
|
assert result_id == ['id-1', 'id-4', 'id-2']
|
|
17
17
|
|
|
18
18
|
|
|
19
|
+
def test_cc_non_overlap():
|
|
20
|
+
sample_ids = (['id-1', 'id-2', 'id-3', 'id-4', 'id-5'],
|
|
21
|
+
['id-6', 'id-4', 'id-3', 'id-7', 'id-2'])
|
|
22
|
+
sample_scores = ([5, 3, 1, 0.4, 0.2], [6, 2, 1, 0.5, 0.1])
|
|
23
|
+
result_id, result_scores = cc_pure(sample_ids, sample_scores,
|
|
24
|
+
weights=(0.3, 0.7), top_k=3)
|
|
25
|
+
assert result_scores == pytest.approx([0.7, 0.3, 0.23792372])
|
|
26
|
+
assert result_id == ['id-6', 'id-1', 'id-4']
|
|
27
|
+
|
|
28
|
+
|
|
19
29
|
def test_hybrid_cc():
|
|
20
30
|
sample_ids = ([
|
|
21
31
|
['id-1', 'id-2', 'id-3', 'id-4', 'id-5'],
|
|
@@ -58,7 +68,7 @@ def test_hybrid_cc_node(pseudo_project_dir, pseudo_node_dir):
|
|
|
58
68
|
['id-1', 'id-2', 'id-3', 'id-4', 'id-5']],
|
|
59
69
|
[['id-1', 'id-4', 'id-3', 'id-5', 'id-2'],
|
|
60
70
|
['id-1', 'id-4', 'id-3', 'id-5', 'id-2']]
|
|
61
|
-
|
|
71
|
+
),
|
|
62
72
|
'scores': ([[5, 3, 1, 0.4, 0.2], [5, 3, 1, 0.4, 0.2]],
|
|
63
73
|
[[6, 2, 1, 0.5, 0.1], [6, 2, 1, 0.5, 0.1]]),
|
|
64
74
|
'top_k': 3,
|
|
@@ -95,7 +95,7 @@ def test_run_retrieval_node(node_line_dir):
|
|
|
95
95
|
assert all(hybrid_summary_df['module_params'].apply(lambda x: 'target_module_params' in x))
|
|
96
96
|
assert all(hybrid_summary_df['module_params'].apply(lambda x: x['target_modules'] == ('bm25', 'vectordb')))
|
|
97
97
|
assert all(hybrid_summary_df['module_params'].apply(
|
|
98
|
-
lambda x: x['target_module_params'] ==
|
|
98
|
+
lambda x: x['target_module_params'] == ({'top_k': 4}, {'top_k': 4, 'embedding_model': 'openai'})))
|
|
99
99
|
|
|
100
100
|
# test the best file is saved properly
|
|
101
101
|
best_filename = summary_df[summary_df['is_best'] == True]['filename'].values[0]
|
|
@@ -181,3 +181,48 @@ def test_select_result_for_hybrid(pseudo_node_dir):
|
|
|
181
181
|
assert scores[1] == [[0.5, 0.6, 0.7],
|
|
182
182
|
[0.5, 0.6, 0.7],
|
|
183
183
|
[0.5, 0.6, 0.7]]
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
def test_run_retrieval_node_only_hybrid(node_line_dir):
|
|
187
|
+
modules = [hybrid_cc]
|
|
188
|
+
module_params = [
|
|
189
|
+
{'top_k': 4, 'target_modules': ('bm25', 'vectordb'), 'weights': (0.3, 0.7),
|
|
190
|
+
'target_module_params': ({'top_k': 3}, {'top_k': 3, 'embedding_model': 'openai'})},
|
|
191
|
+
]
|
|
192
|
+
project_dir = pathlib.PurePath(node_line_dir).parent.parent
|
|
193
|
+
qa_path = os.path.join(project_dir, "data", "qa.parquet")
|
|
194
|
+
strategies = {
|
|
195
|
+
'metrics': ['retrieval_f1', 'retrieval_recall'],
|
|
196
|
+
}
|
|
197
|
+
previous_result = pd.read_parquet(qa_path)
|
|
198
|
+
best_result = run_retrieval_node(modules, module_params, previous_result, node_line_dir, strategies)
|
|
199
|
+
assert os.path.exists(os.path.join(node_line_dir, "retrieval"))
|
|
200
|
+
expect_columns = ['qid', 'query', 'retrieval_gt', 'generation_gt',
|
|
201
|
+
'retrieved_contents', 'retrieved_ids', 'retrieve_scores', 'retrieval_f1', 'retrieval_recall']
|
|
202
|
+
assert all([expect_column in best_result.columns for expect_column in expect_columns])
|
|
203
|
+
# test summary feature
|
|
204
|
+
summary_path = os.path.join(node_line_dir, "retrieval", "summary.csv")
|
|
205
|
+
single_result_path = os.path.join(node_line_dir, "retrieval", "0.parquet")
|
|
206
|
+
assert os.path.exists(single_result_path)
|
|
207
|
+
single_result_df = pd.read_parquet(single_result_path)
|
|
208
|
+
assert os.path.exists(summary_path)
|
|
209
|
+
summary_df = load_summary_file(summary_path)
|
|
210
|
+
assert set(summary_df.columns) == {'filename', 'retrieval_f1', 'retrieval_recall',
|
|
211
|
+
'module_name', 'module_params', 'execution_time', 'is_best'}
|
|
212
|
+
assert len(summary_df) == 1
|
|
213
|
+
assert summary_df['filename'][0] == "0.parquet"
|
|
214
|
+
assert summary_df['retrieval_f1'][0] == single_result_df['retrieval_f1'].mean()
|
|
215
|
+
assert summary_df['retrieval_recall'][0] == single_result_df['retrieval_recall'].mean()
|
|
216
|
+
assert summary_df['module_name'][0] == "hybrid_cc"
|
|
217
|
+
assert summary_df['module_params'][0] == {'top_k': 4, 'target_modules': ('bm25', 'vectordb'), 'weights': (0.3, 0.7),
|
|
218
|
+
'target_module_params': ({'top_k': 3}, {'top_k': 3, 'embedding_model': 'openai'})}
|
|
219
|
+
assert summary_df['execution_time'][0] > 0
|
|
220
|
+
assert summary_df['is_best'][0] == True
|
|
221
|
+
assert summary_df['filename'].nunique() == len(summary_df)
|
|
222
|
+
|
|
223
|
+
# test the best file is saved properly
|
|
224
|
+
best_filename = summary_df[summary_df['is_best'] == True]['filename'].values[0]
|
|
225
|
+
best_path = os.path.join(node_line_dir, "retrieval", f'best_{best_filename}')
|
|
226
|
+
assert os.path.exists(best_path)
|
|
227
|
+
best_df = pd.read_parquet(best_path)
|
|
228
|
+
assert all([expect_column in best_df.columns for expect_column in expect_columns])
|
|
@@ -9,7 +9,8 @@ import yaml
|
|
|
9
9
|
|
|
10
10
|
from click.testing import CliRunner
|
|
11
11
|
from fastapi.testclient import TestClient
|
|
12
|
-
from autorag.deploy import summary_df_to_yaml, extract_best_config, Runner, extract_node_line_names
|
|
12
|
+
from autorag.deploy import summary_df_to_yaml, extract_best_config, Runner, extract_node_line_names, \
|
|
13
|
+
extract_node_strategy
|
|
13
14
|
from autorag.evaluator import Evaluator, cli
|
|
14
15
|
|
|
15
16
|
root_dir = pathlib.PurePath(os.path.dirname(os.path.realpath(__file__))).parent
|
|
@@ -58,6 +59,9 @@ solution_dict = {
|
|
|
58
59
|
'nodes': [
|
|
59
60
|
{
|
|
60
61
|
'node_type': 'retrieval',
|
|
62
|
+
'strategy': {
|
|
63
|
+
'metrics': ['retrieval_f1', 'retrieval_recall', 'retrieval_precision'],
|
|
64
|
+
},
|
|
61
65
|
'modules': [
|
|
62
66
|
{
|
|
63
67
|
'module_type': 'bm25',
|
|
@@ -67,6 +71,10 @@ solution_dict = {
|
|
|
67
71
|
},
|
|
68
72
|
{
|
|
69
73
|
'node_type': 'rerank',
|
|
74
|
+
'strategy': {
|
|
75
|
+
'metrics': ['retrieval_f1', 'retrieval_recall', 'retrieval_precision'],
|
|
76
|
+
'speed_threshold': 10,
|
|
77
|
+
},
|
|
70
78
|
'modules': [
|
|
71
79
|
{
|
|
72
80
|
'module_type': 'upr',
|
|
@@ -82,6 +90,9 @@ solution_dict = {
|
|
|
82
90
|
'nodes': [
|
|
83
91
|
{
|
|
84
92
|
'node_type': 'generation',
|
|
93
|
+
'strategy': {
|
|
94
|
+
'metrics': ['bleu', 'rouge'],
|
|
95
|
+
},
|
|
85
96
|
'modules': [
|
|
86
97
|
{
|
|
87
98
|
'module_type': 'gpt-4',
|
|
@@ -111,6 +122,18 @@ def test_extract_node_line_names(full_config):
|
|
|
111
122
|
assert node_line_names == ['pre_retrieve_node_line', 'retrieve_node_line', 'post_retrieve_node_line']
|
|
112
123
|
|
|
113
124
|
|
|
125
|
+
def test_extract_node_strategy(full_config):
|
|
126
|
+
node_strategies = extract_node_strategy(full_config)
|
|
127
|
+
assert set(list(node_strategies.keys())) == {
|
|
128
|
+
'query_expansion', 'retrieval', 'passage_reranker', 'passage_compressor',
|
|
129
|
+
'prompt_maker', 'generator'
|
|
130
|
+
}
|
|
131
|
+
assert node_strategies['retrieval'] == {
|
|
132
|
+
'metrics': ['retrieval_f1', 'retrieval_recall', 'retrieval_precision'],
|
|
133
|
+
'speed_threshold': 10,
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
|
|
114
137
|
def test_summary_df_to_yaml():
|
|
115
138
|
yaml_dict = summary_df_to_yaml(summary_df, solution_dict)
|
|
116
139
|
assert yaml_dict == solution_dict
|
|
@@ -2,10 +2,12 @@ import os.path
|
|
|
2
2
|
import pathlib
|
|
3
3
|
import shutil
|
|
4
4
|
import subprocess
|
|
5
|
+
import tempfile
|
|
5
6
|
|
|
6
7
|
import pandas as pd
|
|
7
8
|
import pytest
|
|
8
9
|
|
|
10
|
+
from autorag.deploy import extract_best_config
|
|
9
11
|
from autorag.evaluator import Evaluator
|
|
10
12
|
from autorag.nodes.retrieval import bm25, vectordb, hybrid_rrf
|
|
11
13
|
from autorag.nodes.retrieval.run import run_retrieval_node
|
|
@@ -35,6 +37,24 @@ def evaluator():
|
|
|
35
37
|
pass
|
|
36
38
|
|
|
37
39
|
|
|
40
|
+
@pytest.fixture
|
|
41
|
+
def test_evaluator():
|
|
42
|
+
evaluator = Evaluator(os.path.join(resource_dir, 'qa_test_data_sample.parquet'),
|
|
43
|
+
os.path.join(resource_dir, 'corpus_data_sample.parquet'))
|
|
44
|
+
yield evaluator
|
|
45
|
+
paths_to_remove = ['0', 'data', 'resources', 'trial.json']
|
|
46
|
+
|
|
47
|
+
for path in paths_to_remove:
|
|
48
|
+
full_path = os.path.join(os.getcwd(), path)
|
|
49
|
+
try:
|
|
50
|
+
if os.path.isdir(full_path):
|
|
51
|
+
shutil.rmtree(full_path)
|
|
52
|
+
else:
|
|
53
|
+
os.remove(full_path)
|
|
54
|
+
except FileNotFoundError:
|
|
55
|
+
pass
|
|
56
|
+
|
|
57
|
+
|
|
38
58
|
def test_evaluator_init(evaluator):
|
|
39
59
|
validate_qa_dataset(evaluator.qa_data)
|
|
40
60
|
validate_corpus_dataset(evaluator.corpus_data)
|
|
@@ -98,7 +118,7 @@ def test_start_trial(evaluator):
|
|
|
98
118
|
# test node line summary
|
|
99
119
|
node_line_summary_path = os.path.join(os.getcwd(), '0', 'retrieve_node_line', 'summary.csv')
|
|
100
120
|
assert os.path.exists(node_line_summary_path)
|
|
101
|
-
node_line_summary_df = load_summary_file(node_line_summary_path,["best_module_params"])
|
|
121
|
+
node_line_summary_df = load_summary_file(node_line_summary_path, ["best_module_params"])
|
|
102
122
|
assert len(node_line_summary_df) == 1
|
|
103
123
|
assert set(node_line_summary_df.columns) == {'node_type', 'best_module_filename',
|
|
104
124
|
'best_module_name', 'best_module_params', 'best_execution_time'}
|
|
@@ -180,3 +200,28 @@ def start_trial_full(evaluator):
|
|
|
180
200
|
assert os.path.exists(os.path.join(os.getcwd(), '0', 'post_retrieve_node_line', 'generator', '3.parquet'))
|
|
181
201
|
assert os.path.exists(os.path.join(os.getcwd(), '0', 'post_retrieve_node_line', 'generator', '4.parquet'))
|
|
182
202
|
assert os.path.exists(os.path.join(os.getcwd(), '0', 'post_retrieve_node_line', 'generator', '5.parquet'))
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
def test_test_data_evaluate(test_evaluator):
|
|
206
|
+
trial_folder = os.path.join(resource_dir, 'result_project', '0')
|
|
207
|
+
with tempfile.NamedTemporaryFile(mode="w+t", suffix=".yaml") as yaml_file:
|
|
208
|
+
extract_best_config(trial_folder, yaml_file.name)
|
|
209
|
+
test_evaluator.start_trial(yaml_file.name)
|
|
210
|
+
|
|
211
|
+
assert os.path.exists(os.path.join(os.getcwd(), '0'))
|
|
212
|
+
assert os.path.exists(os.path.join(os.getcwd(), 'data'))
|
|
213
|
+
assert os.path.exists(os.path.join(os.getcwd(), 'resources'))
|
|
214
|
+
assert os.path.exists(os.path.join(os.getcwd(), 'trial.json'))
|
|
215
|
+
assert os.path.exists(os.path.join(os.getcwd(), '0', 'config.yaml'))
|
|
216
|
+
assert os.path.exists(os.path.join(os.getcwd(), '0', 'retrieve_node_line'))
|
|
217
|
+
assert os.path.exists(os.path.join(os.getcwd(), '0', 'retrieve_node_line', 'retrieval'))
|
|
218
|
+
assert os.path.exists(os.path.join(os.getcwd(), '0', 'retrieve_node_line', 'retrieval', '0.parquet'))
|
|
219
|
+
assert os.path.exists(os.path.join(os.getcwd(), '0', 'retrieve_node_line', 'retrieval', 'best_0.parquet'))
|
|
220
|
+
assert os.path.exists(os.path.join(os.getcwd(), '0', 'retrieve_node_line', 'summary.csv'))
|
|
221
|
+
assert os.path.exists(os.path.join(os.getcwd(), '0', 'pre_retrieve_node_line'))
|
|
222
|
+
assert os.path.exists(os.path.join(os.getcwd(), '0', 'pre_retrieve_node_line', 'query_expansion'))
|
|
223
|
+
assert os.path.exists(os.path.join(os.getcwd(), '0', 'pre_retrieve_node_line', 'query_expansion', "0.parquet"))
|
|
224
|
+
assert os.path.exists(os.path.join(os.getcwd(), '0', 'post_retrieve_node_line'))
|
|
225
|
+
assert os.path.exists(os.path.join(os.getcwd(), '0', 'post_retrieve_node_line', 'prompt_maker'))
|
|
226
|
+
assert os.path.exists(os.path.join(os.getcwd(), '0', 'post_retrieve_node_line', 'generator'))
|
|
227
|
+
assert os.path.exists(os.path.join(os.getcwd(), '0', 'summary.csv'))
|
|
Binary file
|
AutoRAG-0.0.3/autorag/VERSION
DELETED
|
@@ -1 +0,0 @@
|
|
|
1
|
-
0.0.3
|