AutoRAG 0.2.16__tar.gz → 0.2.18__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {autorag-0.2.16 → autorag-0.2.18}/AutoRAG.egg-info/PKG-INFO +3 -3
- {autorag-0.2.16 → autorag-0.2.18}/AutoRAG.egg-info/SOURCES.txt +15 -2
- {autorag-0.2.16 → autorag-0.2.18}/AutoRAG.egg-info/requires.txt +2 -2
- {autorag-0.2.16 → autorag-0.2.18}/PKG-INFO +3 -3
- autorag-0.2.18/autorag/VERSION +1 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/__init__.py +1 -1
- {autorag-0.2.16 → autorag-0.2.18}/autorag/data/__init__.py +2 -2
- autorag-0.2.18/autorag/data/beta/extract_evidence.py +1 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/data/beta/schema.py +138 -3
- {autorag-0.2.16 → autorag-0.2.18}/autorag/data/chunk/base.py +3 -3
- {autorag-0.2.16 → autorag-0.2.18}/autorag/data/parse/clova.py +17 -1
- {autorag-0.2.16 → autorag-0.2.18}/autorag/data/parse/langchain_parse.py +8 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/data/parse/llamaparse.py +10 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/data/parse/table_hybrid_parse.py +12 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/evaluator.py +39 -9
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/generator/openai_llm.py +45 -2
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/retrieval/vectordb.py +4 -3
- {autorag-0.2.16 → autorag-0.2.18}/autorag/schema/node.py +19 -2
- {autorag-0.2.16 → autorag-0.2.18}/autorag/utils/util.py +24 -0
- autorag-0.2.18/docs/source/_static/data_creation.png +0 -0
- autorag-0.2.18/docs/source/api_spec/autorag.data.beta.filter.rst +29 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/api_spec/autorag.data.beta.rst +8 -0
- autorag-0.2.18/docs/source/api_spec/autorag.data.chunk.rst +45 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/api_spec/autorag.schema.rst +8 -0
- autorag-0.2.18/docs/source/data_creation/beta/chunk/chunk.md +168 -0
- autorag-0.2.18/docs/source/data_creation/beta/chunk/langchain_chunk.md +47 -0
- autorag-0.2.18/docs/source/data_creation/beta/chunk/llama_index_chunk.md +57 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/data_creation/beta/data_creation.md +5 -16
- autorag-0.2.18/docs/source/data_creation/beta/parse/clova.md +30 -0
- autorag-0.2.18/docs/source/data_creation/beta/parse/langchain_parse.md +123 -0
- autorag-0.2.18/docs/source/data_creation/beta/parse/llama_parse.md +29 -0
- autorag-0.2.18/docs/source/data_creation/beta/parse/parse.md +108 -0
- autorag-0.2.18/docs/source/data_creation/beta/parse/table_hybrid_parse.md +52 -0
- autorag-0.2.18/docs/source/data_creation/beta/qa_creation/answer_gen.md +66 -0
- {autorag-0.2.16/docs/source/data_creation/beta → autorag-0.2.18/docs/source/data_creation/beta/qa_creation}/filter.md +5 -0
- autorag-0.2.18/docs/source/data_creation/beta/qa_creation/qa_creation.md +165 -0
- {autorag-0.2.16/docs/source/data_creation/beta → autorag-0.2.18/docs/source/data_creation/beta/qa_creation}/query_gen.md +2 -30
- autorag-0.2.18/docs/source/data_creation/beta/tutorial.md +152 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/tutorial.md +1 -1
- {autorag-0.2.16 → autorag-0.2.18}/requirements.txt +2 -2
- {autorag-0.2.16 → autorag-0.2.18}/sample_config/parse/parse_full.yaml +7 -0
- autorag-0.2.18/tests/autorag/data/beta/test_schema.py +186 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/data/chunk/test_langchain_chunk.py +4 -4
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/data/chunk/test_llama_index_chunk.py +4 -4
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/test_evaluator.py +13 -13
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/utils/test_util.py +45 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/0/summary.csv +1 -1
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/simple.yaml +14 -1
- autorag-0.2.16/autorag/VERSION +0 -1
- autorag-0.2.16/docs/source/_static/data_creation.png +0 -0
- autorag-0.2.16/docs/source/data_creation/beta/tutorial.md +0 -37
- autorag-0.2.16/tests/autorag/data/beta/test_schema.py +0 -61
- {autorag-0.2.16 → autorag-0.2.18}/.github/FUNDING.yml +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/.github/ISSUE_TEMPLATE/bug_report.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/.github/ISSUE_TEMPLATE/feature_request.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/.github/dependabot.yml +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/.github/workflows/publish.yml +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/.github/workflows/sphinx.yml +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/.github/workflows/test.yml +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/.gitignore +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/.pre-commit-config.yaml +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/AutoRAG.egg-info/dependency_links.txt +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/AutoRAG.egg-info/entry_points.txt +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/AutoRAG.egg-info/top_level.txt +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/CODE_OF_CONDUCT.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/CONTRIBUTING.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/LICENSE +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/README.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/chunker.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/cli.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/dashboard.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/data/beta/__init__.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/data/beta/filter/__init__.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/data/beta/filter/dontknow.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/data/beta/filter/prompt.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/data/beta/generation_gt/__init__.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/data/beta/generation_gt/base.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/data/beta/generation_gt/llama_index_gen_gt.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/data/beta/generation_gt/openai_gen_gt.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/data/beta/generation_gt/prompt.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/data/beta/query/__init__.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/data/beta/query/llama_gen_query.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/data/beta/query/openai_gen_query.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/data/beta/query/prompt.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/data/beta/sample.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/data/chunk/__init__.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/data/chunk/langchain_chunk.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/data/chunk/llama_index_chunk.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/data/chunk/run.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/data/corpus/__init__.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/data/corpus/langchain.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/data/corpus/llama_index.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/data/parse/__init__.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/data/parse/base.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/data/parse/run.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/data/qacreation/__init__.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/data/qacreation/base.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/data/qacreation/llama_index.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/data/qacreation/llama_index_default_prompt.txt +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/data/qacreation/ragas.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/data/qacreation/simple.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/data/utils/__init__.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/data/utils/util.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/deploy.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/evaluation/__init__.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/evaluation/generation.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/evaluation/metric/__init__.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/evaluation/metric/g_eval_prompts/coh_detailed.txt +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/evaluation/metric/g_eval_prompts/con_detailed.txt +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/evaluation/metric/g_eval_prompts/flu_detailed.txt +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/evaluation/metric/g_eval_prompts/rel_detailed.txt +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/evaluation/metric/generation.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/evaluation/metric/retrieval.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/evaluation/metric/retrieval_contents.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/evaluation/metric/util.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/evaluation/retrieval.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/evaluation/retrieval_contents.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/evaluation/util.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/node_line.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/__init__.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/generator/__init__.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/generator/base.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/generator/llama_index_llm.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/generator/run.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/generator/vllm.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/passageaugmenter/__init__.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/passageaugmenter/base.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/passageaugmenter/pass_passage_augmenter.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/passageaugmenter/prev_next_augmenter.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/passageaugmenter/run.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/passagecompressor/__init__.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/passagecompressor/base.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/passagecompressor/longllmlingua.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/passagecompressor/pass_compressor.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/passagecompressor/refine.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/passagecompressor/run.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/passagecompressor/tree_summarize.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/passagefilter/__init__.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/passagefilter/base.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/passagefilter/pass_passage_filter.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/passagefilter/percentile_cutoff.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/passagefilter/recency.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/passagefilter/run.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/passagefilter/similarity_percentile_cutoff.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/passagefilter/similarity_threshold_cutoff.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/passagefilter/threshold_cutoff.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/passagereranker/__init__.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/passagereranker/base.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/passagereranker/cohere.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/passagereranker/colbert.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/passagereranker/flag_embedding.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/passagereranker/flag_embedding_llm.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/passagereranker/jina.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/passagereranker/koreranker.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/passagereranker/monot5.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/passagereranker/pass_reranker.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/passagereranker/rankgpt.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/passagereranker/run.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/passagereranker/sentence_transformer.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/passagereranker/tart/__init__.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/passagereranker/tart/modeling_enc_t5.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/passagereranker/tart/tart.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/passagereranker/tart/tokenization_enc_t5.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/passagereranker/time_reranker.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/passagereranker/upr.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/promptmaker/__init__.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/promptmaker/base.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/promptmaker/fstring.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/promptmaker/long_context_reorder.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/promptmaker/run.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/promptmaker/window_replacement.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/queryexpansion/__init__.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/queryexpansion/base.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/queryexpansion/hyde.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/queryexpansion/multi_query_expansion.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/queryexpansion/pass_query_expansion.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/queryexpansion/query_decompose.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/queryexpansion/run.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/retrieval/__init__.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/retrieval/base.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/retrieval/bm25.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/retrieval/hybrid_cc.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/retrieval/hybrid_rrf.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/nodes/retrieval/run.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/parser.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/schema/__init__.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/schema/metricinput.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/schema/module.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/strategy.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/support.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/utils/__init__.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/utils/preprocess.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/validator.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/autorag/web.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/Makefile +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/make.bat +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/requirements.txt +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/CNAME +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/_static/data_folder.png +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/_static/dcg.png +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/_static/f1_score.png +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/_static/map.png +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/_static/mrr.png +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/_static/ndcg.png +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/_static/ndcg_formula.png +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/_static/node_folder.png +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/_static/node_line_folder.png +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/_static/node_line_summary.png +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/_static/node_lines.png +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/_static/node_summary.png +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/_static/normal_distribution.png +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/_static/project_folder_example.png +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/_static/project_folders.png +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/_static/resources_folder.png +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/_static/roadmap/RAG_paradigms.png +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/_static/roadmap/advanced_RAG.png +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/_static/roadmap/cycle.png +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/_static/roadmap/merger.png +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/_static/roadmap/node_line_modular.png +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/_static/roadmap/policy.png +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/_static/samsung_sundae.jpeg +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/_static/score_fusion.png +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/_static/trial_folder.png +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/_static/trial_json.png +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/_static/trial_summary.png +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/_static/web_interface.png +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/_static/web_interface_gradio.png +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/api_spec/autorag.data.beta.generation_gt.rst +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/api_spec/autorag.data.beta.query.rst +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/api_spec/autorag.data.beta.schema.rst +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/api_spec/autorag.data.corpus.rst +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/api_spec/autorag.data.parse.rst +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/api_spec/autorag.data.qacreation.rst +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/api_spec/autorag.data.rst +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/api_spec/autorag.data.utils.rst +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/api_spec/autorag.evaluation.metric.rst +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/api_spec/autorag.evaluation.rst +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/api_spec/autorag.nodes.generator.rst +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/api_spec/autorag.nodes.passageaugmenter.rst +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/api_spec/autorag.nodes.passagecompressor.rst +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/api_spec/autorag.nodes.passagefilter.rst +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/api_spec/autorag.nodes.passagereranker.rst +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/api_spec/autorag.nodes.passagereranker.tart.rst +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/api_spec/autorag.nodes.promptmaker.rst +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/api_spec/autorag.nodes.queryexpansion.rst +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/api_spec/autorag.nodes.retrieval.rst +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/api_spec/autorag.nodes.rst +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/api_spec/autorag.rst +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/api_spec/autorag.utils.rst +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/api_spec/modules.rst +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/conf.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/data_creation/data_format.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/data_creation/ragas.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/data_creation/tutorial.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/deploy/api_endpoint.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/deploy/web.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/evaluate_metrics/generation.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/evaluate_metrics/retrieval.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/evaluate_metrics/retrieval_contents.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/index.rst +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/install.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/local_model.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/nodes/generator/generator.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/nodes/generator/llama_index_llm.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/nodes/generator/openai_llm.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/nodes/generator/vllm.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/nodes/index.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/nodes/passage_augmenter/passage_augmenter.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/nodes/passage_augmenter/prev_next_augmenter.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/nodes/passage_compressor/longllmlingua.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/nodes/passage_compressor/passage_compressor.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/nodes/passage_compressor/refine.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/nodes/passage_compressor/tree_summarize.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/nodes/passage_filter/passage_filter.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/nodes/passage_filter/percentile_cutoff.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/nodes/passage_filter/recency_filter.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/nodes/passage_filter/similarity_percentile_cutoff.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/nodes/passage_filter/similarity_threshold_cutoff.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/nodes/passage_filter/threshold_cutoff.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/nodes/passage_reranker/cohere.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/nodes/passage_reranker/colbert.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/nodes/passage_reranker/flag_embedding_llm_reranker.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/nodes/passage_reranker/flag_embedding_reranker.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/nodes/passage_reranker/jina_reranker.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/nodes/passage_reranker/koreranker.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/nodes/passage_reranker/monot5.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/nodes/passage_reranker/passage_reranker.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/nodes/passage_reranker/rankgpt.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/nodes/passage_reranker/sentence_transformer_reranker.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/nodes/passage_reranker/tart.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/nodes/passage_reranker/time_reranker.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/nodes/passage_reranker/upr.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/nodes/prompt_maker/fstring.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/nodes/prompt_maker/long_context_reorder.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/nodes/prompt_maker/prompt_maker.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/nodes/prompt_maker/window_replacement.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/nodes/query_expansion/hyde.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/nodes/query_expansion/multi_query_expansion.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/nodes/query_expansion/query_decompose.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/nodes/query_expansion/query_expansion.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/nodes/retrieval/bm25.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/nodes/retrieval/hybrid_cc.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/nodes/retrieval/hybrid_rrf.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/nodes/retrieval/retrieval.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/nodes/retrieval/vectordb.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/optimization/custom_config.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/optimization/folder_structure.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/optimization/optimization.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/optimization/strategies.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/roadmap/modular_rag.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/structure.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/docs/source/troubleshooting.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/pyproject.toml +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/sample_config/chunk/chunk_full.yaml +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/sample_config/compact_local.yaml +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/sample_config/compact_openai.yaml +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/sample_config/config_korean.yaml +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/sample_config/extracted_sample.yaml +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/sample_config/full.yaml +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/sample_config/parse/parse_hybird.yaml +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/sample_config/simple_local.yaml +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/sample_config/simple_ollama.yaml +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/sample_config/simple_openai.yaml +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/sample_dataset/README.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/sample_dataset/eli5/load_eli5_dataset.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/sample_dataset/hotpotqa/load_hotpotqa_dataset.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/sample_dataset/msmarco/load_msmarco_dataset.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/sample_dataset/triviaqa/load_triviaqa_dataset.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/setup.cfg +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/data/beta/filter/test_dontknow.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/data/beta/generation_gt/base_test_generation_gt.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/data/beta/generation_gt/test_llama_index_gen_gt.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/data/beta/generation_gt/test_openai_gen_gt.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/data/beta/query/base_test_query_gen.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/data/beta/query/test_llama_gen_query.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/data/beta/query/test_openai_gen_query.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/data/beta/test_data_creation_piepline.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/data/beta/test_sample.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/data/chunk/test_chunk_base.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/data/chunk/test_chunk_run.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/data/corpus/test_base.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/data/corpus/test_langchain.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/data/corpus/test_llama_index_corpus.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/data/parse/test_clova.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/data/parse/test_langchain_parse.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/data/parse/test_llamaparse.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/data/parse/test_parse_base.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/data/parse/test_parse_run.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/data/parse/test_table_hybrid_parse.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/data/qacreation/test_base_qacreation.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/data/qacreation/test_llama_index_qacreation.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/data/qacreation/test_ragas_qa_creation.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/data/qacreation/test_simple.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/evaluate/metric/test_generation_metric.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/evaluate/metric/test_retrieval_contents_metric.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/evaluate/metric/test_retrieval_metric.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/evaluate/test_evaluate_util.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/evaluate/test_generation_evaluate.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/evaluate/test_retrieval_contents_evaluate.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/evaluate/test_retrieval_evaluate.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/generator/test_generator_base.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/generator/test_llama_index_llm.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/generator/test_openai.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/generator/test_run_generator_node.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/generator/test_vllm.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/passageaugmenter/test_base_passage_augmenter.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/passageaugmenter/test_pass_passage_augmenter.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/passageaugmenter/test_prev_next_augmenter.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/passageaugmenter/test_run_passage_augmenter.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/passagecompressor/test_base_passage_compressor.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/passagecompressor/test_longllmlingua.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/passagecompressor/test_pass_compressor.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/passagecompressor/test_refine.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/passagecompressor/test_run_passage_compressor_node.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/passagecompressor/test_tree_summarize.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/passagefilter/test_pass_passage_filter.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/passagefilter/test_passage_filter_base.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/passagefilter/test_passage_filter_run.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/passagefilter/test_percentile_cutoff.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/passagefilter/test_recency_filter.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/passagefilter/test_similarity_percentile_cutoff.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/passagefilter/test_similarity_threshold_cutoff.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/passagefilter/test_threshold_cutoff.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/passagereranker/test_cohere_reranker.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/passagereranker/test_colbert_reranker.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/passagereranker/test_flag_embedding.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/passagereranker/test_flag_embedding_llm.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/passagereranker/test_jina_reranker.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/passagereranker/test_koreranker.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/passagereranker/test_monot5.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/passagereranker/test_pass_reranker.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/passagereranker/test_passage_reranker_base.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/passagereranker/test_passage_reranker_run.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/passagereranker/test_rankgpt.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/passagereranker/test_sentence_transformer.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/passagereranker/test_tart.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/passagereranker/test_time_reranker.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/passagereranker/test_upr.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/promptmaker/test_fstring.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/promptmaker/test_long_context_reorder.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/promptmaker/test_prompt_maker_base.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/promptmaker/test_prompt_maker_run.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/promptmaker/test_window_replacement.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/queryexpansion/test_hyde.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/queryexpansion/test_multi_query_expansion.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/queryexpansion/test_pass_query_expansion.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/queryexpansion/test_query_decompose.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/queryexpansion/test_query_expansion_base.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/queryexpansion/test_query_expansion_run.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/retrieval/test_bm25.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/retrieval/test_hybrid_base.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/retrieval/test_hybrid_cc.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/retrieval/test_hybrid_rrf.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/retrieval/test_retrieval_base.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/retrieval/test_run_retrieval_node.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/nodes/retrieval/test_vectordb.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/schema/test_metricinput_schema.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/schema/test_module_schema.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/schema/test_node_schema.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/test_chunker.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/test_cli.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/test_dashboard.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/test_deploy.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/test_parser.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/test_strategy.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/test_support.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/test_validator.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/test_web.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/autorag/utils/test_preprocess.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/conftest.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/delete_tests.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/mock.py +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/requirements.txt +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/README.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/chunk_data/sample_parsed.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/corpus_data_sample.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/data_creation/raw_dir/sample1.txt +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/data_creation/raw_dir/sample2.txt +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/data_creation/raw_dir/sample3.txt +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/full.yaml +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/parse_data/all_files/baseball_1.pdf +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/parse_data/all_files/csv_sample.csv +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/parse_data/clova_data/result_sample.json +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/parse_data/clova_data/result_table.txt +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/parse_data/clova_data/result_text.txt +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/parse_data/csv_data/csv_sample.csv +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/parse_data/eng_text/baseball_1.pdf +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/parse_data/eng_text/baseball_2.pdf +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/parse_data/html_data/html_sample.html +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/parse_data/hybrid_data/nfl_rulebook_both.pdf +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/parse_data/json_data/json_sample.json +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/parse_data/korean_table/only_table/kbo_only_table.pdf +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/parse_data/korean_table/table_text/kbo_table_text.pdf +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/parse_data/korean_text/korean_texts_two_page.pdf +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/parse_data/markdown_data/markdown_sample.md +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/parse_data/xml_data/xml_sample.xml +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/qa_data_sample.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/qa_gen_prompts/prompt1.txt +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/qa_gen_prompts/prompt2.txt +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/qa_gen_prompts/prompt3.txt +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/qa_test_data_sample.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/0/config.yaml +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/0/post_retrieve_node_line/generator/0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/0/post_retrieve_node_line/generator/1.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/0/post_retrieve_node_line/generator/2.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/0/post_retrieve_node_line/generator/3.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/0/post_retrieve_node_line/generator/4.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/0/post_retrieve_node_line/generator/5.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/0/post_retrieve_node_line/generator/best_5.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/0/post_retrieve_node_line/generator/summary.csv +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/0/post_retrieve_node_line/prompt_maker/0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/0/post_retrieve_node_line/prompt_maker/1.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/0/post_retrieve_node_line/prompt_maker/best_0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/0/post_retrieve_node_line/prompt_maker/summary.csv +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/0/post_retrieve_node_line/summary.csv +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/0/pre_retrieve_node_line/query_expansion/0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/0/pre_retrieve_node_line/query_expansion/1.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/0/pre_retrieve_node_line/query_expansion/2.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/0/pre_retrieve_node_line/query_expansion/best_0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/0/pre_retrieve_node_line/query_expansion/summary.csv +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/0/pre_retrieve_node_line/summary.csv +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/0/retrieve_node_line/passage_compressor/0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/0/retrieve_node_line/passage_compressor/best_0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/0/retrieve_node_line/passage_compressor/summary.csv +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/0/retrieve_node_line/passage_reranker/0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/0/retrieve_node_line/passage_reranker/1.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/0/retrieve_node_line/passage_reranker/best_0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/0/retrieve_node_line/passage_reranker/summary.csv +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/0/retrieve_node_line/retrieval/0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/0/retrieve_node_line/retrieval/1.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/0/retrieve_node_line/retrieval/best_0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/0/retrieve_node_line/retrieval/summary.csv +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/0/retrieve_node_line/summary.csv +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/1/config.yaml +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/1/pre_retrieve_node_line/query_expansion/0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/1/pre_retrieve_node_line/query_expansion/1.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/1/pre_retrieve_node_line/query_expansion/2.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/1/pre_retrieve_node_line/query_expansion/best_0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/1/pre_retrieve_node_line/query_expansion/summary.csv +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/1/pre_retrieve_node_line/summary.csv +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/1/retrieve_node_line/passage_filter/0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/1/retrieve_node_line/passage_reranker/0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/1/retrieve_node_line/passage_reranker/1.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/1/retrieve_node_line/passage_reranker/best_0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/1/retrieve_node_line/passage_reranker/summary.csv +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/1/retrieve_node_line/retrieval/0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/1/retrieve_node_line/retrieval/1.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/1/retrieve_node_line/retrieval/best_0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/1/retrieve_node_line/retrieval/summary.csv +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/2/config.yaml +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/2/post_retrieve_node_line/prompt_maker/0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/2/pre_retrieve_node_line/query_expansion/0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/2/pre_retrieve_node_line/query_expansion/1.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/2/pre_retrieve_node_line/query_expansion/2.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/2/pre_retrieve_node_line/query_expansion/best_0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/2/pre_retrieve_node_line/query_expansion/summary.csv +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/2/pre_retrieve_node_line/summary.csv +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/2/retrieve_node_line/passage_compressor/0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/2/retrieve_node_line/passage_compressor/best_0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/2/retrieve_node_line/passage_compressor/summary.csv +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/2/retrieve_node_line/passage_reranker/0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/2/retrieve_node_line/passage_reranker/1.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/2/retrieve_node_line/passage_reranker/best_0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/2/retrieve_node_line/passage_reranker/summary.csv +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/2/retrieve_node_line/retrieval/0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/2/retrieve_node_line/retrieval/1.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/2/retrieve_node_line/retrieval/best_0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/2/retrieve_node_line/retrieval/summary.csv +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/2/retrieve_node_line/summary.csv +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/3/config.yaml +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/best.yaml +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/data/corpus.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/data/qa.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/resources/bm25_porter_stemmer.pkl +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/resources/chroma/6595d304-5270-4d9f-a267-c191366804ce/data_level0.bin +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/resources/chroma/6595d304-5270-4d9f-a267-c191366804ce/header.bin +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/resources/chroma/6595d304-5270-4d9f-a267-c191366804ce/length.bin +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/resources/chroma/6595d304-5270-4d9f-a267-c191366804ce/link_lists.bin +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/resources/chroma/chroma.sqlite3 +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/result_project/trial.json +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/sample_contents_nqa.csv +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/sample_project/data/corpus.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/sample_project/data/qa.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/sample_project/resources/bm25_gpt2.pkl +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/sample_project/resources/bm25_porter_stemmer.pkl +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/simple_chunk.yaml +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/simple_mock.yaml +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/simple_parse.yaml +0 -0
- {autorag-0.2.16 → autorag-0.2.18}/tests/resources/test_bm25_retrieval.pkl +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: AutoRAG
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.18
|
|
4
4
|
Summary: Automatically Evaluate RAG pipelines with your own data. Find optimal structure for new RAG product.
|
|
5
5
|
Author-email: Marker-Inc <vkehfdl1@gmail.com>
|
|
6
6
|
License: Apache License
|
|
@@ -254,14 +254,14 @@ Requires-Dist: llama-index-core>=0.11.0
|
|
|
254
254
|
Requires-Dist: llama-index-readers-file
|
|
255
255
|
Requires-Dist: llama-index-embeddings-openai
|
|
256
256
|
Requires-Dist: llama-index-embeddings-huggingface
|
|
257
|
-
Requires-Dist: llama-index-llms-openai>=0.
|
|
257
|
+
Requires-Dist: llama-index-llms-openai>=0.2.7
|
|
258
258
|
Requires-Dist: llama-index-llms-huggingface
|
|
259
259
|
Requires-Dist: llama-index-llms-openai-like
|
|
260
260
|
Requires-Dist: llama-index-llms-ollama
|
|
261
261
|
Requires-Dist: llama-index-retrievers-bm25
|
|
262
262
|
Requires-Dist: streamlit
|
|
263
263
|
Requires-Dist: gradio
|
|
264
|
-
Requires-Dist: langchain-core>=0.
|
|
264
|
+
Requires-Dist: langchain-core>=0.3.0
|
|
265
265
|
Requires-Dist: langchain_unstructured
|
|
266
266
|
Requires-Dist: langchain-upstage
|
|
267
267
|
Requires-Dist: panel
|
|
@@ -34,6 +34,7 @@ autorag/validator.py
|
|
|
34
34
|
autorag/web.py
|
|
35
35
|
autorag/data/__init__.py
|
|
36
36
|
autorag/data/beta/__init__.py
|
|
37
|
+
autorag/data/beta/extract_evidence.py
|
|
37
38
|
autorag/data/beta/sample.py
|
|
38
39
|
autorag/data/beta/schema.py
|
|
39
40
|
autorag/data/beta/filter/__init__.py
|
|
@@ -200,10 +201,12 @@ docs/source/_static/roadmap/cycle.png
|
|
|
200
201
|
docs/source/_static/roadmap/merger.png
|
|
201
202
|
docs/source/_static/roadmap/node_line_modular.png
|
|
202
203
|
docs/source/_static/roadmap/policy.png
|
|
204
|
+
docs/source/api_spec/autorag.data.beta.filter.rst
|
|
203
205
|
docs/source/api_spec/autorag.data.beta.generation_gt.rst
|
|
204
206
|
docs/source/api_spec/autorag.data.beta.query.rst
|
|
205
207
|
docs/source/api_spec/autorag.data.beta.rst
|
|
206
208
|
docs/source/api_spec/autorag.data.beta.schema.rst
|
|
209
|
+
docs/source/api_spec/autorag.data.chunk.rst
|
|
207
210
|
docs/source/api_spec/autorag.data.corpus.rst
|
|
208
211
|
docs/source/api_spec/autorag.data.parse.rst
|
|
209
212
|
docs/source/api_spec/autorag.data.qacreation.rst
|
|
@@ -229,9 +232,19 @@ docs/source/data_creation/data_format.md
|
|
|
229
232
|
docs/source/data_creation/ragas.md
|
|
230
233
|
docs/source/data_creation/tutorial.md
|
|
231
234
|
docs/source/data_creation/beta/data_creation.md
|
|
232
|
-
docs/source/data_creation/beta/filter.md
|
|
233
|
-
docs/source/data_creation/beta/query_gen.md
|
|
234
235
|
docs/source/data_creation/beta/tutorial.md
|
|
236
|
+
docs/source/data_creation/beta/chunk/chunk.md
|
|
237
|
+
docs/source/data_creation/beta/chunk/langchain_chunk.md
|
|
238
|
+
docs/source/data_creation/beta/chunk/llama_index_chunk.md
|
|
239
|
+
docs/source/data_creation/beta/parse/clova.md
|
|
240
|
+
docs/source/data_creation/beta/parse/langchain_parse.md
|
|
241
|
+
docs/source/data_creation/beta/parse/llama_parse.md
|
|
242
|
+
docs/source/data_creation/beta/parse/parse.md
|
|
243
|
+
docs/source/data_creation/beta/parse/table_hybrid_parse.md
|
|
244
|
+
docs/source/data_creation/beta/qa_creation/answer_gen.md
|
|
245
|
+
docs/source/data_creation/beta/qa_creation/filter.md
|
|
246
|
+
docs/source/data_creation/beta/qa_creation/qa_creation.md
|
|
247
|
+
docs/source/data_creation/beta/qa_creation/query_gen.md
|
|
235
248
|
docs/source/deploy/api_endpoint.md
|
|
236
249
|
docs/source/deploy/web.md
|
|
237
250
|
docs/source/evaluate_metrics/generation.md
|
|
@@ -32,14 +32,14 @@ llama-index-core>=0.11.0
|
|
|
32
32
|
llama-index-readers-file
|
|
33
33
|
llama-index-embeddings-openai
|
|
34
34
|
llama-index-embeddings-huggingface
|
|
35
|
-
llama-index-llms-openai>=0.
|
|
35
|
+
llama-index-llms-openai>=0.2.7
|
|
36
36
|
llama-index-llms-huggingface
|
|
37
37
|
llama-index-llms-openai-like
|
|
38
38
|
llama-index-llms-ollama
|
|
39
39
|
llama-index-retrievers-bm25
|
|
40
40
|
streamlit
|
|
41
41
|
gradio
|
|
42
|
-
langchain-core>=0.
|
|
42
|
+
langchain-core>=0.3.0
|
|
43
43
|
langchain_unstructured
|
|
44
44
|
langchain-upstage
|
|
45
45
|
panel
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: AutoRAG
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.18
|
|
4
4
|
Summary: Automatically Evaluate RAG pipelines with your own data. Find optimal structure for new RAG product.
|
|
5
5
|
Author-email: Marker-Inc <vkehfdl1@gmail.com>
|
|
6
6
|
License: Apache License
|
|
@@ -254,14 +254,14 @@ Requires-Dist: llama-index-core>=0.11.0
|
|
|
254
254
|
Requires-Dist: llama-index-readers-file
|
|
255
255
|
Requires-Dist: llama-index-embeddings-openai
|
|
256
256
|
Requires-Dist: llama-index-embeddings-huggingface
|
|
257
|
-
Requires-Dist: llama-index-llms-openai>=0.
|
|
257
|
+
Requires-Dist: llama-index-llms-openai>=0.2.7
|
|
258
258
|
Requires-Dist: llama-index-llms-huggingface
|
|
259
259
|
Requires-Dist: llama-index-llms-openai-like
|
|
260
260
|
Requires-Dist: llama-index-llms-ollama
|
|
261
261
|
Requires-Dist: llama-index-retrievers-bm25
|
|
262
262
|
Requires-Dist: streamlit
|
|
263
263
|
Requires-Dist: gradio
|
|
264
|
-
Requires-Dist: langchain-core>=0.
|
|
264
|
+
Requires-Dist: langchain-core>=0.3.0
|
|
265
265
|
Requires-Dist: langchain_unstructured
|
|
266
266
|
Requires-Dist: langchain-upstage
|
|
267
267
|
Requires-Dist: panel
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
0.2.18
|
|
@@ -15,7 +15,7 @@ from llama_index.llms.huggingface import HuggingFaceLLM
|
|
|
15
15
|
from llama_index.llms.ollama import Ollama
|
|
16
16
|
from llama_index.llms.openai import OpenAI
|
|
17
17
|
from llama_index.llms.openai_like import OpenAILike
|
|
18
|
-
from
|
|
18
|
+
from langchain_openai.embeddings import OpenAIEmbeddings
|
|
19
19
|
from rich.logging import RichHandler
|
|
20
20
|
|
|
21
21
|
version_path = os.path.join(os.path.dirname(os.path.realpath(__file__)), "VERSION")
|
|
@@ -15,7 +15,7 @@ from langchain_community.document_loaders import (
|
|
|
15
15
|
DirectoryLoader,
|
|
16
16
|
)
|
|
17
17
|
from langchain_unstructured import UnstructuredLoader
|
|
18
|
-
from langchain_upstage import
|
|
18
|
+
from langchain_upstage import UpstageDocumentParseLoader
|
|
19
19
|
|
|
20
20
|
from llama_index.core.node_parser import (
|
|
21
21
|
TokenTextSplitter,
|
|
@@ -42,7 +42,6 @@ parse_modules = {
|
|
|
42
42
|
"pypdf": PyPDFLoader,
|
|
43
43
|
"pymupdf": PyMuPDFLoader,
|
|
44
44
|
"unstructuredpdf": UnstructuredPDFLoader,
|
|
45
|
-
"upstagelayoutanalysis": UpstageLayoutAnalysisLoader,
|
|
46
45
|
# Common File Types
|
|
47
46
|
# 1. CSV
|
|
48
47
|
"csv": CSVLoader,
|
|
@@ -57,6 +56,7 @@ parse_modules = {
|
|
|
57
56
|
# 6. All files
|
|
58
57
|
"directory": DirectoryLoader,
|
|
59
58
|
"unstructured": UnstructuredLoader,
|
|
59
|
+
"upstagedocumentparse": UpstageDocumentParseLoader,
|
|
60
60
|
}
|
|
61
61
|
|
|
62
62
|
chunk_modules = {
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
# This module is about extracting evidence from the given retrieval gt passage
|
|
@@ -1,9 +1,9 @@
|
|
|
1
1
|
import logging
|
|
2
|
-
from typing import Callable, Optional, Dict, Awaitable, Any
|
|
2
|
+
from typing import Callable, Optional, Dict, Awaitable, Any, Tuple, List
|
|
3
3
|
import pandas as pd
|
|
4
|
+
from autorag.utils.util import process_batch, get_event_loop, fetch_contents
|
|
4
5
|
|
|
5
6
|
from autorag.support import get_support_modules
|
|
6
|
-
from autorag.utils.util import process_batch, get_event_loop, fetch_contents
|
|
7
7
|
|
|
8
8
|
logger = logging.getLogger("AutoRAG")
|
|
9
9
|
|
|
@@ -68,6 +68,20 @@ class Corpus:
|
|
|
68
68
|
def linked_raw(self, raw: Raw):
|
|
69
69
|
raise NotImplementedError("linked_raw is read-only.")
|
|
70
70
|
|
|
71
|
+
def to_parquet(self, save_path: str):
|
|
72
|
+
"""
|
|
73
|
+
Save the corpus to the AutoRAG compatible parquet file.
|
|
74
|
+
It is not for the data creation, for running AutoRAG.
|
|
75
|
+
If you want to save it directly, use the below code.
|
|
76
|
+
`corpus.data.to_parquet(save_path)`
|
|
77
|
+
|
|
78
|
+
:param save_path: The path to save the corpus.
|
|
79
|
+
"""
|
|
80
|
+
if not save_path.endswith(".parquet"):
|
|
81
|
+
raise ValueError("save_path must be ended with .parquet")
|
|
82
|
+
save_df = self.data[["doc_id", "contents", "metadata"]].reset_index(drop=True)
|
|
83
|
+
save_df.to_parquet(save_path)
|
|
84
|
+
|
|
71
85
|
def batch_apply(
|
|
72
86
|
self, fn: Callable[[Dict, Any], Awaitable[Dict]], batch_size: int = 32, **kwargs
|
|
73
87
|
) -> "Corpus":
|
|
@@ -152,6 +166,26 @@ class QA:
|
|
|
152
166
|
)
|
|
153
167
|
return self
|
|
154
168
|
|
|
169
|
+
def to_parquet(self, qa_save_path: str, corpus_save_path: str):
|
|
170
|
+
"""
|
|
171
|
+
Save the qa and corpus to the AutoRAG compatible parquet file.
|
|
172
|
+
It is not for the data creation, for running AutoRAG.
|
|
173
|
+
If you want to save it directly, use the below code.
|
|
174
|
+
`qa.data.to_parquet(save_path)`
|
|
175
|
+
|
|
176
|
+
:param qa_save_path: The path to save the qa dataset.
|
|
177
|
+
:param corpus_save_path: The path to save the corpus.
|
|
178
|
+
"""
|
|
179
|
+
if not qa_save_path.endswith(".parquet"):
|
|
180
|
+
raise ValueError("save_path must be ended with .parquet")
|
|
181
|
+
if not corpus_save_path.endswith(".parquet"):
|
|
182
|
+
raise ValueError("save_path must be ended with .parquet")
|
|
183
|
+
save_df = self.data[
|
|
184
|
+
["qid", "query", "retrieval_gt", "generation_gt"]
|
|
185
|
+
].reset_index(drop=True)
|
|
186
|
+
save_df.to_parquet(qa_save_path)
|
|
187
|
+
self.linked_corpus.to_parquet(corpus_save_path)
|
|
188
|
+
|
|
155
189
|
def update_corpus(self, new_corpus: Corpus) -> "QA":
|
|
156
190
|
"""
|
|
157
191
|
Update linked corpus.
|
|
@@ -163,4 +197,105 @@ class QA:
|
|
|
163
197
|
Must have valid `linked_raw` and `raw_id`, `raw_start_idx`, `raw_end_idx` columns.
|
|
164
198
|
:return: The QA instance that updated linked corpus.
|
|
165
199
|
"""
|
|
166
|
-
|
|
200
|
+
self.data["evidence_path"] = (
|
|
201
|
+
self.data["retrieval_gt"]
|
|
202
|
+
.apply(
|
|
203
|
+
lambda x: fetch_contents(
|
|
204
|
+
self.linked_corpus.data,
|
|
205
|
+
x,
|
|
206
|
+
column_name="path",
|
|
207
|
+
)
|
|
208
|
+
)
|
|
209
|
+
.tolist()
|
|
210
|
+
)
|
|
211
|
+
self.data["evidence_page"] = self.data["retrieval_gt"].apply(
|
|
212
|
+
lambda x: list(
|
|
213
|
+
map(
|
|
214
|
+
lambda lst: list(map(lambda x: x.get("page", -1), lst)),
|
|
215
|
+
fetch_contents(self.linked_corpus.data, x, column_name="metadata"),
|
|
216
|
+
)
|
|
217
|
+
)
|
|
218
|
+
)
|
|
219
|
+
if "evidence_start_end_idx" not in self.data.columns:
|
|
220
|
+
# make evidence start_end_idx
|
|
221
|
+
self.data["evidence_start_end_idx"] = (
|
|
222
|
+
self.data["retrieval_gt"]
|
|
223
|
+
.apply(
|
|
224
|
+
lambda x: fetch_contents(
|
|
225
|
+
self.linked_corpus.data,
|
|
226
|
+
x,
|
|
227
|
+
column_name="start_end_idx",
|
|
228
|
+
)
|
|
229
|
+
)
|
|
230
|
+
.tolist()
|
|
231
|
+
)
|
|
232
|
+
|
|
233
|
+
# matching the new corpus with the old corpus
|
|
234
|
+
path_corpus_dict = QA.__make_path_corpus_dict(new_corpus.data)
|
|
235
|
+
new_retrieval_gt = self.data.apply(
|
|
236
|
+
lambda row: QA.__match_index_row(
|
|
237
|
+
row["evidence_start_end_idx"],
|
|
238
|
+
row["evidence_path"],
|
|
239
|
+
row["evidence_page"],
|
|
240
|
+
path_corpus_dict,
|
|
241
|
+
),
|
|
242
|
+
axis=1,
|
|
243
|
+
).tolist()
|
|
244
|
+
new_qa = self.data.copy(deep=True)[["qid", "query", "generation_gt"]]
|
|
245
|
+
new_qa["retrieval_gt"] = new_retrieval_gt
|
|
246
|
+
return QA(new_qa, new_corpus)
|
|
247
|
+
|
|
248
|
+
@staticmethod
|
|
249
|
+
def __match_index(target_idx: Tuple[int, int], dst_idx: Tuple[int, int]) -> bool:
|
|
250
|
+
"""
|
|
251
|
+
Check if the target_idx is overlap by the dst_idx.
|
|
252
|
+
"""
|
|
253
|
+
target_start, target_end = target_idx
|
|
254
|
+
dst_start, dst_end = dst_idx
|
|
255
|
+
return (
|
|
256
|
+
dst_start <= target_start <= dst_end or dst_start <= target_end <= dst_end
|
|
257
|
+
)
|
|
258
|
+
|
|
259
|
+
@staticmethod
|
|
260
|
+
def __match_index_row(
|
|
261
|
+
evidence_indices: List[List[Tuple[int, int]]],
|
|
262
|
+
evidence_paths: List[List[str]],
|
|
263
|
+
evidence_pages: List[List[int]],
|
|
264
|
+
path_corpus_dict: Dict,
|
|
265
|
+
) -> List[List[str]]:
|
|
266
|
+
"""
|
|
267
|
+
Find the matched passage from new_corpus.
|
|
268
|
+
|
|
269
|
+
:param evidence_indices: The evidence indices at the corresponding Raw.
|
|
270
|
+
Its shape is the same as the retrieval_gt.
|
|
271
|
+
:param evidence_paths: The evidence paths at the corresponding Raw.
|
|
272
|
+
Its shape is the same as the retrieval_gt.
|
|
273
|
+
:param path_corpus_dict: The key is the path name, and the value is the corpus dataframe that only contains the path in the key.
|
|
274
|
+
You can make it using `QA.__make_path_corpus_dict`.
|
|
275
|
+
:return:
|
|
276
|
+
"""
|
|
277
|
+
result = []
|
|
278
|
+
for i, idx_list in enumerate(evidence_indices):
|
|
279
|
+
sub_result = []
|
|
280
|
+
for j, idx in enumerate(idx_list):
|
|
281
|
+
path_corpus_df = path_corpus_dict[evidence_paths[i][j]]
|
|
282
|
+
if evidence_pages[i][j] >= 0:
|
|
283
|
+
path_corpus_df = path_corpus_df.loc[
|
|
284
|
+
path_corpus_df["metadata"].apply(lambda x: x.get("page", -1))
|
|
285
|
+
== evidence_pages[i][j]
|
|
286
|
+
]
|
|
287
|
+
matched_corpus = path_corpus_df.loc[
|
|
288
|
+
path_corpus_df["start_end_idx"].apply(
|
|
289
|
+
lambda x: QA.__match_index(idx, x)
|
|
290
|
+
)
|
|
291
|
+
]
|
|
292
|
+
sub_result.extend(matched_corpus["doc_id"].tolist())
|
|
293
|
+
result.append(sub_result)
|
|
294
|
+
return result
|
|
295
|
+
|
|
296
|
+
@staticmethod
|
|
297
|
+
def __make_path_corpus_dict(corpus_df: pd.DataFrame) -> Dict[str, pd.DataFrame]:
|
|
298
|
+
return {
|
|
299
|
+
path: corpus_df[corpus_df["path"] == path]
|
|
300
|
+
for path in corpus_df["path"].unique()
|
|
301
|
+
}
|
|
@@ -101,14 +101,14 @@ def __get_chunk_instance(module_type: str, chunk_method: str, **kwargs):
|
|
|
101
101
|
def add_file_name(
|
|
102
102
|
file_name_language: str, file_names: List[str], chunk_texts: List[str]
|
|
103
103
|
) -> List[str]:
|
|
104
|
-
if file_name_language == "
|
|
104
|
+
if file_name_language == "en":
|
|
105
105
|
return list(
|
|
106
106
|
map(
|
|
107
107
|
lambda x: f"file_name: {x[1]}\n contents: {x[0]}",
|
|
108
108
|
zip(chunk_texts, file_names),
|
|
109
109
|
)
|
|
110
110
|
)
|
|
111
|
-
elif file_name_language == "
|
|
111
|
+
elif file_name_language == "ko":
|
|
112
112
|
return list(
|
|
113
113
|
map(
|
|
114
114
|
lambda x: f"파일 제목: {x[1]}\n 내용: {x[0]}",
|
|
@@ -117,5 +117,5 @@ def add_file_name(
|
|
|
117
117
|
)
|
|
118
118
|
else:
|
|
119
119
|
raise ValueError(
|
|
120
|
-
f"Unsupported file_name_language: {file_name_language}. Choose from '
|
|
120
|
+
f"Unsupported file_name_language: {file_name_language}. Choose from 'en' or 'ko'."
|
|
121
121
|
)
|
|
@@ -16,9 +16,23 @@ def clova_ocr(
|
|
|
16
16
|
data_path_list: List[str],
|
|
17
17
|
url: Optional[str] = None,
|
|
18
18
|
api_key: Optional[str] = None,
|
|
19
|
-
batch: int =
|
|
19
|
+
batch: int = 5,
|
|
20
20
|
table_detection: bool = False,
|
|
21
21
|
) -> Tuple[List[str], List[str], List[int]]:
|
|
22
|
+
"""
|
|
23
|
+
Parse documents to use Naver Clova OCR.
|
|
24
|
+
|
|
25
|
+
:param data_path_list: The list of data paths to parse.
|
|
26
|
+
:param url: The URL for Clova OCR.
|
|
27
|
+
You can get the URL with the guide at https://guide.ncloud-docs.com/docs/clovaocr-example01
|
|
28
|
+
You can set the environment variable CLOVA_URL, or you can set it directly as a parameter.
|
|
29
|
+
:param api_key: The API key for Clova OCR.
|
|
30
|
+
You can get the API key with the guide at https://guide.ncloud-docs.com/docs/clovaocr-example01
|
|
31
|
+
You can set the environment variable CLOVA_API_KEY, or you can set it directly as a parameter.
|
|
32
|
+
:param batch: The batch size for parse documents. Default is 8.
|
|
33
|
+
:param table_detection: Whether to enable table detection. Default is False.
|
|
34
|
+
:return: tuple of lists containing the parsed texts, path and pages.
|
|
35
|
+
"""
|
|
22
36
|
url = os.getenv("CLOVA_URL", None) if url is None else url
|
|
23
37
|
if url is None:
|
|
24
38
|
raise KeyError(
|
|
@@ -32,6 +46,8 @@ def clova_ocr(
|
|
|
32
46
|
"Please set the API key for Clova OCR in the environment variable CLOVA_API_KEY "
|
|
33
47
|
"or directly set it on the config YAML file."
|
|
34
48
|
)
|
|
49
|
+
if batch > 5:
|
|
50
|
+
raise ValueError("The batch size should be less than or equal to 5.")
|
|
35
51
|
|
|
36
52
|
image_data_lst = list(
|
|
37
53
|
map(lambda data_path: pdf_to_images(data_path), data_path_list)
|
|
@@ -9,6 +9,14 @@ from autorag.data.parse.base import parser_node
|
|
|
9
9
|
def langchain_parse(
|
|
10
10
|
data_path_list: List[str], parse_method: str, **kwargs
|
|
11
11
|
) -> Tuple[List[str], List[str], List[int]]:
|
|
12
|
+
"""
|
|
13
|
+
Parse documents to use langchain document_loaders(parse) method
|
|
14
|
+
|
|
15
|
+
:param data_path_list: The list of data paths to parse.
|
|
16
|
+
:param parse_method: A langchain document_loaders(parse) method to use.
|
|
17
|
+
:param kwargs: The extra parameters for creating the langchain document_loaders(parse) instance.
|
|
18
|
+
:return: tuple of lists containing the parsed texts, path and pages.
|
|
19
|
+
"""
|
|
12
20
|
if parse_method in ["directory", "unstructured"]:
|
|
13
21
|
results = parse_all_files(data_path_list, parse_method, **kwargs)
|
|
14
22
|
texts, path = results[0], results[1]
|
|
@@ -10,6 +10,16 @@ from autorag.utils.util import process_batch, get_event_loop
|
|
|
10
10
|
def llama_parse(
|
|
11
11
|
data_path_list: List[str], batch: int = 8, **kwargs
|
|
12
12
|
) -> Tuple[List[str], List[str], List[int]]:
|
|
13
|
+
"""
|
|
14
|
+
Parse documents to use llama_parse.
|
|
15
|
+
LLAMA_CLOUD_API_KEY environment variable should be set.
|
|
16
|
+
You can get the key from https://cloud.llamaindex.ai/api-key
|
|
17
|
+
|
|
18
|
+
:param data_path_list: The list of data paths to parse.
|
|
19
|
+
:param batch: The batch size for parse documents. Default is 8.
|
|
20
|
+
:param kwargs: The extra parameters for creating the llama_parse instance.
|
|
21
|
+
:return: tuple of lists containing the parsed texts, path and pages.
|
|
22
|
+
"""
|
|
13
23
|
parse_instance = LlamaParse(**kwargs)
|
|
14
24
|
|
|
15
25
|
tasks = [
|
|
@@ -18,6 +18,18 @@ def table_hybrid_parse(
|
|
|
18
18
|
table_parse_module: str,
|
|
19
19
|
table_params: Dict,
|
|
20
20
|
) -> Tuple[List[str], List[str], List[int]]:
|
|
21
|
+
"""
|
|
22
|
+
Parse documents to use table_hybrid_parse method.
|
|
23
|
+
The table_hybrid_parse method is a hybrid method that combines the parsing results of PDFs with and without tables.
|
|
24
|
+
It splits the PDF file into pages, separates pages with and without tables, and then parses and merges the results.
|
|
25
|
+
|
|
26
|
+
:param data_path_list: The list of data paths to parse.
|
|
27
|
+
:param text_parse_module: The text parsing module to use. The type should be a string.
|
|
28
|
+
:param text_params: The extra parameters for the text parsing module. The type should be a dictionary.
|
|
29
|
+
:param table_parse_module: The table parsing module to use. The type should be a string.
|
|
30
|
+
:param table_params: The extra parameters for the table parsing module. The type should be a dictionary.
|
|
31
|
+
:return: tuple of lists containing the parsed texts, path and pages.
|
|
32
|
+
"""
|
|
21
33
|
# make save folder directory
|
|
22
34
|
with tempfile.TemporaryDirectory() as save_dir:
|
|
23
35
|
text_dir = os.path.join(save_dir, "text")
|
|
@@ -18,7 +18,11 @@ from autorag.nodes.retrieval.base import get_bm25_pkl_name
|
|
|
18
18
|
from autorag.nodes.retrieval.bm25 import bm25_ingest
|
|
19
19
|
from autorag.nodes.retrieval.vectordb import vectordb_ingest
|
|
20
20
|
from autorag.schema import Node
|
|
21
|
-
from autorag.schema.node import
|
|
21
|
+
from autorag.schema.node import (
|
|
22
|
+
module_type_exists,
|
|
23
|
+
extract_values_from_nodes,
|
|
24
|
+
extract_values_from_nodes_strategy,
|
|
25
|
+
)
|
|
22
26
|
from autorag.utils import (
|
|
23
27
|
cast_qa_dataset,
|
|
24
28
|
cast_corpus_dataset,
|
|
@@ -82,7 +86,7 @@ class Evaluator:
|
|
|
82
86
|
|
|
83
87
|
validate_qa_from_corpus_dataset(self.qa_data, self.corpus_data)
|
|
84
88
|
|
|
85
|
-
# copy dataset to project directory
|
|
89
|
+
# copy dataset to the project directory
|
|
86
90
|
if not os.path.exists(os.path.join(self.project_dir, "data")):
|
|
87
91
|
os.makedirs(os.path.join(self.project_dir, "data"))
|
|
88
92
|
qa_path_in_project = os.path.join(self.project_dir, "data", "qa.parquet")
|
|
@@ -100,7 +104,7 @@ class Evaluator:
|
|
|
100
104
|
trial_name = self.__get_new_trial_name()
|
|
101
105
|
self.__make_trial_dir(trial_name)
|
|
102
106
|
|
|
103
|
-
# copy
|
|
107
|
+
# copy YAML file to the trial directory
|
|
104
108
|
shutil.copy(
|
|
105
109
|
yaml_path, os.path.join(self.project_dir, trial_name, "config.yaml")
|
|
106
110
|
)
|
|
@@ -148,13 +152,12 @@ class Evaluator:
|
|
|
148
152
|
bm25_tokenizer_list = list(
|
|
149
153
|
chain.from_iterable(
|
|
150
154
|
map(
|
|
151
|
-
lambda nodes:
|
|
152
|
-
nodes, "bm25_tokenizer"
|
|
153
|
-
),
|
|
155
|
+
lambda nodes: self._find_bm25_tokenizer(nodes),
|
|
154
156
|
node_lines.values(),
|
|
155
157
|
)
|
|
156
158
|
)
|
|
157
159
|
)
|
|
160
|
+
|
|
158
161
|
if len(bm25_tokenizer_list) == 0:
|
|
159
162
|
bm25_tokenizer_list = ["porter_stemmer"]
|
|
160
163
|
for bm25_tokenizer in bm25_tokenizer_list:
|
|
@@ -166,6 +169,7 @@ class Evaluator:
|
|
|
166
169
|
# ingest because bm25 supports update new corpus data
|
|
167
170
|
bm25_ingest(bm25_dir, self.corpus_data, bm25_tokenizer=bm25_tokenizer)
|
|
168
171
|
logger.info("BM25 corpus embedding complete.")
|
|
172
|
+
|
|
169
173
|
if any(
|
|
170
174
|
list(
|
|
171
175
|
map(
|
|
@@ -178,9 +182,7 @@ class Evaluator:
|
|
|
178
182
|
embedding_models_list = list(
|
|
179
183
|
chain.from_iterable(
|
|
180
184
|
map(
|
|
181
|
-
lambda nodes:
|
|
182
|
-
nodes, "embedding_model"
|
|
183
|
-
),
|
|
185
|
+
lambda nodes: self._find_embedding_model(nodes),
|
|
184
186
|
node_lines.values(),
|
|
185
187
|
)
|
|
186
188
|
)
|
|
@@ -517,3 +519,31 @@ class Evaluator:
|
|
|
517
519
|
}
|
|
518
520
|
)
|
|
519
521
|
return summary_lst
|
|
522
|
+
|
|
523
|
+
@staticmethod
|
|
524
|
+
def _find_bm25_tokenizer(nodes: List[Node]):
|
|
525
|
+
bm25_tokenizer_list = extract_values_from_nodes(nodes, "bm25_tokenizer")
|
|
526
|
+
strategy_tokenizer_list = list(
|
|
527
|
+
chain.from_iterable(
|
|
528
|
+
extract_values_from_nodes_strategy(nodes, "bm25_tokenizer")
|
|
529
|
+
)
|
|
530
|
+
)
|
|
531
|
+
return list(set(bm25_tokenizer_list + strategy_tokenizer_list))
|
|
532
|
+
|
|
533
|
+
@staticmethod
|
|
534
|
+
def _find_embedding_model(nodes: List[Node]):
|
|
535
|
+
embedding_models_list = extract_values_from_nodes(nodes, "embedding_model")
|
|
536
|
+
retrieval_module_dicts = extract_values_from_nodes_strategy(
|
|
537
|
+
nodes, "retrieval_modules"
|
|
538
|
+
)
|
|
539
|
+
for retrieval_modules in retrieval_module_dicts:
|
|
540
|
+
vectordb_modules = list(
|
|
541
|
+
filter(lambda x: x["module_type"] == "vectordb", retrieval_modules)
|
|
542
|
+
)
|
|
543
|
+
embedding_models_list.extend(
|
|
544
|
+
list(map(lambda x: x.get("embedding_model", None), vectordb_modules))
|
|
545
|
+
)
|
|
546
|
+
embedding_models_list = list(
|
|
547
|
+
filter(lambda x: x is not None, embedding_models_list)
|
|
548
|
+
)
|
|
549
|
+
return list(set(embedding_models_list))
|
|
@@ -12,10 +12,16 @@ from autorag.utils.util import get_event_loop, process_batch
|
|
|
12
12
|
logger = logging.getLogger("AutoRAG")
|
|
13
13
|
|
|
14
14
|
MAX_TOKEN_DICT = { # model name : token limit
|
|
15
|
+
"o1-preview": 128_000,
|
|
16
|
+
"o1-preview-2024-09-12": 128_000,
|
|
17
|
+
"o1-mini": 128_000,
|
|
18
|
+
"o1-mini-2024-09-12": 128_000,
|
|
15
19
|
"gpt-4o-mini": 128_000,
|
|
16
20
|
"gpt-4o-mini-2024-07-18": 128_000,
|
|
17
21
|
"gpt-4o": 128_000,
|
|
22
|
+
"gpt-4o-2024-08-06": 128_000,
|
|
18
23
|
"gpt-4o-2024-05-13": 128_000,
|
|
24
|
+
"chatgpt-4o-latest": 128_000,
|
|
19
25
|
"gpt-4-turbo": 128_000,
|
|
20
26
|
"gpt-4-turbo-2024-04-09": 128_000,
|
|
21
27
|
"gpt-4-turbo-preview": 128_000,
|
|
@@ -82,7 +88,11 @@ def openai_llm(
|
|
|
82
88
|
kwargs.pop("n")
|
|
83
89
|
logger.warning("parameter n does not effective. It always set to 1.")
|
|
84
90
|
|
|
85
|
-
|
|
91
|
+
# TODO: fix this after updating tiktoken for the o1 model. It is not yet supported yet.
|
|
92
|
+
if llm.startswith("o1"):
|
|
93
|
+
tokenizer = tiktoken.get_encoding("o200k_base")
|
|
94
|
+
else:
|
|
95
|
+
tokenizer = tiktoken.encoding_for_model(llm)
|
|
86
96
|
if truncate:
|
|
87
97
|
max_token_size = MAX_TOKEN_DICT.get(llm) - 7 # because of chat token usage
|
|
88
98
|
if max_token_size is None:
|
|
@@ -99,7 +109,15 @@ def openai_llm(
|
|
|
99
109
|
|
|
100
110
|
client = AsyncOpenAI(api_key=api_key)
|
|
101
111
|
loop = get_event_loop()
|
|
102
|
-
|
|
112
|
+
if llm.startswith("o1"):
|
|
113
|
+
tasks = [
|
|
114
|
+
get_result_o1(prompt, client, llm, tokenizer, **kwargs)
|
|
115
|
+
for prompt in prompts
|
|
116
|
+
]
|
|
117
|
+
else:
|
|
118
|
+
tasks = [
|
|
119
|
+
get_result(prompt, client, llm, tokenizer, **kwargs) for prompt in prompts
|
|
120
|
+
]
|
|
103
121
|
result = loop.run_until_complete(process_batch(tasks, batch))
|
|
104
122
|
answer_result = list(map(lambda x: x[0], result))
|
|
105
123
|
token_result = list(map(lambda x: x[1], result))
|
|
@@ -132,6 +150,31 @@ async def get_result(
|
|
|
132
150
|
return answer, tokens, logprobs
|
|
133
151
|
|
|
134
152
|
|
|
153
|
+
async def get_result_o1(
|
|
154
|
+
prompt: str, client: AsyncOpenAI, model: str, tokenizer: Encoding, **kwargs
|
|
155
|
+
):
|
|
156
|
+
assert model.startswith("o1"), "This function only supports o1 model."
|
|
157
|
+
# The default temperature for the o1 model is 1. 1 is only supported.
|
|
158
|
+
# See https://platform.openai.com/docs/guides/reasoning about beta limitation of o1 models.
|
|
159
|
+
kwargs["temperature"] = 1
|
|
160
|
+
kwargs["top_p"] = 1
|
|
161
|
+
kwargs["presence_penalty"] = 0
|
|
162
|
+
kwargs["frequency_penalty"] = 0
|
|
163
|
+
response = await client.chat.completions.create(
|
|
164
|
+
model=model,
|
|
165
|
+
messages=[
|
|
166
|
+
{"role": "user", "content": prompt},
|
|
167
|
+
],
|
|
168
|
+
logprobs=False,
|
|
169
|
+
n=1,
|
|
170
|
+
**kwargs,
|
|
171
|
+
)
|
|
172
|
+
answer = response.choices[0].message.content
|
|
173
|
+
tokens = tokenizer.encode(answer, allowed_special="all")
|
|
174
|
+
pseudo_log_probs = [0.5] * len(tokens)
|
|
175
|
+
return answer, tokens, pseudo_log_probs
|
|
176
|
+
|
|
177
|
+
|
|
135
178
|
def truncate_by_token(prompt: str, tokenizer: Encoding, max_token_size: int):
|
|
136
179
|
tokens = tokenizer.encode(prompt, allowed_special="all")
|
|
137
180
|
return tokenizer.decode(tokens[:max_token_size])
|
|
@@ -105,14 +105,15 @@ async def vectordb_pure(
|
|
|
105
105
|
:param query_embeddings: A list of query embeddings.
|
|
106
106
|
:param top_k: The number of passages to be retrieved.
|
|
107
107
|
:param collection: A chroma collection instance that will be used to retrieve passages.
|
|
108
|
-
|
|
109
|
-
:return: The tuple contains a list of passage ids that retrieved from vectordb and a list of its scores.
|
|
108
|
+
:return: The tuple contains a list of passage ids that are retrieved from vectordb and a list of its scores.
|
|
110
109
|
"""
|
|
111
110
|
id_result, score_result = [], []
|
|
112
111
|
for embedded_query in query_embeddings:
|
|
113
112
|
result = collection.query(query_embeddings=embedded_query, n_results=top_k)
|
|
114
113
|
id_result.extend(result["ids"])
|
|
115
|
-
score_result.extend(
|
|
114
|
+
score_result.extend(
|
|
115
|
+
list(map(lambda lst: list(map(lambda x: 1 - x, lst)), result["distances"]))
|
|
116
|
+
)
|
|
116
117
|
|
|
117
118
|
# Distribute passages evenly
|
|
118
119
|
id_result, score_result = evenly_distribute_passages(id_result, score_result, top_k)
|
|
@@ -2,13 +2,13 @@ import itertools
|
|
|
2
2
|
import logging
|
|
3
3
|
from copy import deepcopy
|
|
4
4
|
from dataclasses import dataclass, field
|
|
5
|
-
from typing import Dict, List, Callable, Tuple
|
|
5
|
+
from typing import Dict, List, Callable, Tuple, Any
|
|
6
6
|
|
|
7
7
|
import pandas as pd
|
|
8
8
|
|
|
9
9
|
from autorag.schema.module import Module
|
|
10
10
|
from autorag.support import get_support_nodes
|
|
11
|
-
from autorag.utils.util import make_combinations, explode
|
|
11
|
+
from autorag.utils.util import make_combinations, explode, find_key_values
|
|
12
12
|
|
|
13
13
|
logger = logging.getLogger("AutoRAG")
|
|
14
14
|
|
|
@@ -101,6 +101,23 @@ def extract_values_from_nodes(nodes: List[Node], key: str) -> List[str]:
|
|
|
101
101
|
return list(set(list(itertools.chain.from_iterable(values))))
|
|
102
102
|
|
|
103
103
|
|
|
104
|
+
def extract_values_from_nodes_strategy(nodes: List[Node], key: str) -> List[Any]:
|
|
105
|
+
"""
|
|
106
|
+
This function extract values from nodes' strategy.
|
|
107
|
+
|
|
108
|
+
:param nodes: The nodes you want to extract values from.
|
|
109
|
+
:param key: The key string that you want to extract.
|
|
110
|
+
:return: The list of extracted values.
|
|
111
|
+
It removes duplicated elements automatically.
|
|
112
|
+
"""
|
|
113
|
+
values = []
|
|
114
|
+
for node in nodes:
|
|
115
|
+
value_list = find_key_values(node.strategy, key)
|
|
116
|
+
if value_list:
|
|
117
|
+
values.extend(value_list)
|
|
118
|
+
return values
|
|
119
|
+
|
|
120
|
+
|
|
104
121
|
def module_type_exists(nodes: List[Node], module_type: str) -> bool:
|
|
105
122
|
"""
|
|
106
123
|
This function check if the module type exists in the nodes.
|