AutoRAG 0.2.16__tar.gz → 0.2.17__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {autorag-0.2.16 → autorag-0.2.17}/AutoRAG.egg-info/PKG-INFO +1 -1
- {autorag-0.2.16 → autorag-0.2.17}/AutoRAG.egg-info/SOURCES.txt +13 -2
- {autorag-0.2.16 → autorag-0.2.17}/PKG-INFO +1 -1
- autorag-0.2.17/autorag/VERSION +1 -0
- autorag-0.2.17/autorag/data/beta/extract_evidence.py +1 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/data/beta/schema.py +138 -3
- {autorag-0.2.16 → autorag-0.2.17}/autorag/data/parse/clova.py +14 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/data/parse/langchain_parse.py +8 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/data/parse/llamaparse.py +10 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/data/parse/table_hybrid_parse.py +12 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/retrieval/vectordb.py +4 -3
- autorag-0.2.17/docs/source/_static/data_creation.png +0 -0
- autorag-0.2.17/docs/source/data_creation/beta/chunk/chunk.md +168 -0
- autorag-0.2.17/docs/source/data_creation/beta/chunk/langchain_chunk.md +47 -0
- autorag-0.2.17/docs/source/data_creation/beta/chunk/llama_index_chunk.md +57 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/data_creation/beta/data_creation.md +5 -16
- autorag-0.2.17/docs/source/data_creation/beta/parse/clova.md +30 -0
- autorag-0.2.17/docs/source/data_creation/beta/parse/langchain_parse.md +123 -0
- autorag-0.2.17/docs/source/data_creation/beta/parse/llama_parse.md +29 -0
- autorag-0.2.17/docs/source/data_creation/beta/parse/parse.md +108 -0
- autorag-0.2.17/docs/source/data_creation/beta/parse/table_hybrid_parse.md +52 -0
- autorag-0.2.17/docs/source/data_creation/beta/qa_creation/answer_gen.md +66 -0
- {autorag-0.2.16/docs/source/data_creation/beta → autorag-0.2.17/docs/source/data_creation/beta/qa_creation}/filter.md +5 -0
- autorag-0.2.17/docs/source/data_creation/beta/qa_creation/qa_creation.md +165 -0
- {autorag-0.2.16/docs/source/data_creation/beta → autorag-0.2.17/docs/source/data_creation/beta/qa_creation}/query_gen.md +2 -30
- autorag-0.2.17/docs/source/data_creation/beta/tutorial.md +152 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/tutorial.md +1 -1
- {autorag-0.2.16 → autorag-0.2.17}/sample_config/parse/parse_full.yaml +7 -0
- autorag-0.2.17/tests/autorag/data/beta/test_schema.py +186 -0
- autorag-0.2.16/autorag/VERSION +0 -1
- autorag-0.2.16/docs/source/_static/data_creation.png +0 -0
- autorag-0.2.16/docs/source/data_creation/beta/tutorial.md +0 -37
- autorag-0.2.16/tests/autorag/data/beta/test_schema.py +0 -61
- {autorag-0.2.16 → autorag-0.2.17}/.github/FUNDING.yml +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/.github/ISSUE_TEMPLATE/bug_report.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/.github/ISSUE_TEMPLATE/feature_request.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/.github/dependabot.yml +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/.github/workflows/publish.yml +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/.github/workflows/sphinx.yml +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/.github/workflows/test.yml +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/.gitignore +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/.pre-commit-config.yaml +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/AutoRAG.egg-info/dependency_links.txt +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/AutoRAG.egg-info/entry_points.txt +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/AutoRAG.egg-info/requires.txt +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/AutoRAG.egg-info/top_level.txt +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/CODE_OF_CONDUCT.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/CONTRIBUTING.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/LICENSE +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/README.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/__init__.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/chunker.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/cli.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/dashboard.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/data/__init__.py +1 -1
- {autorag-0.2.16 → autorag-0.2.17}/autorag/data/beta/__init__.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/data/beta/filter/__init__.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/data/beta/filter/dontknow.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/data/beta/filter/prompt.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/data/beta/generation_gt/__init__.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/data/beta/generation_gt/base.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/data/beta/generation_gt/llama_index_gen_gt.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/data/beta/generation_gt/openai_gen_gt.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/data/beta/generation_gt/prompt.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/data/beta/query/__init__.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/data/beta/query/llama_gen_query.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/data/beta/query/openai_gen_query.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/data/beta/query/prompt.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/data/beta/sample.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/data/chunk/__init__.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/data/chunk/base.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/data/chunk/langchain_chunk.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/data/chunk/llama_index_chunk.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/data/chunk/run.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/data/corpus/__init__.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/data/corpus/langchain.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/data/corpus/llama_index.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/data/parse/__init__.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/data/parse/base.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/data/parse/run.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/data/qacreation/__init__.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/data/qacreation/base.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/data/qacreation/llama_index.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/data/qacreation/llama_index_default_prompt.txt +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/data/qacreation/ragas.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/data/qacreation/simple.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/data/utils/__init__.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/data/utils/util.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/deploy.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/evaluation/__init__.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/evaluation/generation.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/evaluation/metric/__init__.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/evaluation/metric/g_eval_prompts/coh_detailed.txt +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/evaluation/metric/g_eval_prompts/con_detailed.txt +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/evaluation/metric/g_eval_prompts/flu_detailed.txt +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/evaluation/metric/g_eval_prompts/rel_detailed.txt +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/evaluation/metric/generation.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/evaluation/metric/retrieval.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/evaluation/metric/retrieval_contents.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/evaluation/metric/util.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/evaluation/retrieval.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/evaluation/retrieval_contents.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/evaluation/util.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/evaluator.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/node_line.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/__init__.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/generator/__init__.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/generator/base.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/generator/llama_index_llm.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/generator/openai_llm.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/generator/run.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/generator/vllm.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/passageaugmenter/__init__.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/passageaugmenter/base.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/passageaugmenter/pass_passage_augmenter.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/passageaugmenter/prev_next_augmenter.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/passageaugmenter/run.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/passagecompressor/__init__.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/passagecompressor/base.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/passagecompressor/longllmlingua.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/passagecompressor/pass_compressor.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/passagecompressor/refine.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/passagecompressor/run.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/passagecompressor/tree_summarize.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/passagefilter/__init__.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/passagefilter/base.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/passagefilter/pass_passage_filter.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/passagefilter/percentile_cutoff.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/passagefilter/recency.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/passagefilter/run.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/passagefilter/similarity_percentile_cutoff.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/passagefilter/similarity_threshold_cutoff.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/passagefilter/threshold_cutoff.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/passagereranker/__init__.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/passagereranker/base.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/passagereranker/cohere.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/passagereranker/colbert.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/passagereranker/flag_embedding.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/passagereranker/flag_embedding_llm.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/passagereranker/jina.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/passagereranker/koreranker.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/passagereranker/monot5.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/passagereranker/pass_reranker.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/passagereranker/rankgpt.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/passagereranker/run.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/passagereranker/sentence_transformer.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/passagereranker/tart/__init__.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/passagereranker/tart/modeling_enc_t5.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/passagereranker/tart/tart.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/passagereranker/tart/tokenization_enc_t5.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/passagereranker/time_reranker.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/passagereranker/upr.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/promptmaker/__init__.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/promptmaker/base.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/promptmaker/fstring.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/promptmaker/long_context_reorder.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/promptmaker/run.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/promptmaker/window_replacement.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/queryexpansion/__init__.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/queryexpansion/base.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/queryexpansion/hyde.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/queryexpansion/multi_query_expansion.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/queryexpansion/pass_query_expansion.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/queryexpansion/query_decompose.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/queryexpansion/run.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/retrieval/__init__.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/retrieval/base.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/retrieval/bm25.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/retrieval/hybrid_cc.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/retrieval/hybrid_rrf.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/nodes/retrieval/run.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/parser.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/schema/__init__.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/schema/metricinput.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/schema/module.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/schema/node.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/strategy.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/support.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/utils/__init__.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/utils/preprocess.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/utils/util.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/validator.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/autorag/web.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/Makefile +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/make.bat +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/requirements.txt +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/CNAME +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/_static/data_folder.png +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/_static/dcg.png +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/_static/f1_score.png +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/_static/map.png +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/_static/mrr.png +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/_static/ndcg.png +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/_static/ndcg_formula.png +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/_static/node_folder.png +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/_static/node_line_folder.png +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/_static/node_line_summary.png +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/_static/node_lines.png +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/_static/node_summary.png +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/_static/normal_distribution.png +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/_static/project_folder_example.png +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/_static/project_folders.png +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/_static/resources_folder.png +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/_static/roadmap/RAG_paradigms.png +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/_static/roadmap/advanced_RAG.png +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/_static/roadmap/cycle.png +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/_static/roadmap/merger.png +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/_static/roadmap/node_line_modular.png +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/_static/roadmap/policy.png +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/_static/samsung_sundae.jpeg +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/_static/score_fusion.png +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/_static/trial_folder.png +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/_static/trial_json.png +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/_static/trial_summary.png +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/_static/web_interface.png +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/_static/web_interface_gradio.png +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/api_spec/autorag.data.beta.generation_gt.rst +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/api_spec/autorag.data.beta.query.rst +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/api_spec/autorag.data.beta.rst +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/api_spec/autorag.data.beta.schema.rst +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/api_spec/autorag.data.corpus.rst +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/api_spec/autorag.data.parse.rst +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/api_spec/autorag.data.qacreation.rst +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/api_spec/autorag.data.rst +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/api_spec/autorag.data.utils.rst +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/api_spec/autorag.evaluation.metric.rst +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/api_spec/autorag.evaluation.rst +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/api_spec/autorag.nodes.generator.rst +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/api_spec/autorag.nodes.passageaugmenter.rst +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/api_spec/autorag.nodes.passagecompressor.rst +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/api_spec/autorag.nodes.passagefilter.rst +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/api_spec/autorag.nodes.passagereranker.rst +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/api_spec/autorag.nodes.passagereranker.tart.rst +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/api_spec/autorag.nodes.promptmaker.rst +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/api_spec/autorag.nodes.queryexpansion.rst +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/api_spec/autorag.nodes.retrieval.rst +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/api_spec/autorag.nodes.rst +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/api_spec/autorag.rst +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/api_spec/autorag.schema.rst +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/api_spec/autorag.utils.rst +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/api_spec/modules.rst +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/conf.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/data_creation/data_format.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/data_creation/ragas.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/data_creation/tutorial.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/deploy/api_endpoint.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/deploy/web.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/evaluate_metrics/generation.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/evaluate_metrics/retrieval.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/evaluate_metrics/retrieval_contents.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/index.rst +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/install.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/local_model.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/nodes/generator/generator.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/nodes/generator/llama_index_llm.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/nodes/generator/openai_llm.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/nodes/generator/vllm.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/nodes/index.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/nodes/passage_augmenter/passage_augmenter.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/nodes/passage_augmenter/prev_next_augmenter.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/nodes/passage_compressor/longllmlingua.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/nodes/passage_compressor/passage_compressor.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/nodes/passage_compressor/refine.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/nodes/passage_compressor/tree_summarize.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/nodes/passage_filter/passage_filter.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/nodes/passage_filter/percentile_cutoff.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/nodes/passage_filter/recency_filter.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/nodes/passage_filter/similarity_percentile_cutoff.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/nodes/passage_filter/similarity_threshold_cutoff.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/nodes/passage_filter/threshold_cutoff.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/nodes/passage_reranker/cohere.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/nodes/passage_reranker/colbert.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/nodes/passage_reranker/flag_embedding_llm_reranker.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/nodes/passage_reranker/flag_embedding_reranker.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/nodes/passage_reranker/jina_reranker.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/nodes/passage_reranker/koreranker.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/nodes/passage_reranker/monot5.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/nodes/passage_reranker/passage_reranker.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/nodes/passage_reranker/rankgpt.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/nodes/passage_reranker/sentence_transformer_reranker.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/nodes/passage_reranker/tart.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/nodes/passage_reranker/time_reranker.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/nodes/passage_reranker/upr.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/nodes/prompt_maker/fstring.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/nodes/prompt_maker/long_context_reorder.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/nodes/prompt_maker/prompt_maker.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/nodes/prompt_maker/window_replacement.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/nodes/query_expansion/hyde.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/nodes/query_expansion/multi_query_expansion.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/nodes/query_expansion/query_decompose.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/nodes/query_expansion/query_expansion.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/nodes/retrieval/bm25.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/nodes/retrieval/hybrid_cc.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/nodes/retrieval/hybrid_rrf.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/nodes/retrieval/retrieval.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/nodes/retrieval/vectordb.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/optimization/custom_config.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/optimization/folder_structure.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/optimization/optimization.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/optimization/strategies.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/roadmap/modular_rag.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/structure.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/docs/source/troubleshooting.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/pyproject.toml +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/requirements.txt +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/sample_config/chunk/chunk_full.yaml +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/sample_config/compact_local.yaml +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/sample_config/compact_openai.yaml +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/sample_config/config_korean.yaml +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/sample_config/extracted_sample.yaml +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/sample_config/full.yaml +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/sample_config/parse/parse_hybird.yaml +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/sample_config/simple_local.yaml +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/sample_config/simple_ollama.yaml +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/sample_config/simple_openai.yaml +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/sample_dataset/README.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/sample_dataset/eli5/load_eli5_dataset.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/sample_dataset/hotpotqa/load_hotpotqa_dataset.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/sample_dataset/msmarco/load_msmarco_dataset.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/sample_dataset/triviaqa/load_triviaqa_dataset.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/setup.cfg +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/data/beta/filter/test_dontknow.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/data/beta/generation_gt/base_test_generation_gt.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/data/beta/generation_gt/test_llama_index_gen_gt.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/data/beta/generation_gt/test_openai_gen_gt.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/data/beta/query/base_test_query_gen.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/data/beta/query/test_llama_gen_query.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/data/beta/query/test_openai_gen_query.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/data/beta/test_data_creation_piepline.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/data/beta/test_sample.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/data/chunk/test_chunk_base.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/data/chunk/test_chunk_run.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/data/chunk/test_langchain_chunk.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/data/chunk/test_llama_index_chunk.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/data/corpus/test_base.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/data/corpus/test_langchain.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/data/corpus/test_llama_index_corpus.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/data/parse/test_clova.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/data/parse/test_langchain_parse.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/data/parse/test_llamaparse.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/data/parse/test_parse_base.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/data/parse/test_parse_run.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/data/parse/test_table_hybrid_parse.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/data/qacreation/test_base_qacreation.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/data/qacreation/test_llama_index_qacreation.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/data/qacreation/test_ragas_qa_creation.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/data/qacreation/test_simple.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/evaluate/metric/test_generation_metric.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/evaluate/metric/test_retrieval_contents_metric.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/evaluate/metric/test_retrieval_metric.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/evaluate/test_evaluate_util.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/evaluate/test_generation_evaluate.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/evaluate/test_retrieval_contents_evaluate.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/evaluate/test_retrieval_evaluate.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/generator/test_generator_base.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/generator/test_llama_index_llm.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/generator/test_openai.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/generator/test_run_generator_node.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/generator/test_vllm.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/passageaugmenter/test_base_passage_augmenter.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/passageaugmenter/test_pass_passage_augmenter.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/passageaugmenter/test_prev_next_augmenter.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/passageaugmenter/test_run_passage_augmenter.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/passagecompressor/test_base_passage_compressor.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/passagecompressor/test_longllmlingua.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/passagecompressor/test_pass_compressor.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/passagecompressor/test_refine.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/passagecompressor/test_run_passage_compressor_node.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/passagecompressor/test_tree_summarize.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/passagefilter/test_pass_passage_filter.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/passagefilter/test_passage_filter_base.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/passagefilter/test_passage_filter_run.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/passagefilter/test_percentile_cutoff.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/passagefilter/test_recency_filter.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/passagefilter/test_similarity_percentile_cutoff.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/passagefilter/test_similarity_threshold_cutoff.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/passagefilter/test_threshold_cutoff.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/passagereranker/test_cohere_reranker.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/passagereranker/test_colbert_reranker.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/passagereranker/test_flag_embedding.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/passagereranker/test_flag_embedding_llm.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/passagereranker/test_jina_reranker.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/passagereranker/test_koreranker.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/passagereranker/test_monot5.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/passagereranker/test_pass_reranker.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/passagereranker/test_passage_reranker_base.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/passagereranker/test_passage_reranker_run.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/passagereranker/test_rankgpt.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/passagereranker/test_sentence_transformer.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/passagereranker/test_tart.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/passagereranker/test_time_reranker.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/passagereranker/test_upr.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/promptmaker/test_fstring.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/promptmaker/test_long_context_reorder.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/promptmaker/test_prompt_maker_base.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/promptmaker/test_prompt_maker_run.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/promptmaker/test_window_replacement.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/queryexpansion/test_hyde.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/queryexpansion/test_multi_query_expansion.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/queryexpansion/test_pass_query_expansion.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/queryexpansion/test_query_decompose.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/queryexpansion/test_query_expansion_base.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/queryexpansion/test_query_expansion_run.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/retrieval/test_bm25.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/retrieval/test_hybrid_base.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/retrieval/test_hybrid_cc.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/retrieval/test_hybrid_rrf.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/retrieval/test_retrieval_base.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/retrieval/test_run_retrieval_node.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/nodes/retrieval/test_vectordb.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/schema/test_metricinput_schema.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/schema/test_module_schema.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/schema/test_node_schema.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/test_chunker.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/test_cli.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/test_dashboard.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/test_deploy.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/test_evaluator.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/test_parser.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/test_strategy.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/test_support.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/test_validator.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/test_web.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/utils/test_preprocess.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/autorag/utils/test_util.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/conftest.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/delete_tests.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/mock.py +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/requirements.txt +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/README.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/chunk_data/sample_parsed.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/corpus_data_sample.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/data_creation/raw_dir/sample1.txt +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/data_creation/raw_dir/sample2.txt +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/data_creation/raw_dir/sample3.txt +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/full.yaml +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/parse_data/all_files/baseball_1.pdf +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/parse_data/all_files/csv_sample.csv +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/parse_data/clova_data/result_sample.json +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/parse_data/clova_data/result_table.txt +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/parse_data/clova_data/result_text.txt +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/parse_data/csv_data/csv_sample.csv +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/parse_data/eng_text/baseball_1.pdf +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/parse_data/eng_text/baseball_2.pdf +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/parse_data/html_data/html_sample.html +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/parse_data/hybrid_data/nfl_rulebook_both.pdf +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/parse_data/json_data/json_sample.json +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/parse_data/korean_table/only_table/kbo_only_table.pdf +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/parse_data/korean_table/table_text/kbo_table_text.pdf +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/parse_data/korean_text/korean_texts_two_page.pdf +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/parse_data/markdown_data/markdown_sample.md +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/parse_data/xml_data/xml_sample.xml +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/qa_data_sample.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/qa_gen_prompts/prompt1.txt +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/qa_gen_prompts/prompt2.txt +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/qa_gen_prompts/prompt3.txt +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/qa_test_data_sample.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/0/config.yaml +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/0/post_retrieve_node_line/generator/0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/0/post_retrieve_node_line/generator/1.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/0/post_retrieve_node_line/generator/2.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/0/post_retrieve_node_line/generator/3.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/0/post_retrieve_node_line/generator/4.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/0/post_retrieve_node_line/generator/5.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/0/post_retrieve_node_line/generator/best_5.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/0/post_retrieve_node_line/generator/summary.csv +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/0/post_retrieve_node_line/prompt_maker/0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/0/post_retrieve_node_line/prompt_maker/1.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/0/post_retrieve_node_line/prompt_maker/best_0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/0/post_retrieve_node_line/prompt_maker/summary.csv +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/0/post_retrieve_node_line/summary.csv +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/0/pre_retrieve_node_line/query_expansion/0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/0/pre_retrieve_node_line/query_expansion/1.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/0/pre_retrieve_node_line/query_expansion/2.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/0/pre_retrieve_node_line/query_expansion/best_0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/0/pre_retrieve_node_line/query_expansion/summary.csv +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/0/pre_retrieve_node_line/summary.csv +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/0/retrieve_node_line/passage_compressor/0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/0/retrieve_node_line/passage_compressor/best_0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/0/retrieve_node_line/passage_compressor/summary.csv +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/0/retrieve_node_line/passage_reranker/0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/0/retrieve_node_line/passage_reranker/1.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/0/retrieve_node_line/passage_reranker/best_0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/0/retrieve_node_line/passage_reranker/summary.csv +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/0/retrieve_node_line/retrieval/0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/0/retrieve_node_line/retrieval/1.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/0/retrieve_node_line/retrieval/best_0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/0/retrieve_node_line/retrieval/summary.csv +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/0/retrieve_node_line/summary.csv +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/0/summary.csv +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/1/config.yaml +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/1/pre_retrieve_node_line/query_expansion/0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/1/pre_retrieve_node_line/query_expansion/1.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/1/pre_retrieve_node_line/query_expansion/2.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/1/pre_retrieve_node_line/query_expansion/best_0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/1/pre_retrieve_node_line/query_expansion/summary.csv +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/1/pre_retrieve_node_line/summary.csv +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/1/retrieve_node_line/passage_filter/0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/1/retrieve_node_line/passage_reranker/0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/1/retrieve_node_line/passage_reranker/1.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/1/retrieve_node_line/passage_reranker/best_0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/1/retrieve_node_line/passage_reranker/summary.csv +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/1/retrieve_node_line/retrieval/0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/1/retrieve_node_line/retrieval/1.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/1/retrieve_node_line/retrieval/best_0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/1/retrieve_node_line/retrieval/summary.csv +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/2/config.yaml +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/2/post_retrieve_node_line/prompt_maker/0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/2/pre_retrieve_node_line/query_expansion/0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/2/pre_retrieve_node_line/query_expansion/1.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/2/pre_retrieve_node_line/query_expansion/2.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/2/pre_retrieve_node_line/query_expansion/best_0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/2/pre_retrieve_node_line/query_expansion/summary.csv +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/2/pre_retrieve_node_line/summary.csv +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/2/retrieve_node_line/passage_compressor/0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/2/retrieve_node_line/passage_compressor/best_0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/2/retrieve_node_line/passage_compressor/summary.csv +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/2/retrieve_node_line/passage_reranker/0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/2/retrieve_node_line/passage_reranker/1.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/2/retrieve_node_line/passage_reranker/best_0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/2/retrieve_node_line/passage_reranker/summary.csv +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/2/retrieve_node_line/retrieval/0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/2/retrieve_node_line/retrieval/1.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/2/retrieve_node_line/retrieval/best_0.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/2/retrieve_node_line/retrieval/summary.csv +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/2/retrieve_node_line/summary.csv +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/3/config.yaml +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/best.yaml +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/data/corpus.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/data/qa.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/resources/bm25_porter_stemmer.pkl +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/resources/chroma/6595d304-5270-4d9f-a267-c191366804ce/data_level0.bin +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/resources/chroma/6595d304-5270-4d9f-a267-c191366804ce/header.bin +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/resources/chroma/6595d304-5270-4d9f-a267-c191366804ce/length.bin +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/resources/chroma/6595d304-5270-4d9f-a267-c191366804ce/link_lists.bin +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/resources/chroma/chroma.sqlite3 +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/result_project/trial.json +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/sample_contents_nqa.csv +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/sample_project/data/corpus.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/sample_project/data/qa.parquet +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/sample_project/resources/bm25_gpt2.pkl +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/sample_project/resources/bm25_porter_stemmer.pkl +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/simple.yaml +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/simple_chunk.yaml +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/simple_mock.yaml +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/simple_parse.yaml +0 -0
- {autorag-0.2.16 → autorag-0.2.17}/tests/resources/test_bm25_retrieval.pkl +0 -0
|
@@ -34,6 +34,7 @@ autorag/validator.py
|
|
|
34
34
|
autorag/web.py
|
|
35
35
|
autorag/data/__init__.py
|
|
36
36
|
autorag/data/beta/__init__.py
|
|
37
|
+
autorag/data/beta/extract_evidence.py
|
|
37
38
|
autorag/data/beta/sample.py
|
|
38
39
|
autorag/data/beta/schema.py
|
|
39
40
|
autorag/data/beta/filter/__init__.py
|
|
@@ -229,9 +230,19 @@ docs/source/data_creation/data_format.md
|
|
|
229
230
|
docs/source/data_creation/ragas.md
|
|
230
231
|
docs/source/data_creation/tutorial.md
|
|
231
232
|
docs/source/data_creation/beta/data_creation.md
|
|
232
|
-
docs/source/data_creation/beta/filter.md
|
|
233
|
-
docs/source/data_creation/beta/query_gen.md
|
|
234
233
|
docs/source/data_creation/beta/tutorial.md
|
|
234
|
+
docs/source/data_creation/beta/chunk/chunk.md
|
|
235
|
+
docs/source/data_creation/beta/chunk/langchain_chunk.md
|
|
236
|
+
docs/source/data_creation/beta/chunk/llama_index_chunk.md
|
|
237
|
+
docs/source/data_creation/beta/parse/clova.md
|
|
238
|
+
docs/source/data_creation/beta/parse/langchain_parse.md
|
|
239
|
+
docs/source/data_creation/beta/parse/llama_parse.md
|
|
240
|
+
docs/source/data_creation/beta/parse/parse.md
|
|
241
|
+
docs/source/data_creation/beta/parse/table_hybrid_parse.md
|
|
242
|
+
docs/source/data_creation/beta/qa_creation/answer_gen.md
|
|
243
|
+
docs/source/data_creation/beta/qa_creation/filter.md
|
|
244
|
+
docs/source/data_creation/beta/qa_creation/qa_creation.md
|
|
245
|
+
docs/source/data_creation/beta/qa_creation/query_gen.md
|
|
235
246
|
docs/source/deploy/api_endpoint.md
|
|
236
247
|
docs/source/deploy/web.md
|
|
237
248
|
docs/source/evaluate_metrics/generation.md
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
0.2.17
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
# This module is about extracting evidence from the given retrieval gt passage
|
|
@@ -1,9 +1,9 @@
|
|
|
1
1
|
import logging
|
|
2
|
-
from typing import Callable, Optional, Dict, Awaitable, Any
|
|
2
|
+
from typing import Callable, Optional, Dict, Awaitable, Any, Tuple, List
|
|
3
3
|
import pandas as pd
|
|
4
|
+
from autorag.utils.util import process_batch, get_event_loop, fetch_contents
|
|
4
5
|
|
|
5
6
|
from autorag.support import get_support_modules
|
|
6
|
-
from autorag.utils.util import process_batch, get_event_loop, fetch_contents
|
|
7
7
|
|
|
8
8
|
logger = logging.getLogger("AutoRAG")
|
|
9
9
|
|
|
@@ -68,6 +68,20 @@ class Corpus:
|
|
|
68
68
|
def linked_raw(self, raw: Raw):
|
|
69
69
|
raise NotImplementedError("linked_raw is read-only.")
|
|
70
70
|
|
|
71
|
+
def to_parquet(self, save_path: str):
|
|
72
|
+
"""
|
|
73
|
+
Save the corpus to the AutoRAG compatible parquet file.
|
|
74
|
+
It is not for the data creation, for running AutoRAG.
|
|
75
|
+
If you want to save it directly, use the below code.
|
|
76
|
+
`corpus.data.to_parquet(save_path)`
|
|
77
|
+
|
|
78
|
+
:param save_path: The path to save the corpus.
|
|
79
|
+
"""
|
|
80
|
+
if not save_path.endswith(".parquet"):
|
|
81
|
+
raise ValueError("save_path must be ended with .parquet")
|
|
82
|
+
save_df = self.data[["doc_id", "contents", "metadata"]].reset_index(drop=True)
|
|
83
|
+
save_df.to_parquet(save_path)
|
|
84
|
+
|
|
71
85
|
def batch_apply(
|
|
72
86
|
self, fn: Callable[[Dict, Any], Awaitable[Dict]], batch_size: int = 32, **kwargs
|
|
73
87
|
) -> "Corpus":
|
|
@@ -152,6 +166,26 @@ class QA:
|
|
|
152
166
|
)
|
|
153
167
|
return self
|
|
154
168
|
|
|
169
|
+
def to_parquet(self, qa_save_path: str, corpus_save_path: str):
|
|
170
|
+
"""
|
|
171
|
+
Save the qa and corpus to the AutoRAG compatible parquet file.
|
|
172
|
+
It is not for the data creation, for running AutoRAG.
|
|
173
|
+
If you want to save it directly, use the below code.
|
|
174
|
+
`qa.data.to_parquet(save_path)`
|
|
175
|
+
|
|
176
|
+
:param qa_save_path: The path to save the qa dataset.
|
|
177
|
+
:param corpus_save_path: The path to save the corpus.
|
|
178
|
+
"""
|
|
179
|
+
if not qa_save_path.endswith(".parquet"):
|
|
180
|
+
raise ValueError("save_path must be ended with .parquet")
|
|
181
|
+
if not corpus_save_path.endswith(".parquet"):
|
|
182
|
+
raise ValueError("save_path must be ended with .parquet")
|
|
183
|
+
save_df = self.data[
|
|
184
|
+
["qid", "query", "retrieval_gt", "generation_gt"]
|
|
185
|
+
].reset_index(drop=True)
|
|
186
|
+
save_df.to_parquet(qa_save_path)
|
|
187
|
+
self.linked_corpus.to_parquet(corpus_save_path)
|
|
188
|
+
|
|
155
189
|
def update_corpus(self, new_corpus: Corpus) -> "QA":
|
|
156
190
|
"""
|
|
157
191
|
Update linked corpus.
|
|
@@ -163,4 +197,105 @@ class QA:
|
|
|
163
197
|
Must have valid `linked_raw` and `raw_id`, `raw_start_idx`, `raw_end_idx` columns.
|
|
164
198
|
:return: The QA instance that updated linked corpus.
|
|
165
199
|
"""
|
|
166
|
-
|
|
200
|
+
self.data["evidence_path"] = (
|
|
201
|
+
self.data["retrieval_gt"]
|
|
202
|
+
.apply(
|
|
203
|
+
lambda x: fetch_contents(
|
|
204
|
+
self.linked_corpus.data,
|
|
205
|
+
x,
|
|
206
|
+
column_name="path",
|
|
207
|
+
)
|
|
208
|
+
)
|
|
209
|
+
.tolist()
|
|
210
|
+
)
|
|
211
|
+
self.data["evidence_page"] = self.data["retrieval_gt"].apply(
|
|
212
|
+
lambda x: list(
|
|
213
|
+
map(
|
|
214
|
+
lambda lst: list(map(lambda x: x.get("page", -1), lst)),
|
|
215
|
+
fetch_contents(self.linked_corpus.data, x, column_name="metadata"),
|
|
216
|
+
)
|
|
217
|
+
)
|
|
218
|
+
)
|
|
219
|
+
if "evidence_start_end_idx" not in self.data.columns:
|
|
220
|
+
# make evidence start_end_idx
|
|
221
|
+
self.data["evidence_start_end_idx"] = (
|
|
222
|
+
self.data["retrieval_gt"]
|
|
223
|
+
.apply(
|
|
224
|
+
lambda x: fetch_contents(
|
|
225
|
+
self.linked_corpus.data,
|
|
226
|
+
x,
|
|
227
|
+
column_name="start_end_idx",
|
|
228
|
+
)
|
|
229
|
+
)
|
|
230
|
+
.tolist()
|
|
231
|
+
)
|
|
232
|
+
|
|
233
|
+
# matching the new corpus with the old corpus
|
|
234
|
+
path_corpus_dict = QA.__make_path_corpus_dict(new_corpus.data)
|
|
235
|
+
new_retrieval_gt = self.data.apply(
|
|
236
|
+
lambda row: QA.__match_index_row(
|
|
237
|
+
row["evidence_start_end_idx"],
|
|
238
|
+
row["evidence_path"],
|
|
239
|
+
row["evidence_page"],
|
|
240
|
+
path_corpus_dict,
|
|
241
|
+
),
|
|
242
|
+
axis=1,
|
|
243
|
+
).tolist()
|
|
244
|
+
new_qa = self.data.copy(deep=True)[["qid", "query", "generation_gt"]]
|
|
245
|
+
new_qa["retrieval_gt"] = new_retrieval_gt
|
|
246
|
+
return QA(new_qa, new_corpus)
|
|
247
|
+
|
|
248
|
+
@staticmethod
|
|
249
|
+
def __match_index(target_idx: Tuple[int, int], dst_idx: Tuple[int, int]) -> bool:
|
|
250
|
+
"""
|
|
251
|
+
Check if the target_idx is overlap by the dst_idx.
|
|
252
|
+
"""
|
|
253
|
+
target_start, target_end = target_idx
|
|
254
|
+
dst_start, dst_end = dst_idx
|
|
255
|
+
return (
|
|
256
|
+
dst_start <= target_start <= dst_end or dst_start <= target_end <= dst_end
|
|
257
|
+
)
|
|
258
|
+
|
|
259
|
+
@staticmethod
|
|
260
|
+
def __match_index_row(
|
|
261
|
+
evidence_indices: List[List[Tuple[int, int]]],
|
|
262
|
+
evidence_paths: List[List[str]],
|
|
263
|
+
evidence_pages: List[List[int]],
|
|
264
|
+
path_corpus_dict: Dict,
|
|
265
|
+
) -> List[List[str]]:
|
|
266
|
+
"""
|
|
267
|
+
Find the matched passage from new_corpus.
|
|
268
|
+
|
|
269
|
+
:param evidence_indices: The evidence indices at the corresponding Raw.
|
|
270
|
+
Its shape is the same as the retrieval_gt.
|
|
271
|
+
:param evidence_paths: The evidence paths at the corresponding Raw.
|
|
272
|
+
Its shape is the same as the retrieval_gt.
|
|
273
|
+
:param path_corpus_dict: The key is the path name, and the value is the corpus dataframe that only contains the path in the key.
|
|
274
|
+
You can make it using `QA.__make_path_corpus_dict`.
|
|
275
|
+
:return:
|
|
276
|
+
"""
|
|
277
|
+
result = []
|
|
278
|
+
for i, idx_list in enumerate(evidence_indices):
|
|
279
|
+
sub_result = []
|
|
280
|
+
for j, idx in enumerate(idx_list):
|
|
281
|
+
path_corpus_df = path_corpus_dict[evidence_paths[i][j]]
|
|
282
|
+
if evidence_pages[i][j] >= 0:
|
|
283
|
+
path_corpus_df = path_corpus_df.loc[
|
|
284
|
+
path_corpus_df["metadata"].apply(lambda x: x.get("page", -1))
|
|
285
|
+
== evidence_pages[i][j]
|
|
286
|
+
]
|
|
287
|
+
matched_corpus = path_corpus_df.loc[
|
|
288
|
+
path_corpus_df["start_end_idx"].apply(
|
|
289
|
+
lambda x: QA.__match_index(idx, x)
|
|
290
|
+
)
|
|
291
|
+
]
|
|
292
|
+
sub_result.extend(matched_corpus["doc_id"].tolist())
|
|
293
|
+
result.append(sub_result)
|
|
294
|
+
return result
|
|
295
|
+
|
|
296
|
+
@staticmethod
|
|
297
|
+
def __make_path_corpus_dict(corpus_df: pd.DataFrame) -> Dict[str, pd.DataFrame]:
|
|
298
|
+
return {
|
|
299
|
+
path: corpus_df[corpus_df["path"] == path]
|
|
300
|
+
for path in corpus_df["path"].unique()
|
|
301
|
+
}
|
|
@@ -19,6 +19,20 @@ def clova_ocr(
|
|
|
19
19
|
batch: int = 8,
|
|
20
20
|
table_detection: bool = False,
|
|
21
21
|
) -> Tuple[List[str], List[str], List[int]]:
|
|
22
|
+
"""
|
|
23
|
+
Parse documents to use Naver Clova OCR.
|
|
24
|
+
|
|
25
|
+
:param data_path_list: The list of data paths to parse.
|
|
26
|
+
:param url: The URL for Clova OCR.
|
|
27
|
+
You can get the URL with the guide at https://guide.ncloud-docs.com/docs/clovaocr-example01
|
|
28
|
+
You can set the environment variable CLOVA_URL, or you can set it directly as a parameter.
|
|
29
|
+
:param api_key: The API key for Clova OCR.
|
|
30
|
+
You can get the API key with the guide at https://guide.ncloud-docs.com/docs/clovaocr-example01
|
|
31
|
+
You can set the environment variable CLOVA_API_KEY, or you can set it directly as a parameter.
|
|
32
|
+
:param batch: The batch size for parse documents. Default is 8.
|
|
33
|
+
:param table_detection: Whether to enable table detection. Default is False.
|
|
34
|
+
:return: tuple of lists containing the parsed texts, path and pages.
|
|
35
|
+
"""
|
|
22
36
|
url = os.getenv("CLOVA_URL", None) if url is None else url
|
|
23
37
|
if url is None:
|
|
24
38
|
raise KeyError(
|
|
@@ -9,6 +9,14 @@ from autorag.data.parse.base import parser_node
|
|
|
9
9
|
def langchain_parse(
|
|
10
10
|
data_path_list: List[str], parse_method: str, **kwargs
|
|
11
11
|
) -> Tuple[List[str], List[str], List[int]]:
|
|
12
|
+
"""
|
|
13
|
+
Parse documents to use langchain document_loaders(parse) method
|
|
14
|
+
|
|
15
|
+
:param data_path_list: The list of data paths to parse.
|
|
16
|
+
:param parse_method: A langchain document_loaders(parse) method to use.
|
|
17
|
+
:param kwargs: The extra parameters for creating the langchain document_loaders(parse) instance.
|
|
18
|
+
:return: tuple of lists containing the parsed texts, path and pages.
|
|
19
|
+
"""
|
|
12
20
|
if parse_method in ["directory", "unstructured"]:
|
|
13
21
|
results = parse_all_files(data_path_list, parse_method, **kwargs)
|
|
14
22
|
texts, path = results[0], results[1]
|
|
@@ -10,6 +10,16 @@ from autorag.utils.util import process_batch, get_event_loop
|
|
|
10
10
|
def llama_parse(
|
|
11
11
|
data_path_list: List[str], batch: int = 8, **kwargs
|
|
12
12
|
) -> Tuple[List[str], List[str], List[int]]:
|
|
13
|
+
"""
|
|
14
|
+
Parse documents to use llama_parse.
|
|
15
|
+
LLAMA_CLOUD_API_KEY environment variable should be set.
|
|
16
|
+
You can get the key from https://cloud.llamaindex.ai/api-key
|
|
17
|
+
|
|
18
|
+
:param data_path_list: The list of data paths to parse.
|
|
19
|
+
:param batch: The batch size for parse documents. Default is 8.
|
|
20
|
+
:param kwargs: The extra parameters for creating the llama_parse instance.
|
|
21
|
+
:return: tuple of lists containing the parsed texts, path and pages.
|
|
22
|
+
"""
|
|
13
23
|
parse_instance = LlamaParse(**kwargs)
|
|
14
24
|
|
|
15
25
|
tasks = [
|
|
@@ -18,6 +18,18 @@ def table_hybrid_parse(
|
|
|
18
18
|
table_parse_module: str,
|
|
19
19
|
table_params: Dict,
|
|
20
20
|
) -> Tuple[List[str], List[str], List[int]]:
|
|
21
|
+
"""
|
|
22
|
+
Parse documents to use table_hybrid_parse method.
|
|
23
|
+
The table_hybrid_parse method is a hybrid method that combines the parsing results of PDFs with and without tables.
|
|
24
|
+
It splits the PDF file into pages, separates pages with and without tables, and then parses and merges the results.
|
|
25
|
+
|
|
26
|
+
:param data_path_list: The list of data paths to parse.
|
|
27
|
+
:param text_parse_module: The text parsing module to use. The type should be a string.
|
|
28
|
+
:param text_params: The extra parameters for the text parsing module. The type should be a dictionary.
|
|
29
|
+
:param table_parse_module: The table parsing module to use. The type should be a string.
|
|
30
|
+
:param table_params: The extra parameters for the table parsing module. The type should be a dictionary.
|
|
31
|
+
:return: tuple of lists containing the parsed texts, path and pages.
|
|
32
|
+
"""
|
|
21
33
|
# make save folder directory
|
|
22
34
|
with tempfile.TemporaryDirectory() as save_dir:
|
|
23
35
|
text_dir = os.path.join(save_dir, "text")
|
|
@@ -105,14 +105,15 @@ async def vectordb_pure(
|
|
|
105
105
|
:param query_embeddings: A list of query embeddings.
|
|
106
106
|
:param top_k: The number of passages to be retrieved.
|
|
107
107
|
:param collection: A chroma collection instance that will be used to retrieve passages.
|
|
108
|
-
|
|
109
|
-
:return: The tuple contains a list of passage ids that retrieved from vectordb and a list of its scores.
|
|
108
|
+
:return: The tuple contains a list of passage ids that are retrieved from vectordb and a list of its scores.
|
|
110
109
|
"""
|
|
111
110
|
id_result, score_result = [], []
|
|
112
111
|
for embedded_query in query_embeddings:
|
|
113
112
|
result = collection.query(query_embeddings=embedded_query, n_results=top_k)
|
|
114
113
|
id_result.extend(result["ids"])
|
|
115
|
-
score_result.extend(
|
|
114
|
+
score_result.extend(
|
|
115
|
+
list(map(lambda lst: list(map(lambda x: 1 - x, lst)), result["distances"]))
|
|
116
|
+
)
|
|
116
117
|
|
|
117
118
|
# Distribute passages evenly
|
|
118
119
|
id_result, score_result = evenly_distribute_passages(id_result, score_result, top_k)
|
|
Binary file
|
|
@@ -0,0 +1,168 @@
|
|
|
1
|
+
# Chunk
|
|
2
|
+
|
|
3
|
+
In this section, we will cover how to chunk parsed result.
|
|
4
|
+
|
|
5
|
+
It is a crucial step because if the parsed result is not chunked well, the RAG will not be optimized well.
|
|
6
|
+
|
|
7
|
+
Using only YAML files, you can easily use the various chunk methods.
|
|
8
|
+
The chunked result is saved according to the data format used by AutoRAG.
|
|
9
|
+
|
|
10
|
+
## Overview
|
|
11
|
+
|
|
12
|
+
The sample chunk pipeline looks like this.
|
|
13
|
+
|
|
14
|
+
```python
|
|
15
|
+
from autorag.chunker import Chunker
|
|
16
|
+
|
|
17
|
+
chunker = Chunker.from_parquet(parsed_data_path="your/parsed/data/path")
|
|
18
|
+
chunker.start_chunking("your/path/to/chunk_config.yaml")
|
|
19
|
+
```
|
|
20
|
+
|
|
21
|
+
## Features
|
|
22
|
+
|
|
23
|
+
### 1. Add File Name
|
|
24
|
+
You need to set one of 'English' and 'Korean'
|
|
25
|
+
The 'add_file_name' feature is to add a file_name to chunked_contents.
|
|
26
|
+
This is used to prevent hallucination by retrieving contents from the wrong document.
|
|
27
|
+
Default form of English is `"file_name: {file_name}\n contents: {content}"`
|
|
28
|
+
|
|
29
|
+
#### Example YAML
|
|
30
|
+
|
|
31
|
+
```yaml
|
|
32
|
+
modules:
|
|
33
|
+
- module_type: llama_index_chunk
|
|
34
|
+
chunk_method: [ Token, Sentence ]
|
|
35
|
+
chunk_size: [ 1024, 512 ]
|
|
36
|
+
chunk_overlap: 24
|
|
37
|
+
add_file_name: english
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
### 2. Sentence Splitter
|
|
41
|
+
|
|
42
|
+
The following chunk methods in the `llama_index_chunk` module use the sentence splitter.
|
|
43
|
+
|
|
44
|
+
- `Semantic_llama_index`
|
|
45
|
+
- `SemanticDoubling`
|
|
46
|
+
- `SentenceWindow`
|
|
47
|
+
|
|
48
|
+
The following methods use `PunktSentenceTokenizer` as the default sentence splitter.
|
|
49
|
+
|
|
50
|
+
See below for the available languages of `PunktSentenceTokenizer`.
|
|
51
|
+
|
|
52
|
+
["Czech, Danish, Dutch, English, Estonian, Finnish, French, German, Greek, Italian, Malayalam, Norwegian, Polish, Portuguese, Russian, Slovenian, Spanish, Swedish, Turkish"]
|
|
53
|
+
|
|
54
|
+
So if the language you want to use is not in the list, or you want to use a different sentence splitter, you can use the sentence_splitter parameter.
|
|
55
|
+
|
|
56
|
+
#### Available Sentence Splitter
|
|
57
|
+
- [kiwi](https://github.com/bab2min/kiwipiepy) : For Korean 🇰🇷
|
|
58
|
+
|
|
59
|
+
#### Example YAML
|
|
60
|
+
|
|
61
|
+
```yaml
|
|
62
|
+
modules:
|
|
63
|
+
- module_type: llama_index_chunk
|
|
64
|
+
chunk_method: [ SentenceWindow ]
|
|
65
|
+
sentence_splitter: kiwi
|
|
66
|
+
window_size: 3
|
|
67
|
+
add_file_name: english
|
|
68
|
+
```
|
|
69
|
+
|
|
70
|
+
#### Using sentence splitter that is not in the Available Sentence Splitter
|
|
71
|
+
|
|
72
|
+
If you want to use `kiwi`, you can use the following code.
|
|
73
|
+
|
|
74
|
+
```python
|
|
75
|
+
from autorag.data import sentence_splitter_modules, LazyInit
|
|
76
|
+
|
|
77
|
+
def split_by_sentence_kiwi() -> Callable[[str], List[str]]:
|
|
78
|
+
from kiwipiepy import Kiwi
|
|
79
|
+
|
|
80
|
+
kiwi = Kiwi()
|
|
81
|
+
|
|
82
|
+
def split(text: str) -> List[str]:
|
|
83
|
+
kiwi_result = kiwi.split_into_sents(text)
|
|
84
|
+
sentences = list(map(lambda x: x.text, kiwi_result))
|
|
85
|
+
|
|
86
|
+
return sentences
|
|
87
|
+
|
|
88
|
+
return split
|
|
89
|
+
|
|
90
|
+
sentence_splitter_modules["kiwi"] = LazyInit(split_by_sentence_kiwi)
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
## Run Chunk Pipeline
|
|
94
|
+
|
|
95
|
+
### 1. Set chunker instance
|
|
96
|
+
|
|
97
|
+
```python
|
|
98
|
+
from autorag.chunker import Chunker
|
|
99
|
+
|
|
100
|
+
chunker = Chunker.from_parquet(parsed_data_path="your/parsed/data/path")
|
|
101
|
+
```
|
|
102
|
+
|
|
103
|
+
```{admonition} Want to specify project folder?
|
|
104
|
+
You can specify project directory with `--project_dir` option or project_dir parameter.
|
|
105
|
+
```
|
|
106
|
+
|
|
107
|
+
### 2. Set YAML file
|
|
108
|
+
|
|
109
|
+
Here is an example of how to use the `llama_index_chunk` module.
|
|
110
|
+
|
|
111
|
+
```yaml
|
|
112
|
+
modules:
|
|
113
|
+
- module_type: llama_index_chunk
|
|
114
|
+
chunk_method: [ Token, Sentence ]
|
|
115
|
+
chunk_size: [ 1024, 512 ]
|
|
116
|
+
chunk_overlap: 24
|
|
117
|
+
```
|
|
118
|
+
|
|
119
|
+
### 3. Start chunking
|
|
120
|
+
|
|
121
|
+
Use `start_chunking` function to start parsing.
|
|
122
|
+
|
|
123
|
+
```python
|
|
124
|
+
chunker.start_chunking("your/path/to/chunk_config.yaml")
|
|
125
|
+
```
|
|
126
|
+
|
|
127
|
+
### 4. Check the result
|
|
128
|
+
|
|
129
|
+
If you set `project_dir` parameter, you can check the result in the project directory.
|
|
130
|
+
If not, you can check the result in the current directory.
|
|
131
|
+
|
|
132
|
+
The way to check the result is the same as the `Evaluator` and `Parser` in AutoRAG.
|
|
133
|
+
|
|
134
|
+
A `trial_folder` is created in `project_dir` first.
|
|
135
|
+
|
|
136
|
+
If the chunking is completed successfully, the following three types of files are created in the trial_folder.
|
|
137
|
+
|
|
138
|
+
1. Chunked Result
|
|
139
|
+
2. Used YAML file
|
|
140
|
+
3. Summary file
|
|
141
|
+
|
|
142
|
+
For example, if chunking is performed using three chunk methods, the following files are created.
|
|
143
|
+
`0.parquet`, `1.parquet`, `2.parquet`, `parse_config.yaml`, `summary.csv`
|
|
144
|
+
|
|
145
|
+
Finally, in the summary.csv file, you can see information about the chunked result, such as what chunk method was used to chunk it.
|
|
146
|
+
|
|
147
|
+
## Output Columns
|
|
148
|
+
- `doc_id`: Document ID. The type is string.
|
|
149
|
+
- `contents`: The contents of the chunked data. The type is string.
|
|
150
|
+
- `path`: The path of the document. The type is string.
|
|
151
|
+
- `start_end_idx`:
|
|
152
|
+
- Store index of chunked_str based on original_str before chunking
|
|
153
|
+
- stored to map the retrieval_gt of Evaluation QA Dataset according to various chunk methods.
|
|
154
|
+
- `metadata`: It is also stored in the passage after the data of the parsed result is chunked. The type is dictionary.
|
|
155
|
+
- Depending on the dataformat of AutoRAG's `Parsed Result`, metadata should have the following keys: `page`, `last_modified_datetime`, `path`.
|
|
156
|
+
|
|
157
|
+
#### Supported Chunk Modules
|
|
158
|
+
|
|
159
|
+
📌 You can check our all Chunk modules
|
|
160
|
+
at [here](https://edai.notion.site/Supporting-Chunk-Modules-8db803dba2ec4cd0a8789659106e86a3?pvs=4)
|
|
161
|
+
|
|
162
|
+
```{toctree}
|
|
163
|
+
---
|
|
164
|
+
maxdepth: 1
|
|
165
|
+
---
|
|
166
|
+
langchain_chunk.md
|
|
167
|
+
llama_index_chunk.md
|
|
168
|
+
```
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
# Langchain Chunk
|
|
2
|
+
|
|
3
|
+
Chunk parsed results to use [langchain text splitters](https://api.python.langchain.com/en/latest/text_splitters_api_reference.html#).
|
|
4
|
+
|
|
5
|
+
## Available Chunk Method
|
|
6
|
+
|
|
7
|
+
### 1. Token
|
|
8
|
+
|
|
9
|
+
- [SentenceTransformersToken](https://api.python.langchain.com/en/latest/sentence_transformers/langchain_text_splitters.sentence_transformers.SentenceTransformersTokenTextSplitter.html)
|
|
10
|
+
|
|
11
|
+
### 2. Character
|
|
12
|
+
|
|
13
|
+
- [RecursiveCharacter](https://api.python.langchain.com/en/latest/character/langchain_text_splitters.character.RecursiveCharacterTextSplitter.html)
|
|
14
|
+
- [character](https://api.python.langchain.com/en/latest/character/langchain_text_splitters.character.CharacterTextSplitter.html)
|
|
15
|
+
|
|
16
|
+
### 3. Sentence
|
|
17
|
+
|
|
18
|
+
- [konlpy](https://api.python.langchain.com/en/latest/konlpy/langchain_text_splitters.konlpy.KonlpyTextSplitter.html): For Korean 🇰🇷
|
|
19
|
+
|
|
20
|
+
#### Example YAML
|
|
21
|
+
|
|
22
|
+
```yaml
|
|
23
|
+
modules:
|
|
24
|
+
- module_type: langchain_chunk
|
|
25
|
+
parse_method: konlpy
|
|
26
|
+
add_file_name: korean
|
|
27
|
+
```
|
|
28
|
+
|
|
29
|
+
## Using Langchain Chunk Method that is not in the Available Chunk Method
|
|
30
|
+
|
|
31
|
+
You can find more information about the langchain chunk method at
|
|
32
|
+
[here](https://api.python.langchain.com/en/latest/text_splitters_api_reference.html#)
|
|
33
|
+
|
|
34
|
+
### How to Use
|
|
35
|
+
|
|
36
|
+
If you want to use `PythonCodeTextSplitter` that is not in the available chunk method, you can use the following code.
|
|
37
|
+
|
|
38
|
+
```python
|
|
39
|
+
from autorag.data import chunk_modules
|
|
40
|
+
from langchain.text_splitter import PythonCodeTextSplitter
|
|
41
|
+
|
|
42
|
+
chunk_modules["python"] = PythonCodeTextSplitter
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
```{attention}
|
|
46
|
+
The key value in chunk_modules must always be written in lowercase.
|
|
47
|
+
```
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
# Llama Index Chunk
|
|
2
|
+
|
|
3
|
+
Chunk parsed results to use [Llama Index Node_Parsers & Text Splitters](https://docs.llamaindex.ai/en/stable/api_reference/node_parsers/).
|
|
4
|
+
|
|
5
|
+
## Available Chunk Method
|
|
6
|
+
|
|
7
|
+
### 1. Token
|
|
8
|
+
|
|
9
|
+
- [Token](https://docs.llamaindex.ai/en/stable/api_reference/node_parsers/token_text_splitter/)
|
|
10
|
+
|
|
11
|
+
### 2. Sentence
|
|
12
|
+
|
|
13
|
+
- [Sentence](https://docs.llamaindex.ai/en/stable/api_reference/node_parsers/sentence_splitter/)
|
|
14
|
+
|
|
15
|
+
### 3. Window
|
|
16
|
+
|
|
17
|
+
- [SentenceWindow](https://docs.llamaindex.ai/en/stable/api_reference/node_parsers/sentence_window/)
|
|
18
|
+
|
|
19
|
+
### 4. Semantic
|
|
20
|
+
|
|
21
|
+
- [semantic_llama_index](https://docs.llamaindex.ai/en/stable/api_reference/node_parsers/semantic_splitter/)
|
|
22
|
+
- [SemanticDoubleMerging](https://docs.llamaindex.ai/en/stable/examples/node_parsers/semantic_double_merging_chunking/)
|
|
23
|
+
|
|
24
|
+
### 5. Simple
|
|
25
|
+
|
|
26
|
+
- [Simple](https://docs.llamaindex.ai/en/v0.10.19/api/llama_index.core.node_parser.SimpleFileNodeParser.html)
|
|
27
|
+
|
|
28
|
+
#### Example YAML
|
|
29
|
+
|
|
30
|
+
```yaml
|
|
31
|
+
modules:
|
|
32
|
+
- module_type: llama_index_chunk
|
|
33
|
+
chunk_method: [ Token, Sentence ]
|
|
34
|
+
chunk_size: [ 1024, 512 ]
|
|
35
|
+
chunk_overlap: 24
|
|
36
|
+
add_file_name: english
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
## Using Llama Index Chunk Method that is not in the Available Chunk Method
|
|
40
|
+
|
|
41
|
+
You can find more information about the llama index chunk method at
|
|
42
|
+
[here](https://docs.llamaindex.ai/en/stable/api_reference/node_parsers/).
|
|
43
|
+
|
|
44
|
+
### How to Use
|
|
45
|
+
|
|
46
|
+
If you want to use `HTMLNodeParser` that is not in the available chunk method, you can use the following code.
|
|
47
|
+
|
|
48
|
+
```python
|
|
49
|
+
from autorag.data import chunk_modules
|
|
50
|
+
from llama_index.core.node_parser import HTMLNodeParser
|
|
51
|
+
|
|
52
|
+
chunk_modules["html"] = HTMLNodeParser
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
```{attention}
|
|
56
|
+
The key value in chunk_modules must always be written in lowercase.
|
|
57
|
+
```
|
|
@@ -22,25 +22,14 @@ In this new data creation pipeline, we have three schemas. `Raw`, `QA`, and `Cor
|
|
|
22
22
|
You can use the corpus to generate the answer for the question.
|
|
23
23
|
You have to make corpus data from your documents using parsing and chunking.
|
|
24
24
|
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
We provide some functions for customizing and running a data creation process.
|
|
28
|
-
Here are the basic concepts of each function.
|
|
29
|
-
|
|
30
|
-
### `QA` and `Corpus`
|
|
31
|
-
|
|
32
|
-
- `batch_apply`: Apply the function to each row of the dataset, but run in parallel using `asyncio`.
|
|
33
|
-
You have to use `async` functions for using this function.
|
|
34
|
-
Plus, you can specify the batch size.
|
|
35
|
-
- `map`: You can use this function to do something to its pd.DataFrame value. The input function must be get input of pd.DataFrame and return pd.DataFrame.
|
|
36
|
-
|
|
37
|
-
|
|
25
|
+
To see the tutorial of the data creation, check [here](./tutorial.md).
|
|
38
26
|
|
|
39
27
|
```{toctree}
|
|
40
28
|
---
|
|
41
|
-
maxdepth:
|
|
29
|
+
maxdepth: 2
|
|
42
30
|
---
|
|
43
31
|
tutorial.md
|
|
44
|
-
|
|
45
|
-
|
|
32
|
+
qa_creation/qa_creation.md
|
|
33
|
+
chunk/chunk.md
|
|
34
|
+
parse/parse.md
|
|
46
35
|
```
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
# Clova
|
|
2
|
+
|
|
3
|
+
Parse raw documents to use Naver
|
|
4
|
+
[Clova OCR](https://guide.ncloud-docs.com/docs/clovaocr-overview).
|
|
5
|
+
|
|
6
|
+
Clova OCR divides the document into pages for parsing.
|
|
7
|
+
|
|
8
|
+
## Table Detection
|
|
9
|
+
|
|
10
|
+
If you have tables in your raw document, set `table_detection: true` to use clova ocr table detection feature.
|
|
11
|
+
|
|
12
|
+
### Point
|
|
13
|
+
|
|
14
|
+
#### 1. HTML Parser
|
|
15
|
+
Clova OCR provides parsed table information in complex JSON format.
|
|
16
|
+
It converts the complex JSON form of the table to HTML for storage in the LLM.
|
|
17
|
+
|
|
18
|
+
The parser was created by our own AutoRAG team and you can find the detailed code in the `json_to_html_table` function in `autorag.data.parse.clova`.
|
|
19
|
+
|
|
20
|
+
#### 2. The text information comes separately from the table information.
|
|
21
|
+
If your document is a table + text, the text information comes separately from the table information.
|
|
22
|
+
So when using table_detection, it will be saved in `{text}\n\ntable html:\n{table}` format.
|
|
23
|
+
|
|
24
|
+
## Example YAML
|
|
25
|
+
|
|
26
|
+
```yaml
|
|
27
|
+
modules:
|
|
28
|
+
- module_type: clova
|
|
29
|
+
table_detection: true
|
|
30
|
+
```
|