AutoRAG 0.2.9__tar.gz → 0.2.10__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (455) hide show
  1. {autorag-0.2.9 → autorag-0.2.10}/AutoRAG.egg-info/PKG-INFO +1 -1
  2. {autorag-0.2.9 → autorag-0.2.10}/AutoRAG.egg-info/SOURCES.txt +0 -7
  3. {autorag-0.2.9 → autorag-0.2.10}/PKG-INFO +1 -1
  4. autorag-0.2.10/autorag/VERSION +1 -0
  5. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/retrieval/__init__.py +0 -2
  6. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/retrieval/base.py +3 -4
  7. autorag-0.2.10/autorag/nodes/retrieval/hybrid_cc.py +137 -0
  8. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/retrieval/hybrid_rrf.py +14 -5
  9. autorag-0.2.10/autorag/nodes/retrieval/run.py +285 -0
  10. {autorag-0.2.9 → autorag-0.2.10}/autorag/support.py +0 -2
  11. {autorag-0.2.9 → autorag-0.2.10}/docs/source/api_spec/autorag.nodes.retrieval.rst +0 -16
  12. {autorag-0.2.9 → autorag-0.2.10}/docs/source/evaluate_metrics/generation.md +16 -5
  13. autorag-0.2.10/docs/source/nodes/retrieval/hybrid_cc.md +59 -0
  14. autorag-0.2.10/docs/source/nodes/retrieval/hybrid_rrf.md +34 -0
  15. {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/retrieval/retrieval.md +5 -10
  16. {autorag-0.2.9 → autorag-0.2.10}/sample_config/compact_local.yaml +1 -7
  17. autorag-0.2.10/sample_config/compact_openai.yaml +59 -0
  18. {autorag-0.2.9 → autorag-0.2.10}/sample_config/full.yaml +4 -19
  19. {autorag-0.2.9 → autorag-0.2.10}/sample_config/simple_ollama.yaml +2 -19
  20. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/retrieval/test_hybrid_base.py +18 -1
  21. autorag-0.2.10/tests/autorag/nodes/retrieval/test_hybrid_cc.py +54 -0
  22. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/retrieval/test_hybrid_rrf.py +3 -3
  23. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/retrieval/test_run_retrieval_node.py +17 -38
  24. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/test_evaluator.py +9 -13
  25. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/full.yaml +4 -7
  26. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/3/config.yaml +1 -2
  27. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/simple.yaml +1 -2
  28. autorag-0.2.9/autorag/VERSION +0 -1
  29. autorag-0.2.9/autorag/nodes/retrieval/hybrid_cc.py +0 -63
  30. autorag-0.2.9/autorag/nodes/retrieval/hybrid_dbsf.py +0 -70
  31. autorag-0.2.9/autorag/nodes/retrieval/hybrid_rsf.py +0 -96
  32. autorag-0.2.9/autorag/nodes/retrieval/run.py +0 -217
  33. autorag-0.2.9/docs/source/nodes/retrieval/hybrid_cc.md +0 -44
  34. autorag-0.2.9/docs/source/nodes/retrieval/hybrid_dbsf.md +0 -131
  35. autorag-0.2.9/docs/source/nodes/retrieval/hybrid_rrf.md +0 -40
  36. autorag-0.2.9/docs/source/nodes/retrieval/hybrid_rsf.md +0 -134
  37. autorag-0.2.9/docs/source/optimization/sample_full_config.yaml +0 -78
  38. autorag-0.2.9/sample_config/compact_openai.yaml +0 -65
  39. autorag-0.2.9/tests/autorag/nodes/retrieval/test_hybrid_cc.py +0 -33
  40. autorag-0.2.9/tests/autorag/nodes/retrieval/test_hybrid_dbsf.py +0 -20
  41. autorag-0.2.9/tests/autorag/nodes/retrieval/test_hybrid_rsf.py +0 -20
  42. {autorag-0.2.9 → autorag-0.2.10}/.github/FUNDING.yml +0 -0
  43. {autorag-0.2.9 → autorag-0.2.10}/.github/dependabot.yml +0 -0
  44. {autorag-0.2.9 → autorag-0.2.10}/.github/workflows/publish.yml +0 -0
  45. {autorag-0.2.9 → autorag-0.2.10}/.github/workflows/sphinx.yml +0 -0
  46. {autorag-0.2.9 → autorag-0.2.10}/.github/workflows/test.yml +0 -0
  47. {autorag-0.2.9 → autorag-0.2.10}/.gitignore +0 -0
  48. {autorag-0.2.9 → autorag-0.2.10}/AutoRAG.egg-info/dependency_links.txt +0 -0
  49. {autorag-0.2.9 → autorag-0.2.10}/AutoRAG.egg-info/entry_points.txt +0 -0
  50. {autorag-0.2.9 → autorag-0.2.10}/AutoRAG.egg-info/requires.txt +0 -0
  51. {autorag-0.2.9 → autorag-0.2.10}/AutoRAG.egg-info/top_level.txt +0 -0
  52. {autorag-0.2.9 → autorag-0.2.10}/CODE_OF_CONDUCT.md +0 -0
  53. {autorag-0.2.9 → autorag-0.2.10}/CONTRIBUTING.md +0 -0
  54. {autorag-0.2.9 → autorag-0.2.10}/LICENSE +0 -0
  55. {autorag-0.2.9 → autorag-0.2.10}/README.md +0 -0
  56. {autorag-0.2.9 → autorag-0.2.10}/autorag/__init__.py +0 -0
  57. {autorag-0.2.9 → autorag-0.2.10}/autorag/cli.py +0 -0
  58. {autorag-0.2.9 → autorag-0.2.10}/autorag/dashboard.py +0 -0
  59. {autorag-0.2.9 → autorag-0.2.10}/autorag/data/__init__.py +0 -0
  60. {autorag-0.2.9 → autorag-0.2.10}/autorag/data/corpus/__init__.py +0 -0
  61. {autorag-0.2.9 → autorag-0.2.10}/autorag/data/corpus/langchain.py +0 -0
  62. {autorag-0.2.9 → autorag-0.2.10}/autorag/data/corpus/llama_index.py +0 -0
  63. {autorag-0.2.9 → autorag-0.2.10}/autorag/data/qacreation/__init__.py +0 -0
  64. {autorag-0.2.9 → autorag-0.2.10}/autorag/data/qacreation/base.py +0 -0
  65. {autorag-0.2.9 → autorag-0.2.10}/autorag/data/qacreation/llama_index.py +0 -0
  66. {autorag-0.2.9 → autorag-0.2.10}/autorag/data/qacreation/llama_index_default_prompt.txt +0 -0
  67. {autorag-0.2.9 → autorag-0.2.10}/autorag/data/qacreation/ragas.py +0 -0
  68. {autorag-0.2.9 → autorag-0.2.10}/autorag/data/qacreation/simple.py +0 -0
  69. {autorag-0.2.9 → autorag-0.2.10}/autorag/data/utils/__init__.py +0 -0
  70. {autorag-0.2.9 → autorag-0.2.10}/autorag/data/utils/util.py +0 -0
  71. {autorag-0.2.9 → autorag-0.2.10}/autorag/deploy.py +0 -0
  72. {autorag-0.2.9 → autorag-0.2.10}/autorag/evaluation/__init__.py +0 -0
  73. {autorag-0.2.9 → autorag-0.2.10}/autorag/evaluation/generation.py +0 -0
  74. {autorag-0.2.9 → autorag-0.2.10}/autorag/evaluation/metric/__init__.py +0 -0
  75. {autorag-0.2.9 → autorag-0.2.10}/autorag/evaluation/metric/g_eval_prompts/coh_detailed.txt +0 -0
  76. {autorag-0.2.9 → autorag-0.2.10}/autorag/evaluation/metric/g_eval_prompts/con_detailed.txt +0 -0
  77. {autorag-0.2.9 → autorag-0.2.10}/autorag/evaluation/metric/g_eval_prompts/flu_detailed.txt +0 -0
  78. {autorag-0.2.9 → autorag-0.2.10}/autorag/evaluation/metric/g_eval_prompts/rel_detailed.txt +0 -0
  79. {autorag-0.2.9 → autorag-0.2.10}/autorag/evaluation/metric/generation.py +0 -0
  80. {autorag-0.2.9 → autorag-0.2.10}/autorag/evaluation/metric/retrieval.py +0 -0
  81. {autorag-0.2.9 → autorag-0.2.10}/autorag/evaluation/metric/retrieval_contents.py +0 -0
  82. {autorag-0.2.9 → autorag-0.2.10}/autorag/evaluation/metric/util.py +0 -0
  83. {autorag-0.2.9 → autorag-0.2.10}/autorag/evaluation/retrieval.py +0 -0
  84. {autorag-0.2.9 → autorag-0.2.10}/autorag/evaluation/retrieval_contents.py +0 -0
  85. {autorag-0.2.9 → autorag-0.2.10}/autorag/evaluation/util.py +0 -0
  86. {autorag-0.2.9 → autorag-0.2.10}/autorag/evaluator.py +0 -0
  87. {autorag-0.2.9 → autorag-0.2.10}/autorag/node_line.py +0 -0
  88. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/__init__.py +0 -0
  89. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/generator/__init__.py +0 -0
  90. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/generator/base.py +0 -0
  91. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/generator/llama_index_llm.py +0 -0
  92. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/generator/openai_llm.py +0 -0
  93. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/generator/run.py +0 -0
  94. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/generator/vllm.py +0 -0
  95. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passageaugmenter/__init__.py +0 -0
  96. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passageaugmenter/base.py +0 -0
  97. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passageaugmenter/pass_passage_augmenter.py +0 -0
  98. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passageaugmenter/prev_next_augmenter.py +0 -0
  99. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passageaugmenter/run.py +0 -0
  100. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagecompressor/__init__.py +0 -0
  101. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagecompressor/base.py +0 -0
  102. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagecompressor/longllmlingua.py +0 -0
  103. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagecompressor/pass_compressor.py +0 -0
  104. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagecompressor/refine.py +0 -0
  105. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagecompressor/run.py +0 -0
  106. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagecompressor/tree_summarize.py +0 -0
  107. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagefilter/__init__.py +0 -0
  108. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagefilter/base.py +0 -0
  109. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagefilter/pass_passage_filter.py +0 -0
  110. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagefilter/percentile_cutoff.py +0 -0
  111. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagefilter/recency.py +0 -0
  112. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagefilter/run.py +0 -0
  113. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagefilter/similarity_percentile_cutoff.py +0 -0
  114. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagefilter/similarity_threshold_cutoff.py +0 -0
  115. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagefilter/threshold_cutoff.py +0 -0
  116. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagereranker/__init__.py +0 -0
  117. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagereranker/base.py +0 -0
  118. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagereranker/cohere.py +0 -0
  119. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagereranker/colbert.py +0 -0
  120. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagereranker/flag_embedding.py +0 -0
  121. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagereranker/flag_embedding_llm.py +0 -0
  122. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagereranker/jina.py +0 -0
  123. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagereranker/koreranker.py +0 -0
  124. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagereranker/monot5.py +0 -0
  125. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagereranker/pass_reranker.py +0 -0
  126. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagereranker/rankgpt.py +0 -0
  127. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagereranker/run.py +0 -0
  128. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagereranker/sentence_transformer.py +0 -0
  129. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagereranker/tart/__init__.py +0 -0
  130. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagereranker/tart/modeling_enc_t5.py +0 -0
  131. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagereranker/tart/tart.py +0 -0
  132. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagereranker/tart/tokenization_enc_t5.py +0 -0
  133. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagereranker/time_reranker.py +0 -0
  134. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/passagereranker/upr.py +0 -0
  135. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/promptmaker/__init__.py +0 -0
  136. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/promptmaker/base.py +0 -0
  137. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/promptmaker/fstring.py +0 -0
  138. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/promptmaker/long_context_reorder.py +0 -0
  139. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/promptmaker/run.py +0 -0
  140. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/promptmaker/window_replacement.py +0 -0
  141. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/queryexpansion/__init__.py +0 -0
  142. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/queryexpansion/base.py +0 -0
  143. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/queryexpansion/hyde.py +0 -0
  144. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/queryexpansion/multi_query_expansion.py +0 -0
  145. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/queryexpansion/pass_query_expansion.py +0 -0
  146. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/queryexpansion/query_decompose.py +0 -0
  147. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/queryexpansion/run.py +0 -0
  148. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/retrieval/bm25.py +0 -0
  149. {autorag-0.2.9 → autorag-0.2.10}/autorag/nodes/retrieval/vectordb.py +0 -0
  150. {autorag-0.2.9 → autorag-0.2.10}/autorag/schema/__init__.py +0 -0
  151. {autorag-0.2.9 → autorag-0.2.10}/autorag/schema/module.py +0 -0
  152. {autorag-0.2.9 → autorag-0.2.10}/autorag/schema/node.py +0 -0
  153. {autorag-0.2.9 → autorag-0.2.10}/autorag/strategy.py +0 -0
  154. {autorag-0.2.9 → autorag-0.2.10}/autorag/utils/__init__.py +0 -0
  155. {autorag-0.2.9 → autorag-0.2.10}/autorag/utils/preprocess.py +0 -0
  156. {autorag-0.2.9 → autorag-0.2.10}/autorag/utils/util.py +0 -0
  157. {autorag-0.2.9 → autorag-0.2.10}/autorag/web.py +0 -0
  158. {autorag-0.2.9 → autorag-0.2.10}/docs/Makefile +0 -0
  159. {autorag-0.2.9 → autorag-0.2.10}/docs/make.bat +0 -0
  160. {autorag-0.2.9 → autorag-0.2.10}/docs/requirements.txt +0 -0
  161. {autorag-0.2.9 → autorag-0.2.10}/docs/source/CNAME +0 -0
  162. {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/data_creation.png +0 -0
  163. {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/data_folder.png +0 -0
  164. {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/dcg.png +0 -0
  165. {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/f1_score.png +0 -0
  166. {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/map.png +0 -0
  167. {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/mrr.png +0 -0
  168. {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/ndcg.png +0 -0
  169. {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/ndcg_formula.png +0 -0
  170. {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/node_folder.png +0 -0
  171. {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/node_line_folder.png +0 -0
  172. {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/node_line_summary.png +0 -0
  173. {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/node_lines.png +0 -0
  174. {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/node_summary.png +0 -0
  175. {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/normal_distribution.png +0 -0
  176. {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/project_folder_example.png +0 -0
  177. {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/project_folders.png +0 -0
  178. {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/resources_folder.png +0 -0
  179. {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/roadmap/RAG_paradigms.png +0 -0
  180. {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/roadmap/advanced_RAG.png +0 -0
  181. {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/roadmap/cycle.png +0 -0
  182. {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/roadmap/merger.png +0 -0
  183. {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/roadmap/node_line_modular.png +0 -0
  184. {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/roadmap/policy.png +0 -0
  185. {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/samsung_sundae.jpeg +0 -0
  186. {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/score_fusion.png +0 -0
  187. {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/trial_folder.png +0 -0
  188. {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/trial_json.png +0 -0
  189. {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/trial_summary.png +0 -0
  190. {autorag-0.2.9 → autorag-0.2.10}/docs/source/_static/web_interface.png +0 -0
  191. {autorag-0.2.9 → autorag-0.2.10}/docs/source/api_spec/autorag.data.corpus.rst +0 -0
  192. {autorag-0.2.9 → autorag-0.2.10}/docs/source/api_spec/autorag.data.qacreation.rst +0 -0
  193. {autorag-0.2.9 → autorag-0.2.10}/docs/source/api_spec/autorag.data.rst +0 -0
  194. {autorag-0.2.9 → autorag-0.2.10}/docs/source/api_spec/autorag.data.utils.rst +0 -0
  195. {autorag-0.2.9 → autorag-0.2.10}/docs/source/api_spec/autorag.evaluation.metric.rst +0 -0
  196. {autorag-0.2.9 → autorag-0.2.10}/docs/source/api_spec/autorag.evaluation.rst +0 -0
  197. {autorag-0.2.9 → autorag-0.2.10}/docs/source/api_spec/autorag.nodes.generator.rst +0 -0
  198. {autorag-0.2.9 → autorag-0.2.10}/docs/source/api_spec/autorag.nodes.passageaugmenter.rst +0 -0
  199. {autorag-0.2.9 → autorag-0.2.10}/docs/source/api_spec/autorag.nodes.passagecompressor.rst +0 -0
  200. {autorag-0.2.9 → autorag-0.2.10}/docs/source/api_spec/autorag.nodes.passagefilter.rst +0 -0
  201. {autorag-0.2.9 → autorag-0.2.10}/docs/source/api_spec/autorag.nodes.passagereranker.rst +0 -0
  202. {autorag-0.2.9 → autorag-0.2.10}/docs/source/api_spec/autorag.nodes.passagereranker.tart.rst +0 -0
  203. {autorag-0.2.9 → autorag-0.2.10}/docs/source/api_spec/autorag.nodes.promptmaker.rst +0 -0
  204. {autorag-0.2.9 → autorag-0.2.10}/docs/source/api_spec/autorag.nodes.queryexpansion.rst +0 -0
  205. {autorag-0.2.9 → autorag-0.2.10}/docs/source/api_spec/autorag.nodes.rst +0 -0
  206. {autorag-0.2.9 → autorag-0.2.10}/docs/source/api_spec/autorag.rst +0 -0
  207. {autorag-0.2.9 → autorag-0.2.10}/docs/source/api_spec/autorag.schema.rst +0 -0
  208. {autorag-0.2.9 → autorag-0.2.10}/docs/source/api_spec/autorag.utils.rst +0 -0
  209. {autorag-0.2.9 → autorag-0.2.10}/docs/source/api_spec/modules.rst +0 -0
  210. {autorag-0.2.9 → autorag-0.2.10}/docs/source/conf.py +0 -0
  211. {autorag-0.2.9 → autorag-0.2.10}/docs/source/data_creation/data_format.md +0 -0
  212. {autorag-0.2.9 → autorag-0.2.10}/docs/source/data_creation/ragas.md +0 -0
  213. {autorag-0.2.9 → autorag-0.2.10}/docs/source/data_creation/tutorial.md +0 -0
  214. {autorag-0.2.9 → autorag-0.2.10}/docs/source/deploy/api_endpoint.md +0 -0
  215. {autorag-0.2.9 → autorag-0.2.10}/docs/source/deploy/web.md +0 -0
  216. {autorag-0.2.9 → autorag-0.2.10}/docs/source/evaluate_metrics/retrieval.md +0 -0
  217. {autorag-0.2.9 → autorag-0.2.10}/docs/source/evaluate_metrics/retrieval_contents.md +0 -0
  218. {autorag-0.2.9 → autorag-0.2.10}/docs/source/index.rst +0 -0
  219. {autorag-0.2.9 → autorag-0.2.10}/docs/source/install.md +0 -0
  220. {autorag-0.2.9 → autorag-0.2.10}/docs/source/local_model.md +0 -0
  221. {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/generator/generator.md +0 -0
  222. {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/generator/llama_index_llm.md +0 -0
  223. {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/generator/openai_llm.md +0 -0
  224. {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/generator/vllm.md +0 -0
  225. {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/index.md +0 -0
  226. {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/passage_augmenter/passage_augmenter.md +0 -0
  227. {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/passage_augmenter/prev_next_augmenter.md +0 -0
  228. {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/passage_compressor/longllmlingua.md +0 -0
  229. {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/passage_compressor/passage_compressor.md +0 -0
  230. {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/passage_compressor/refine.md +0 -0
  231. {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/passage_compressor/tree_summarize.md +0 -0
  232. {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/passage_filter/passage_filter.md +0 -0
  233. {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/passage_filter/percentile_cutoff.md +0 -0
  234. {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/passage_filter/recency_filter.md +0 -0
  235. {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/passage_filter/similarity_percentile_cutoff.md +0 -0
  236. {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/passage_filter/similarity_threshold_cutoff.md +0 -0
  237. {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/passage_filter/threshold_cutoff.md +0 -0
  238. {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/passage_reranker/cohere.md +0 -0
  239. {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/passage_reranker/colbert.md +0 -0
  240. {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/passage_reranker/flag_embedding_llm_reranker.md +0 -0
  241. {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/passage_reranker/flag_embedding_reranker.md +0 -0
  242. {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/passage_reranker/jina_reranker.md +0 -0
  243. {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/passage_reranker/koreranker.md +0 -0
  244. {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/passage_reranker/monot5.md +0 -0
  245. {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/passage_reranker/passage_reranker.md +0 -0
  246. {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/passage_reranker/rankgpt.md +0 -0
  247. {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/passage_reranker/sentence_transformer_reranker.md +0 -0
  248. {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/passage_reranker/tart.md +0 -0
  249. {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/passage_reranker/time_reranker.md +0 -0
  250. {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/passage_reranker/upr.md +0 -0
  251. {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/prompt_maker/fstring.md +0 -0
  252. {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/prompt_maker/long_context_reorder.md +0 -0
  253. {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/prompt_maker/prompt_maker.md +0 -0
  254. {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/prompt_maker/window_replacement.md +0 -0
  255. {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/query_expansion/hyde.md +0 -0
  256. {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/query_expansion/multi_query_expansion.md +0 -0
  257. {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/query_expansion/query_decompose.md +0 -0
  258. {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/query_expansion/query_expansion.md +0 -0
  259. {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/retrieval/bm25.md +0 -0
  260. {autorag-0.2.9 → autorag-0.2.10}/docs/source/nodes/retrieval/vectordb.md +0 -0
  261. {autorag-0.2.9 → autorag-0.2.10}/docs/source/optimization/custom_config.md +0 -0
  262. {autorag-0.2.9 → autorag-0.2.10}/docs/source/optimization/folder_structure.md +0 -0
  263. {autorag-0.2.9 → autorag-0.2.10}/docs/source/optimization/optimization.md +0 -0
  264. {autorag-0.2.9 → autorag-0.2.10}/docs/source/optimization/strategies.md +0 -0
  265. {autorag-0.2.9 → autorag-0.2.10}/docs/source/roadmap/modular_rag.md +0 -0
  266. {autorag-0.2.9 → autorag-0.2.10}/docs/source/structure.md +0 -0
  267. {autorag-0.2.9 → autorag-0.2.10}/docs/source/troubleshooting.md +0 -0
  268. {autorag-0.2.9 → autorag-0.2.10}/docs/source/tutorial.md +0 -0
  269. {autorag-0.2.9 → autorag-0.2.10}/pyproject.toml +0 -0
  270. {autorag-0.2.9 → autorag-0.2.10}/requirements.txt +0 -0
  271. {autorag-0.2.9 → autorag-0.2.10}/sample_config/config_korean.yaml +0 -0
  272. {autorag-0.2.9 → autorag-0.2.10}/sample_config/extracted_sample.yaml +0 -0
  273. {autorag-0.2.9 → autorag-0.2.10}/sample_config/simple_local.yaml +0 -0
  274. {autorag-0.2.9 → autorag-0.2.10}/sample_config/simple_openai.yaml +0 -0
  275. {autorag-0.2.9 → autorag-0.2.10}/sample_dataset/README.md +0 -0
  276. {autorag-0.2.9 → autorag-0.2.10}/sample_dataset/eli5/load_eli5_dataset.py +0 -0
  277. {autorag-0.2.9 → autorag-0.2.10}/sample_dataset/hotpotqa/load_hotpotqa_dataset.py +0 -0
  278. {autorag-0.2.9 → autorag-0.2.10}/sample_dataset/msmarco/load_msmarco_dataset.py +0 -0
  279. {autorag-0.2.9 → autorag-0.2.10}/sample_dataset/triviaqa/load_triviaqa_dataset.py +0 -0
  280. {autorag-0.2.9 → autorag-0.2.10}/setup.cfg +0 -0
  281. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/data/corpus/test_base.py +0 -0
  282. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/data/corpus/test_langchain.py +0 -0
  283. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/data/corpus/test_llama_index_corpus.py +0 -0
  284. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/data/qacreation/test_base_qacreation.py +0 -0
  285. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/data/qacreation/test_llama_index_qacreation.py +0 -0
  286. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/data/qacreation/test_ragas_qa_creation.py +0 -0
  287. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/data/qacreation/test_simple.py +0 -0
  288. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/evaluate/metric/test_generation_metric.py +0 -0
  289. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/evaluate/metric/test_retrieval_contents_metric.py +0 -0
  290. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/evaluate/metric/test_retrieval_metric.py +0 -0
  291. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/evaluate/test_evaluate_util.py +0 -0
  292. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/evaluate/test_generation_evaluate.py +0 -0
  293. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/evaluate/test_retrieval_contents_evaluate.py +0 -0
  294. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/evaluate/test_retrieval_evaluate.py +0 -0
  295. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/generator/test_generator_base.py +0 -0
  296. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/generator/test_llama_index_llm.py +0 -0
  297. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/generator/test_openai.py +0 -0
  298. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/generator/test_run_generator_node.py +0 -0
  299. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/generator/test_vllm.py +0 -0
  300. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passageaugmenter/test_base_passage_augmenter.py +0 -0
  301. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passageaugmenter/test_pass_passage_augmenter.py +0 -0
  302. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passageaugmenter/test_prev_next_augmenter.py +0 -0
  303. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passageaugmenter/test_run_passage_augmenter.py +0 -0
  304. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagecompressor/test_base_passage_compressor.py +0 -0
  305. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagecompressor/test_longllmlingua.py +0 -0
  306. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagecompressor/test_pass_compressor.py +0 -0
  307. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagecompressor/test_refine.py +0 -0
  308. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagecompressor/test_run_passage_compressor_node.py +0 -0
  309. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagecompressor/test_tree_summarize.py +0 -0
  310. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagefilter/test_pass_passage_filter.py +0 -0
  311. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagefilter/test_passage_filter_base.py +0 -0
  312. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagefilter/test_passage_filter_run.py +0 -0
  313. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagefilter/test_percentile_cutoff.py +0 -0
  314. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagefilter/test_recency_filter.py +0 -0
  315. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagefilter/test_similarity_percentile_cutoff.py +0 -0
  316. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagefilter/test_similarity_threshold_cutoff.py +0 -0
  317. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagefilter/test_threshold_cutoff.py +0 -0
  318. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagereranker/test_cohere_reranker.py +0 -0
  319. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagereranker/test_colbert_reranker.py +0 -0
  320. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagereranker/test_flag_embedding.py +0 -0
  321. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagereranker/test_flag_embedding_llm.py +0 -0
  322. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagereranker/test_jina_reranker.py +0 -0
  323. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagereranker/test_koreranker.py +0 -0
  324. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagereranker/test_monot5.py +0 -0
  325. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagereranker/test_pass_reranker.py +0 -0
  326. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagereranker/test_passage_reranker_base.py +0 -0
  327. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagereranker/test_passage_reranker_run.py +0 -0
  328. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagereranker/test_rankgpt.py +0 -0
  329. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagereranker/test_sentence_transformer.py +0 -0
  330. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagereranker/test_tart.py +0 -0
  331. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagereranker/test_time_reranker.py +0 -0
  332. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/passagereranker/test_upr.py +0 -0
  333. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/promptmaker/test_fstring.py +0 -0
  334. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/promptmaker/test_long_context_reorder.py +0 -0
  335. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/promptmaker/test_prompt_maker_base.py +0 -0
  336. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/promptmaker/test_prompt_maker_run.py +0 -0
  337. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/promptmaker/test_window_replacement.py +0 -0
  338. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/queryexpansion/test_hyde.py +0 -0
  339. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/queryexpansion/test_multi_query_expansion.py +0 -0
  340. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/queryexpansion/test_pass_query_expansion.py +0 -0
  341. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/queryexpansion/test_query_decompose.py +0 -0
  342. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/queryexpansion/test_query_expansion_base.py +0 -0
  343. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/queryexpansion/test_query_expansion_run.py +0 -0
  344. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/retrieval/test_bm25.py +0 -0
  345. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/retrieval/test_retrieval_base.py +0 -0
  346. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/nodes/retrieval/test_vectordb.py +0 -0
  347. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/schema/test_module_schema.py +0 -0
  348. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/schema/test_node_schema.py +0 -0
  349. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/test_cli.py +0 -0
  350. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/test_dashboard.py +0 -0
  351. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/test_deploy.py +0 -0
  352. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/test_strategy.py +0 -0
  353. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/test_support.py +0 -0
  354. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/test_web.py +0 -0
  355. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/utils/test_preprocess.py +0 -0
  356. {autorag-0.2.9 → autorag-0.2.10}/tests/autorag/utils/test_util.py +0 -0
  357. {autorag-0.2.9 → autorag-0.2.10}/tests/conftest.py +0 -0
  358. {autorag-0.2.9 → autorag-0.2.10}/tests/delete_tests.py +0 -0
  359. {autorag-0.2.9 → autorag-0.2.10}/tests/mock.py +0 -0
  360. {autorag-0.2.9 → autorag-0.2.10}/tests/requirements.txt +0 -0
  361. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/README.md +0 -0
  362. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/corpus_data_sample.parquet +0 -0
  363. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/data_creation/raw_dir/sample1.txt +0 -0
  364. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/data_creation/raw_dir/sample2.txt +0 -0
  365. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/data_creation/raw_dir/sample3.txt +0 -0
  366. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/qa_data_sample.parquet +0 -0
  367. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/qa_gen_prompts/prompt1.txt +0 -0
  368. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/qa_gen_prompts/prompt2.txt +0 -0
  369. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/qa_gen_prompts/prompt3.txt +0 -0
  370. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/qa_test_data_sample.parquet +0 -0
  371. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/config.yaml +0 -0
  372. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/post_retrieve_node_line/generator/0.parquet +0 -0
  373. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/post_retrieve_node_line/generator/1.parquet +0 -0
  374. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/post_retrieve_node_line/generator/2.parquet +0 -0
  375. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/post_retrieve_node_line/generator/3.parquet +0 -0
  376. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/post_retrieve_node_line/generator/4.parquet +0 -0
  377. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/post_retrieve_node_line/generator/5.parquet +0 -0
  378. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/post_retrieve_node_line/generator/best_5.parquet +0 -0
  379. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/post_retrieve_node_line/generator/summary.csv +0 -0
  380. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/post_retrieve_node_line/prompt_maker/0.parquet +0 -0
  381. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/post_retrieve_node_line/prompt_maker/1.parquet +0 -0
  382. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/post_retrieve_node_line/prompt_maker/best_0.parquet +0 -0
  383. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/post_retrieve_node_line/prompt_maker/summary.csv +0 -0
  384. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/post_retrieve_node_line/summary.csv +0 -0
  385. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/pre_retrieve_node_line/query_expansion/0.parquet +0 -0
  386. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/pre_retrieve_node_line/query_expansion/1.parquet +0 -0
  387. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/pre_retrieve_node_line/query_expansion/2.parquet +0 -0
  388. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/pre_retrieve_node_line/query_expansion/best_0.parquet +0 -0
  389. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/pre_retrieve_node_line/query_expansion/summary.csv +0 -0
  390. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/pre_retrieve_node_line/summary.csv +0 -0
  391. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/retrieve_node_line/passage_compressor/0.parquet +0 -0
  392. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/retrieve_node_line/passage_compressor/best_0.parquet +0 -0
  393. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/retrieve_node_line/passage_compressor/summary.csv +0 -0
  394. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/retrieve_node_line/passage_reranker/0.parquet +0 -0
  395. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/retrieve_node_line/passage_reranker/1.parquet +0 -0
  396. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/retrieve_node_line/passage_reranker/best_0.parquet +0 -0
  397. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/retrieve_node_line/passage_reranker/summary.csv +0 -0
  398. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/retrieve_node_line/retrieval/0.parquet +0 -0
  399. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/retrieve_node_line/retrieval/1.parquet +0 -0
  400. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/retrieve_node_line/retrieval/best_0.parquet +0 -0
  401. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/retrieve_node_line/retrieval/summary.csv +0 -0
  402. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/retrieve_node_line/summary.csv +0 -0
  403. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/0/summary.csv +0 -0
  404. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/1/config.yaml +0 -0
  405. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/1/pre_retrieve_node_line/query_expansion/0.parquet +0 -0
  406. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/1/pre_retrieve_node_line/query_expansion/1.parquet +0 -0
  407. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/1/pre_retrieve_node_line/query_expansion/2.parquet +0 -0
  408. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/1/pre_retrieve_node_line/query_expansion/best_0.parquet +0 -0
  409. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/1/pre_retrieve_node_line/query_expansion/summary.csv +0 -0
  410. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/1/pre_retrieve_node_line/summary.csv +0 -0
  411. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/1/retrieve_node_line/passage_filter/0.parquet +0 -0
  412. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/1/retrieve_node_line/passage_reranker/0.parquet +0 -0
  413. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/1/retrieve_node_line/passage_reranker/1.parquet +0 -0
  414. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/1/retrieve_node_line/passage_reranker/best_0.parquet +0 -0
  415. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/1/retrieve_node_line/passage_reranker/summary.csv +0 -0
  416. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/1/retrieve_node_line/retrieval/0.parquet +0 -0
  417. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/1/retrieve_node_line/retrieval/1.parquet +0 -0
  418. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/1/retrieve_node_line/retrieval/best_0.parquet +0 -0
  419. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/1/retrieve_node_line/retrieval/summary.csv +0 -0
  420. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/2/config.yaml +0 -0
  421. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/2/post_retrieve_node_line/prompt_maker/0.parquet +0 -0
  422. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/2/pre_retrieve_node_line/query_expansion/0.parquet +0 -0
  423. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/2/pre_retrieve_node_line/query_expansion/1.parquet +0 -0
  424. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/2/pre_retrieve_node_line/query_expansion/2.parquet +0 -0
  425. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/2/pre_retrieve_node_line/query_expansion/best_0.parquet +0 -0
  426. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/2/pre_retrieve_node_line/query_expansion/summary.csv +0 -0
  427. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/2/pre_retrieve_node_line/summary.csv +0 -0
  428. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/2/retrieve_node_line/passage_compressor/0.parquet +0 -0
  429. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/2/retrieve_node_line/passage_compressor/best_0.parquet +0 -0
  430. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/2/retrieve_node_line/passage_compressor/summary.csv +0 -0
  431. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/2/retrieve_node_line/passage_reranker/0.parquet +0 -0
  432. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/2/retrieve_node_line/passage_reranker/1.parquet +0 -0
  433. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/2/retrieve_node_line/passage_reranker/best_0.parquet +0 -0
  434. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/2/retrieve_node_line/passage_reranker/summary.csv +0 -0
  435. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/2/retrieve_node_line/retrieval/0.parquet +0 -0
  436. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/2/retrieve_node_line/retrieval/1.parquet +0 -0
  437. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/2/retrieve_node_line/retrieval/best_0.parquet +0 -0
  438. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/2/retrieve_node_line/retrieval/summary.csv +0 -0
  439. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/2/retrieve_node_line/summary.csv +0 -0
  440. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/best.yaml +0 -0
  441. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/data/corpus.parquet +0 -0
  442. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/data/qa.parquet +0 -0
  443. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/resources/bm25_porter_stemmer.pkl +0 -0
  444. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/resources/chroma/6595d304-5270-4d9f-a267-c191366804ce/data_level0.bin +0 -0
  445. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/resources/chroma/6595d304-5270-4d9f-a267-c191366804ce/header.bin +0 -0
  446. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/resources/chroma/6595d304-5270-4d9f-a267-c191366804ce/length.bin +0 -0
  447. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/resources/chroma/6595d304-5270-4d9f-a267-c191366804ce/link_lists.bin +0 -0
  448. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/resources/chroma/chroma.sqlite3 +0 -0
  449. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/result_project/trial.json +0 -0
  450. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/sample_contents_nqa.csv +0 -0
  451. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/sample_project/data/corpus.parquet +0 -0
  452. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/sample_project/data/qa.parquet +0 -0
  453. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/sample_project/resources/bm25_gpt2.pkl +0 -0
  454. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/sample_project/resources/bm25_porter_stemmer.pkl +0 -0
  455. {autorag-0.2.9 → autorag-0.2.10}/tests/resources/test_bm25_retrieval.pkl +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: AutoRAG
3
- Version: 0.2.9
3
+ Version: 0.2.10
4
4
  Summary: Automatically Evaluate RAG pipelines with your own data. Find optimal structure for new RAG product.
5
5
  Author-email: Marker-Inc <vkehfdl1@gmail.com>
6
6
  License: Apache License
@@ -116,9 +116,7 @@ autorag/nodes/retrieval/__init__.py
116
116
  autorag/nodes/retrieval/base.py
117
117
  autorag/nodes/retrieval/bm25.py
118
118
  autorag/nodes/retrieval/hybrid_cc.py
119
- autorag/nodes/retrieval/hybrid_dbsf.py
120
119
  autorag/nodes/retrieval/hybrid_rrf.py
121
- autorag/nodes/retrieval/hybrid_rsf.py
122
120
  autorag/nodes/retrieval/run.py
123
121
  autorag/nodes/retrieval/vectordb.py
124
122
  autorag/schema/__init__.py
@@ -235,15 +233,12 @@ docs/source/nodes/query_expansion/query_decompose.md
235
233
  docs/source/nodes/query_expansion/query_expansion.md
236
234
  docs/source/nodes/retrieval/bm25.md
237
235
  docs/source/nodes/retrieval/hybrid_cc.md
238
- docs/source/nodes/retrieval/hybrid_dbsf.md
239
236
  docs/source/nodes/retrieval/hybrid_rrf.md
240
- docs/source/nodes/retrieval/hybrid_rsf.md
241
237
  docs/source/nodes/retrieval/retrieval.md
242
238
  docs/source/nodes/retrieval/vectordb.md
243
239
  docs/source/optimization/custom_config.md
244
240
  docs/source/optimization/folder_structure.md
245
241
  docs/source/optimization/optimization.md
246
- docs/source/optimization/sample_full_config.yaml
247
242
  docs/source/optimization/strategies.md
248
243
  docs/source/roadmap/modular_rag.md
249
244
  sample_config/compact_local.yaml
@@ -336,9 +331,7 @@ tests/autorag/nodes/queryexpansion/test_query_expansion_run.py
336
331
  tests/autorag/nodes/retrieval/test_bm25.py
337
332
  tests/autorag/nodes/retrieval/test_hybrid_base.py
338
333
  tests/autorag/nodes/retrieval/test_hybrid_cc.py
339
- tests/autorag/nodes/retrieval/test_hybrid_dbsf.py
340
334
  tests/autorag/nodes/retrieval/test_hybrid_rrf.py
341
- tests/autorag/nodes/retrieval/test_hybrid_rsf.py
342
335
  tests/autorag/nodes/retrieval/test_retrieval_base.py
343
336
  tests/autorag/nodes/retrieval/test_run_retrieval_node.py
344
337
  tests/autorag/nodes/retrieval/test_vectordb.py
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: AutoRAG
3
- Version: 0.2.9
3
+ Version: 0.2.10
4
4
  Summary: Automatically Evaluate RAG pipelines with your own data. Find optimal structure for new RAG product.
5
5
  Author-email: Marker-Inc <vkehfdl1@gmail.com>
6
6
  License: Apache License
@@ -0,0 +1 @@
1
+ 0.2.10
@@ -1,7 +1,5 @@
1
1
  from .base import retrieval_node
2
2
  from .bm25 import bm25
3
3
  from .hybrid_cc import hybrid_cc
4
- from .hybrid_dbsf import hybrid_dbsf
5
4
  from .hybrid_rrf import hybrid_rrf
6
- from .hybrid_rsf import hybrid_rsf
7
5
  from .vectordb import vectordb
@@ -32,7 +32,6 @@ def retrieval_node(func):
32
32
  project_dir: Union[str, Path],
33
33
  previous_result: pd.DataFrame,
34
34
  **kwargs) -> Tuple[List[List[str]], List[List[str]], List[List[float]]]:
35
- logger.info(f"Running retrieval node - {func.__name__} module...")
36
35
  validate_qa_dataset(previous_result)
37
36
  resources_dir = os.path.join(project_dir, "resources")
38
37
  data_dir = os.path.join(project_dir, "data")
@@ -75,10 +74,10 @@ def retrieval_node(func):
75
74
  del embedding_model
76
75
  if torch.cuda.is_available():
77
76
  torch.cuda.empty_cache()
78
- elif func.__name__ in ["hybrid_rrf", "hybrid_cc", "hybrid_rsf", "hybrid_dbsf"]:
79
- if 'ids' in kwargs and 'scores' in kwargs:
77
+ elif func.__name__ in ["hybrid_rrf", "hybrid_cc"]:
78
+ if 'ids' in kwargs and 'scores' in kwargs: # ordinary run_evaluate
80
79
  ids, scores = func(**kwargs)
81
- else:
80
+ else: # => for Runner.run
82
81
  if not ('target_modules' in kwargs and 'target_module_params' in kwargs):
83
82
  raise ValueError(
84
83
  f"If there are no ids and scores specified, target_modules and target_module_params must be specified for using {func.__name__}.")
@@ -0,0 +1,137 @@
1
+ from typing import Tuple, List
2
+
3
+ import numpy as np
4
+ import pandas as pd
5
+
6
+ from autorag.nodes.retrieval import retrieval_node
7
+
8
+
9
+ def normalize_mm(scores: List[str], fixed_min_value: float = 0):
10
+ arr = np.array(scores)
11
+ max_value = np.max(arr)
12
+ min_value = np.min(arr)
13
+ norm_score = (arr - min_value) / (max_value - min_value)
14
+ return norm_score
15
+
16
+
17
+ def normalize_tmm(scores: List[str], fixed_min_value: float):
18
+ arr = np.array(scores)
19
+ max_value = np.max(arr)
20
+ norm_score = (arr - fixed_min_value) / (max_value - fixed_min_value)
21
+ return norm_score
22
+
23
+
24
+ def normalize_z(scores: List[str], fixed_min_value: float = 0):
25
+ arr = np.array(scores)
26
+ mean_value = np.mean(arr)
27
+ std_value = np.std(arr)
28
+ norm_score = (arr - mean_value) / std_value
29
+ return norm_score
30
+
31
+
32
+ def normalize_dbsf(scores: List[str], fixed_min_value: float = 0):
33
+ arr = np.array(scores)
34
+ mean_value = np.mean(arr)
35
+ std_value = np.std(arr)
36
+ min_value = mean_value - 3 * std_value
37
+ max_value = mean_value + 3 * std_value
38
+ norm_score = (arr - min_value) / (max_value - min_value)
39
+ return norm_score
40
+
41
+
42
+ normalize_method_dict = {
43
+ 'mm': normalize_mm,
44
+ 'tmm': normalize_tmm,
45
+ 'z': normalize_z,
46
+ 'dbsf': normalize_dbsf,
47
+ }
48
+
49
+
50
+ @retrieval_node
51
+ def hybrid_cc(
52
+ ids: Tuple,
53
+ scores: Tuple,
54
+ top_k: int,
55
+ weight: float,
56
+ normalize_method: str = 'mm',
57
+ semantic_theoretical_min_value: float = -1.0,
58
+ lexical_theoretical_min_value: float = 0.0,
59
+ ) -> Tuple[List[List[str]], List[List[float]]]:
60
+ """
61
+ Hybrid CC function.
62
+ CC (convex combination) is a method to fuse lexical and semantic retrieval results.
63
+ It is a method that first normalizes the scores of each retrieval result,
64
+ and then combines them with the given weights.
65
+ It is uniquer than other retrieval modules, because it does not really execute retrieval,
66
+ but just fuse the results of other retrieval functions.
67
+ So you have to run more than two retrieval modules before running this function.
68
+ And collect ids and scores result from each retrieval module.
69
+ Make it as tuple and input it to this function.
70
+
71
+ :param ids: The tuple of ids that you want to fuse.
72
+ The length of this must be the same as the length of scores.
73
+ The semantic retrieval ids must be the first index.
74
+ :param scores: The retrieve scores that you want to fuse.
75
+ The length of this must be the same as the length of ids.
76
+ The semantic retrieval scores must be the first index.
77
+ :param top_k: The number of passages to be retrieved.
78
+ :param normalize_method: The normalization method to use.
79
+ There are some normalization method that you can use at the hybrid cc method.
80
+ AutoRAG support following.
81
+ - `mm`: Min-max scaling
82
+ - `tmm`: Theoretical min-max scaling
83
+ - `z`: z-score normalization
84
+ - `dbsf`: 3-sigma normalization
85
+ :param weight: The weight value. If the weight is 1.0, it means the
86
+ weight to the semantic module will be 1.0 and weight to the lexical module will be 0.0.
87
+ :param semantic_theoretical_min_value: This value used by `tmm` normalization method. You can set the
88
+ theoretical minimum value by yourself. Default is -1.
89
+ :param lexical_theoretical_min_value: This value used by `tmm` normalization method. You can set the
90
+ theoretical minimum value by yourself. Default is 0.
91
+ :return: The tuple of ids and fused scores that fused by CC. Plus, the third element is selected weight value.
92
+ """
93
+ assert len(ids) == len(scores), "The length of ids and scores must be the same."
94
+ assert len(ids) > 1, "You must input more than one retrieval results."
95
+ assert top_k > 0, "top_k must be greater than 0."
96
+ assert weight >= 0, "The weight must be greater than 0."
97
+ assert weight <= 1, "The weight must be less than 1."
98
+
99
+ df = pd.DataFrame({
100
+ 'semantic_ids': ids[0],
101
+ 'lexical_ids': ids[1],
102
+ 'semantic_score': scores[0],
103
+ 'lexical_score': scores[1],
104
+ })
105
+
106
+ def cc_pure_apply(row):
107
+ return fuse_per_query(row['semantic_ids'], row['lexical_ids'],
108
+ row['semantic_score'], row['lexical_score'],
109
+ normalize_method=normalize_method,
110
+ weight=weight, top_k=top_k,
111
+ semantic_theoretical_min_value=semantic_theoretical_min_value,
112
+ lexical_theoretical_min_value=lexical_theoretical_min_value)
113
+
114
+ # fixed weight
115
+ df[['cc_id', 'cc_score']] = df.apply(lambda row: cc_pure_apply(row), axis=1,
116
+ result_type='expand')
117
+ return df['cc_id'].tolist(), df['cc_score'].tolist()
118
+
119
+
120
+ def fuse_per_query(semantic_ids: List[str], lexical_ids: List[str],
121
+ semantic_scores: List[float], lexical_scores: List[float],
122
+ normalize_method: str,
123
+ weight: float,
124
+ top_k: int,
125
+ semantic_theoretical_min_value: float,
126
+ lexical_theoretical_min_value: float):
127
+ normalize_func = normalize_method_dict[normalize_method]
128
+ norm_semantic_scores = normalize_func(semantic_scores, semantic_theoretical_min_value)
129
+ norm_lexical_scores = normalize_func(lexical_scores, lexical_theoretical_min_value)
130
+ ids = [semantic_ids, lexical_ids]
131
+ scores = [norm_semantic_scores, norm_lexical_scores]
132
+ df = pd.concat([pd.Series(dict(zip(_id, score))) for _id, score in zip(ids, scores)], axis=1)
133
+ df.columns = ['semantic', 'lexical']
134
+ df = df.fillna(0)
135
+ df['weighted_sum'] = df.mul((weight, 1.0 - weight)).sum(axis=1)
136
+ df = df.sort_values(by='weighted_sum', ascending=False)
137
+ return df.index.tolist()[:top_k], df['weighted_sum'][:top_k].tolist()
@@ -10,7 +10,8 @@ def hybrid_rrf(
10
10
  ids: Tuple,
11
11
  scores: Tuple,
12
12
  top_k: int,
13
- rrf_k: int = 60) -> Tuple[List[List[str]], List[List[float]]]:
13
+ weight: int = 60,
14
+ rrf_k: int = -1, ) -> Tuple[List[List[str]], List[List[float]]]:
14
15
  """
15
16
  Hybrid RRF function.
16
17
  RRF (Rank Reciprocal Fusion) is a method to fuse multiple retrieval results.
@@ -27,15 +28,23 @@ def hybrid_rrf(
27
28
  :param scores: The retrieve scores that you want to fuse.
28
29
  The length of this must be the same as the length of ids.
29
30
  :param top_k: The number of passages to be retrieved.
30
- :param rrf_k: Hyperparameter for RRF.
31
+ :param weight: Hyperparameter for RRF.
32
+ It was originally rrf_k value.
31
33
  Default is 60.
32
34
  For more information, please visit our documentation.
35
+ :param rrf_k: (Deprecated) Hyperparameter for RRF.
36
+ It was originally rrf_k value. Will remove at further version.
33
37
  :return: The tuple of ids and fused scores that fused by RRF.
34
38
  """
35
39
  assert len(ids) == len(scores), "The length of ids and scores must be the same."
36
40
  assert len(ids) > 1, "You must input more than one retrieval results."
37
41
  assert top_k > 0, "top_k must be greater than 0."
38
- assert rrf_k > 0, "rrf_k must be greater than 0."
42
+ assert weight > 0, "rrf_k must be greater than 0."
43
+
44
+ if rrf_k != -1:
45
+ weight = int(rrf_k)
46
+ else:
47
+ weight = int(weight)
39
48
 
40
49
  id_df = pd.DataFrame({f'id_{i}': id_list for i, id_list in enumerate(ids)})
41
50
  score_df = pd.DataFrame({f'score_{i}': score_list for i, score_list in enumerate(scores)})
@@ -44,9 +53,9 @@ def hybrid_rrf(
44
53
  def rrf_pure_apply(row):
45
54
  ids_tuple = tuple(row[[f'id_{i}' for i in range(len(ids))]].values)
46
55
  scores_tuple = tuple(row[[f'score_{i}' for i in range(len(scores))]].values)
47
- return pd.Series(rrf_pure(ids_tuple, scores_tuple, rrf_k, top_k))
56
+ return pd.Series(rrf_pure(ids_tuple, scores_tuple, weight, top_k))
48
57
 
49
- df[['rrf_id', 'rrf_score']] = df.swifter.apply(rrf_pure_apply, axis=1)
58
+ df[['rrf_id', 'rrf_score']] = df.apply(rrf_pure_apply, axis=1)
50
59
  return df['rrf_id'].tolist(), df['rrf_score'].tolist()
51
60
 
52
61
 
@@ -0,0 +1,285 @@
1
+ import logging
2
+ import os
3
+ import pathlib
4
+ from typing import List, Callable, Dict, Tuple
5
+
6
+ import numpy as np
7
+ import pandas as pd
8
+ from tqdm import tqdm
9
+
10
+ from autorag.evaluation import evaluate_retrieval
11
+ from autorag.strategy import measure_speed, filter_by_threshold, select_best
12
+
13
+ logger = logging.getLogger("AutoRAG")
14
+
15
+ semantic_module_names = ['vectordb']
16
+ lexical_module_names = ['bm25']
17
+ hybrid_module_names = ['hybrid_rrf', 'hybrid_cc']
18
+
19
+
20
+ def run_retrieval_node(modules: List[Callable],
21
+ module_params: List[Dict],
22
+ previous_result: pd.DataFrame,
23
+ node_line_dir: str,
24
+ strategies: Dict,
25
+ ) -> pd.DataFrame:
26
+ """
27
+ Run evaluation and select the best module among retrieval node results.
28
+
29
+ :param modules: Retrieval modules to run.
30
+ :param module_params: Retrieval module parameters.
31
+ :param previous_result: Previous result dataframe.
32
+ Could be query expansion's best result or qa data.
33
+ :param node_line_dir: This node line's directory.
34
+ :param strategies: Strategies for retrieval node.
35
+ :return: The best result dataframe.
36
+ It contains previous result columns and retrieval node's result columns.
37
+ """
38
+ if not os.path.exists(node_line_dir):
39
+ os.makedirs(node_line_dir)
40
+ project_dir = pathlib.PurePath(node_line_dir).parent.parent
41
+ qa_df = pd.read_parquet(os.path.join(project_dir, "data", "qa.parquet"), engine='pyarrow')
42
+ retrieval_gt = qa_df['retrieval_gt'].tolist()
43
+ retrieval_gt = [[[str(uuid) for uuid in sub_array] if sub_array.size > 0 else [] for sub_array in inner_array]
44
+ for inner_array in retrieval_gt]
45
+
46
+ save_dir = os.path.join(node_line_dir, "retrieval") # node name
47
+ if not os.path.exists(save_dir):
48
+ os.makedirs(save_dir)
49
+
50
+ def run(input_modules, input_module_params) -> Tuple[List[pd.DataFrame], List]:
51
+ """
52
+ Run input modules and parameters.
53
+
54
+ :param input_modules: Input modules
55
+ :param input_module_params: Input module parameters
56
+ :return: First, it returns list of result dataframe.
57
+ Second, it returns list of execution times.
58
+ """
59
+ result, execution_times = zip(*map(lambda task: measure_speed(
60
+ task[0], project_dir=project_dir, previous_result=previous_result, **task[1]),
61
+ zip(input_modules, input_module_params)))
62
+ average_times = list(map(lambda x: x / len(result[0]), execution_times))
63
+
64
+ # run metrics before filtering
65
+ if strategies.get('metrics') is None:
66
+ raise ValueError("You must at least one metrics for retrieval evaluation.")
67
+ result = list(map(lambda x: evaluate_retrieval_node(x, retrieval_gt, strategies.get('metrics'),
68
+ qa_df['query'].tolist(),
69
+ qa_df['generation_gt'].tolist()), result))
70
+
71
+ return result, average_times
72
+
73
+ def save_and_summary(input_modules, input_module_params, result_list,
74
+ execution_time_list, filename_start: int):
75
+ """
76
+ Save the result and make summary file
77
+
78
+ :param input_modules: Input modules
79
+ :param input_module_params: Input module parameters
80
+ :param result_list: Result list
81
+ :param execution_time_list: Execution times
82
+ :param filename_start: The first filename to use
83
+ :return: First, it returns list of result dataframe.
84
+ Second, it returns list of execution times.
85
+ """
86
+
87
+ # save results to folder
88
+ filepaths = list(map(lambda x: os.path.join(save_dir, f'{x}.parquet'),
89
+ range(filename_start, filename_start + len(input_modules))))
90
+ list(map(lambda x: x[0].to_parquet(x[1], index=False), zip(result_list, filepaths))) # execute save to parquet
91
+ filename_list = list(map(lambda x: os.path.basename(x), filepaths))
92
+
93
+ summary_df = pd.DataFrame({
94
+ 'filename': filename_list,
95
+ 'module_name': list(map(lambda module: module.__name__, input_modules)),
96
+ 'module_params': input_module_params,
97
+ 'execution_time': execution_time_list,
98
+ **{metric: list(map(lambda result: result[metric].mean(), result_list)) for metric in
99
+ strategies.get('metrics')},
100
+ })
101
+ summary_df.to_csv(os.path.join(save_dir, 'summary.csv'), index=False)
102
+ return summary_df
103
+
104
+ def find_best(results, average_times, filenames):
105
+ # filter by strategies
106
+ if strategies.get('speed_threshold') is not None:
107
+ results, filenames = filter_by_threshold(results, average_times, strategies['speed_threshold'], filenames)
108
+ selected_result, selected_filename = select_best(results, strategies.get('metrics'), filenames,
109
+ strategies.get('strategy', 'mean'))
110
+ return selected_result, selected_filename
111
+
112
+ filename_first = 0
113
+ # run semantic modules
114
+ logger.info(f"Running retrieval node - semantic retrieval module...")
115
+ if any([module.__name__ in semantic_module_names for module in modules]):
116
+ semantic_modules, semantic_module_params = zip(*filter(lambda x: x[0].__name__ in semantic_module_names,
117
+ zip(modules, module_params)))
118
+ semantic_results, semantic_times = run(semantic_modules, semantic_module_params)
119
+ semantic_summary_df = save_and_summary(semantic_modules, semantic_module_params,
120
+ semantic_results, semantic_times, filename_first)
121
+ semantic_selected_result, semantic_selected_filename = find_best(semantic_results, semantic_times,
122
+ semantic_summary_df['filename'].tolist())
123
+ semantic_summary_df['is_best'] = semantic_summary_df['filename'] == semantic_selected_filename
124
+ filename_first += len(semantic_modules)
125
+ else:
126
+ semantic_selected_filename, semantic_summary_df, semantic_results, semantic_times = None, pd.DataFrame(), [], []
127
+ # run lexical modules
128
+ logger.info(f"Running retrieval node - lexical retrieval module...")
129
+ if any([module.__name__ in lexical_module_names for module in modules]):
130
+ lexical_modules, lexical_module_params = zip(*filter(lambda x: x[0].__name__ in lexical_module_names,
131
+ zip(modules, module_params)))
132
+ lexical_results, lexical_times = run(lexical_modules, lexical_module_params)
133
+ lexical_summary_df = save_and_summary(lexical_modules, lexical_module_params,
134
+ lexical_results, lexical_times, filename_first)
135
+ lexical_selected_result, lexical_selected_filename = find_best(lexical_results, lexical_times,
136
+ lexical_summary_df['filename'].tolist())
137
+ lexical_summary_df['is_best'] = lexical_summary_df['filename'] == lexical_selected_filename
138
+ filename_first += len(lexical_modules)
139
+ else:
140
+ lexical_selected_filename, lexical_summary_df, lexical_results, lexical_times = None, pd.DataFrame(), [], []
141
+
142
+ logger.info(f"Running retrieval node - hybrid retrieval module...")
143
+ # Next, run hybrid retrieval
144
+ if any([module.__name__ in hybrid_module_names for module in modules]):
145
+ hybrid_modules, hybrid_module_params = zip(*filter(lambda x: x[0].__name__ in hybrid_module_names,
146
+ zip(modules, module_params)))
147
+ if all(['target_module_params' in x for x in hybrid_module_params]): # for Runner.run
148
+ # If target_module_params are already given, run hybrid retrieval directly
149
+ hybrid_results, hybrid_times = run(hybrid_modules, hybrid_module_params)
150
+ hybrid_summary_df = save_and_summary(hybrid_modules, hybrid_module_params,
151
+ hybrid_results, hybrid_times, filename_first)
152
+ filename_first += len(hybrid_modules)
153
+ else: # for Evaluator
154
+ # get id and score
155
+ ids_scores = get_ids_and_scores(save_dir, [semantic_selected_filename, lexical_selected_filename])
156
+ hybrid_module_params = list(map(lambda x: {**x, **ids_scores}, hybrid_module_params))
157
+
158
+ # optimize each modules
159
+ real_hybrid_times = [get_hybrid_execution_times(semantic_summary_df, lexical_summary_df)
160
+ ] * len(hybrid_module_params)
161
+ hybrid_times = real_hybrid_times.copy()
162
+ hybrid_results = []
163
+ for module, module_param in zip(hybrid_modules, hybrid_module_params):
164
+ module_result_df, module_best_weight = optimize_hybrid(module, module_param, strategies,
165
+ retrieval_gt, qa_df,
166
+ project_dir, previous_result)
167
+ module_param['weight'] = module_best_weight
168
+ hybrid_results.append(module_result_df)
169
+
170
+ hybrid_summary_df = save_and_summary(hybrid_modules, hybrid_module_params,
171
+ hybrid_results, hybrid_times, filename_first)
172
+ filename_first += len(hybrid_modules)
173
+ hybrid_summary_df['execution_time'] = hybrid_times
174
+ best_semantic_summary_row = semantic_summary_df.loc[semantic_summary_df['is_best'] == True].iloc[0]
175
+ best_lexical_summary_row = lexical_summary_df.loc[lexical_summary_df['is_best'] == True].iloc[0]
176
+ target_modules = (best_semantic_summary_row['module_name'], best_lexical_summary_row['module_name'])
177
+ target_module_params = (
178
+ best_semantic_summary_row['module_params'], best_lexical_summary_row['module_params'])
179
+ hybrid_summary_df = edit_summary_df_params(hybrid_summary_df, target_modules, target_module_params)
180
+ else:
181
+ if any([module.__name__ in hybrid_module_names for module in modules]):
182
+ logger.warning("You must at least one semantic module and lexical module for hybrid evaluation."
183
+ "Passing hybrid module.")
184
+ hybrid_selected_filename, hybrid_summary_df, hybrid_results, hybrid_times = None, pd.DataFrame(), [], []
185
+
186
+ summary = pd.concat([semantic_summary_df, lexical_summary_df, hybrid_summary_df], ignore_index=True)
187
+ results = semantic_results + lexical_results + hybrid_results
188
+ average_times = semantic_times + lexical_times + hybrid_times
189
+ filenames = summary['filename'].tolist()
190
+
191
+ # filter by strategies
192
+ selected_result, selected_filename = find_best(results, average_times, filenames)
193
+ best_result = pd.concat([previous_result, selected_result], axis=1)
194
+
195
+ # add summary.csv 'is_best' column
196
+ summary['is_best'] = summary['filename'] == selected_filename
197
+
198
+ # save the result files
199
+ best_result.to_parquet(os.path.join(save_dir, f'best_{os.path.splitext(selected_filename)[0]}.parquet'),
200
+ index=False)
201
+ summary.to_csv(os.path.join(save_dir, 'summary.csv'), index=False)
202
+ return best_result
203
+
204
+
205
+ def evaluate_retrieval_node(result_df: pd.DataFrame, retrieval_gt, metrics,
206
+ queries: List[str], generation_gt: List[List[str]]) -> pd.DataFrame:
207
+ """
208
+ Evaluate retrieval node from retrieval node result dataframe.
209
+
210
+ :param result_df: The result dataframe from a retrieval node.
211
+ :param retrieval_gt: Ground truth for retrieval from qa dataset.
212
+ :param metrics: Metric list from input strategies.
213
+ :param queries: Query list from input strategies.
214
+ :param generation_gt: Ground truth for generation from qa dataset.
215
+ :return: Return result_df with metrics columns.
216
+ The columns will be 'retrieved_contents', 'retrieved_ids', 'retrieve_scores', and metric names.
217
+ """
218
+
219
+ @evaluate_retrieval(retrieval_gt=retrieval_gt, metrics=metrics, queries=queries, generation_gt=generation_gt)
220
+ def evaluate_this_module(df: pd.DataFrame):
221
+ return df['retrieved_contents'].tolist(), df['retrieved_ids'].tolist(), df['retrieve_scores'].tolist()
222
+
223
+ return evaluate_this_module(result_df)
224
+
225
+
226
+ def edit_summary_df_params(summary_df: pd.DataFrame, target_modules, target_module_params) -> pd.DataFrame:
227
+ def delete_ids_scores(x):
228
+ del x['ids']
229
+ del x['scores']
230
+ return x
231
+
232
+ summary_df['module_params'] = summary_df['module_params'].apply(delete_ids_scores)
233
+ summary_df['new_params'] = [{'target_modules': target_modules,
234
+ 'target_module_params': target_module_params}] * len(summary_df)
235
+ summary_df['module_params'] = summary_df.apply(lambda row: {**row['module_params'], **row['new_params']}, axis=1)
236
+ summary_df = summary_df.drop(columns=['new_params'])
237
+ return summary_df
238
+
239
+
240
+ def get_ids_and_scores(node_dir: str, filenames: List[str]) -> Dict:
241
+ best_results_df = list(
242
+ map(lambda filename: pd.read_parquet(os.path.join(node_dir, filename), engine='pyarrow'), filenames))
243
+ ids = tuple(map(lambda df: df['retrieved_ids'].apply(list).tolist(), best_results_df))
244
+ scores = tuple(map(lambda df: df['retrieve_scores'].apply(list).tolist(), best_results_df))
245
+ return {
246
+ 'ids': ids,
247
+ 'scores': scores,
248
+ }
249
+
250
+
251
+ def get_hybrid_execution_times(lexical_summary, semantic_summary) -> float:
252
+ lexical_execution_time = lexical_summary.loc[lexical_summary['is_best'] == True].iloc[0]['execution_time']
253
+ semantic_execution_time = semantic_summary.loc[semantic_summary['is_best'] == True].iloc[0]['execution_time']
254
+ return lexical_execution_time + semantic_execution_time
255
+
256
+
257
+ def optimize_hybrid(hybrid_module_func: Callable, hybrid_module_param: Dict,
258
+ strategy: Dict, retrieval_gt, qa_df: pd.DataFrame,
259
+ project_dir, previous_result):
260
+ if hybrid_module_func.__name__ == 'hybrid_rrf':
261
+ weight_range = hybrid_module_param.pop('weight_range', (4, 80))
262
+ test_weight_size = weight_range[1] - weight_range[0] + 1
263
+ else:
264
+ weight_range = hybrid_module_param.pop('weight_range', (0.0, 1.0))
265
+ test_weight_size = hybrid_module_param.pop('test_weight_size', 101)
266
+
267
+ weight_candidates = np.linspace(weight_range[0], weight_range[1], test_weight_size).tolist()
268
+
269
+ result_list = []
270
+ for weight_value in tqdm(weight_candidates):
271
+ result_df = hybrid_module_func(project_dir=project_dir, previous_result=previous_result,
272
+ weight=weight_value, **hybrid_module_param)
273
+ result_list.append(result_df)
274
+
275
+ # evaluate here
276
+ if strategy.get('metrics') is None:
277
+ raise ValueError("You must at least one metrics for retrieval evaluation.")
278
+ result_list = list(map(lambda x: evaluate_retrieval_node(x, retrieval_gt, strategy.get('metrics'),
279
+ qa_df['query'].tolist(),
280
+ qa_df['generation_gt'].tolist()), result_list))
281
+
282
+ # select best result
283
+ best_result_df, best_weight = select_best(result_list, strategy.get('metrics'), metadatas=weight_candidates,
284
+ strategy_name=strategy.get('strategy', 'normalize_mean'))
285
+ return best_result_df, best_weight
@@ -24,8 +24,6 @@ def get_support_modules(module_name: str) -> Callable:
24
24
  'vectordb': ('autorag.nodes.retrieval', 'vectordb'),
25
25
  'hybrid_rrf': ('autorag.nodes.retrieval', 'hybrid_rrf'),
26
26
  'hybrid_cc': ('autorag.nodes.retrieval', 'hybrid_cc'),
27
- 'hybrid_rsf': ('autorag.nodes.retrieval', 'hybrid_rsf'),
28
- 'hybrid_dbsf': ('autorag.nodes.retrieval', 'hybrid_dbsf'),
29
27
  # passage_augmenter
30
28
  'prev_next_augmenter': ('autorag.nodes.passageaugmenter', 'prev_next_augmenter'),
31
29
  'pass_passage_augmenter': ('autorag.nodes.passageaugmenter', 'pass_passage_augmenter'),
@@ -28,14 +28,6 @@ autorag.nodes.retrieval.hybrid\_cc module
28
28
  :undoc-members:
29
29
  :show-inheritance:
30
30
 
31
- autorag.nodes.retrieval.hybrid\_dbsf module
32
- -------------------------------------------
33
-
34
- .. automodule:: autorag.nodes.retrieval.hybrid_dbsf
35
- :members:
36
- :undoc-members:
37
- :show-inheritance:
38
-
39
31
  autorag.nodes.retrieval.hybrid\_rrf module
40
32
  ------------------------------------------
41
33
 
@@ -44,14 +36,6 @@ autorag.nodes.retrieval.hybrid\_rrf module
44
36
  :undoc-members:
45
37
  :show-inheritance:
46
38
 
47
- autorag.nodes.retrieval.hybrid\_rsf module
48
- ------------------------------------------
49
-
50
- .. automodule:: autorag.nodes.retrieval.hybrid_rsf
51
- :members:
52
- :undoc-members:
53
- :show-inheritance:
54
-
55
39
  autorag.nodes.retrieval.run module
56
40
  ----------------------------------
57
41
 
@@ -63,7 +63,7 @@ at [here](https://medium.com/@autorag/sem-score-maybe-the-answer-to-rag-evaluati
63
63
 
64
64
  ## 5. G-Eval
65
65
 
66
- ### 📌Definition
66
+ ### 📌 Definition
67
67
 
68
68
  Here is the [link](https://arxiv.org/abs/2303.16634) that introduced ***G-Eval***
69
69
 
@@ -77,14 +77,14 @@ So, in AutoRAG, we use **G-Eval with GPT-4**
77
77
 
78
78
  ---
79
79
 
80
- ### 🍀1. Coherence
80
+ ### 5-1. Coherence
81
81
 
82
82
  - Evaluate whether the answer is logically consistent and flows naturally.
83
83
  - Evaluate the connections between sentences and how they fit into the overall context.
84
84
 
85
85
  ---
86
86
 
87
- ### 🍀2. Consistency
87
+ ### 5-2. Consistency
88
88
 
89
89
  - Evaluate whether the answer is consistent with and does not contradict the question asked or the information
90
90
  presented.
@@ -93,17 +93,28 @@ So, in AutoRAG, we use **G-Eval with GPT-4**
93
93
 
94
94
  ---
95
95
 
96
- ### 🍀3. Fluency
96
+ ### 5-3. Fluency
97
97
 
98
98
  - Evaluate answers for fluency
99
99
 
100
100
  ---
101
101
 
102
- ### 🍀4. Relevance
102
+ ### 5-4. Relevance
103
103
 
104
104
  - Evaluate how well the answer meets the question's requirements
105
105
  - A highly relevant answer should be directly related to the question's core topic or keyword.
106
106
 
107
+ ### ❗How to use specific G-Eval metrics
108
+
109
+ You can use specific G-Eval metrics to use `metrics` parameter.
110
+
111
+ Here is an example yaml file that uses **G-Eval consistency** metric.
112
+
113
+ ```yaml
114
+ - metric_name: g_eval
115
+ metrics: [ consistency ]
116
+ ```
117
+
107
118
  ## 6. Bert Score
108
119
 
109
120
  ### 📌Definition