AutoRAG 0.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (184) hide show
  1. autorag/__init__.py +82 -0
  2. autorag/chunker.py +51 -0
  3. autorag/cli.py +209 -0
  4. autorag/dashboard.py +199 -0
  5. autorag/data/__init__.py +109 -0
  6. autorag/data/chunk/__init__.py +2 -0
  7. autorag/data/chunk/base.py +128 -0
  8. autorag/data/chunk/langchain_chunk.py +76 -0
  9. autorag/data/chunk/llama_index_chunk.py +96 -0
  10. autorag/data/chunk/run.py +38 -0
  11. autorag/data/legacy/__init__.py +0 -0
  12. autorag/data/legacy/corpus/__init__.py +2 -0
  13. autorag/data/legacy/corpus/langchain.py +47 -0
  14. autorag/data/legacy/corpus/llama_index.py +93 -0
  15. autorag/data/legacy/qacreation/__init__.py +6 -0
  16. autorag/data/legacy/qacreation/base.py +239 -0
  17. autorag/data/legacy/qacreation/llama_index.py +253 -0
  18. autorag/data/legacy/qacreation/llama_index_default_prompt.txt +54 -0
  19. autorag/data/legacy/qacreation/ragas.py +75 -0
  20. autorag/data/legacy/qacreation/simple.py +99 -0
  21. autorag/data/parse/__init__.py +1 -0
  22. autorag/data/parse/base.py +79 -0
  23. autorag/data/parse/clova.py +194 -0
  24. autorag/data/parse/langchain_parse.py +87 -0
  25. autorag/data/parse/llamaparse.py +126 -0
  26. autorag/data/parse/run.py +141 -0
  27. autorag/data/parse/table_hybrid_parse.py +134 -0
  28. autorag/data/qa/__init__.py +3 -0
  29. autorag/data/qa/evolve/__init__.py +0 -0
  30. autorag/data/qa/evolve/llama_index_query_evolve.py +64 -0
  31. autorag/data/qa/evolve/openai_query_evolve.py +81 -0
  32. autorag/data/qa/evolve/prompt.py +288 -0
  33. autorag/data/qa/extract_evidence.py +1 -0
  34. autorag/data/qa/filter/__init__.py +0 -0
  35. autorag/data/qa/filter/dontknow.py +117 -0
  36. autorag/data/qa/filter/passage_dependency.py +88 -0
  37. autorag/data/qa/filter/prompt.py +73 -0
  38. autorag/data/qa/generation_gt/__init__.py +0 -0
  39. autorag/data/qa/generation_gt/base.py +16 -0
  40. autorag/data/qa/generation_gt/llama_index_gen_gt.py +41 -0
  41. autorag/data/qa/generation_gt/openai_gen_gt.py +84 -0
  42. autorag/data/qa/generation_gt/prompt.py +27 -0
  43. autorag/data/qa/query/__init__.py +0 -0
  44. autorag/data/qa/query/llama_gen_query.py +82 -0
  45. autorag/data/qa/query/openai_gen_query.py +95 -0
  46. autorag/data/qa/query/prompt.py +201 -0
  47. autorag/data/qa/sample.py +26 -0
  48. autorag/data/qa/schema.py +322 -0
  49. autorag/data/utils/__init__.py +0 -0
  50. autorag/data/utils/util.py +103 -0
  51. autorag/deploy/__init__.py +9 -0
  52. autorag/deploy/api.py +303 -0
  53. autorag/deploy/base.py +235 -0
  54. autorag/deploy/gradio.py +74 -0
  55. autorag/deploy/swagger.yml +202 -0
  56. autorag/embedding/__init__.py +0 -0
  57. autorag/embedding/base.py +144 -0
  58. autorag/embedding/vllm.py +256 -0
  59. autorag/evaluation/__init__.py +3 -0
  60. autorag/evaluation/generation.py +88 -0
  61. autorag/evaluation/metric/__init__.py +22 -0
  62. autorag/evaluation/metric/deepeval_prompt.py +322 -0
  63. autorag/evaluation/metric/g_eval_prompts/coh_detailed.txt +32 -0
  64. autorag/evaluation/metric/g_eval_prompts/con_detailed.txt +33 -0
  65. autorag/evaluation/metric/g_eval_prompts/flu_detailed.txt +26 -0
  66. autorag/evaluation/metric/g_eval_prompts/rel_detailed.txt +33 -0
  67. autorag/evaluation/metric/generation.py +504 -0
  68. autorag/evaluation/metric/retrieval.py +115 -0
  69. autorag/evaluation/metric/retrieval_contents.py +65 -0
  70. autorag/evaluation/metric/util.py +88 -0
  71. autorag/evaluation/retrieval.py +83 -0
  72. autorag/evaluation/retrieval_contents.py +65 -0
  73. autorag/evaluation/util.py +43 -0
  74. autorag/evaluator.py +559 -0
  75. autorag/node_line.py +65 -0
  76. autorag/nodes/__init__.py +0 -0
  77. autorag/nodes/generator/__init__.py +4 -0
  78. autorag/nodes/generator/base.py +103 -0
  79. autorag/nodes/generator/llama_index_llm.py +169 -0
  80. autorag/nodes/generator/openai_llm.py +329 -0
  81. autorag/nodes/generator/run.py +148 -0
  82. autorag/nodes/generator/vllm.py +147 -0
  83. autorag/nodes/generator/vllm_api.py +191 -0
  84. autorag/nodes/hybridretrieval/__init__.py +2 -0
  85. autorag/nodes/hybridretrieval/base.py +58 -0
  86. autorag/nodes/hybridretrieval/hybrid_cc.py +227 -0
  87. autorag/nodes/hybridretrieval/hybrid_rrf.py +149 -0
  88. autorag/nodes/hybridretrieval/run.py +137 -0
  89. autorag/nodes/lexicalretrieval/__init__.py +1 -0
  90. autorag/nodes/lexicalretrieval/bm25.py +381 -0
  91. autorag/nodes/lexicalretrieval/run.py +148 -0
  92. autorag/nodes/passageaugmenter/__init__.py +2 -0
  93. autorag/nodes/passageaugmenter/base.py +76 -0
  94. autorag/nodes/passageaugmenter/pass_passage_augmenter.py +43 -0
  95. autorag/nodes/passageaugmenter/prev_next_augmenter.py +155 -0
  96. autorag/nodes/passageaugmenter/run.py +131 -0
  97. autorag/nodes/passagecompressor/__init__.py +4 -0
  98. autorag/nodes/passagecompressor/base.py +78 -0
  99. autorag/nodes/passagecompressor/longllmlingua.py +115 -0
  100. autorag/nodes/passagecompressor/pass_compressor.py +16 -0
  101. autorag/nodes/passagecompressor/refine.py +54 -0
  102. autorag/nodes/passagecompressor/run.py +186 -0
  103. autorag/nodes/passagecompressor/tree_summarize.py +56 -0
  104. autorag/nodes/passagefilter/__init__.py +6 -0
  105. autorag/nodes/passagefilter/base.py +40 -0
  106. autorag/nodes/passagefilter/pass_passage_filter.py +14 -0
  107. autorag/nodes/passagefilter/percentile_cutoff.py +58 -0
  108. autorag/nodes/passagefilter/recency.py +105 -0
  109. autorag/nodes/passagefilter/run.py +138 -0
  110. autorag/nodes/passagefilter/similarity_percentile_cutoff.py +134 -0
  111. autorag/nodes/passagefilter/similarity_threshold_cutoff.py +112 -0
  112. autorag/nodes/passagefilter/threshold_cutoff.py +78 -0
  113. autorag/nodes/passagereranker/__init__.py +16 -0
  114. autorag/nodes/passagereranker/base.py +44 -0
  115. autorag/nodes/passagereranker/cohere.py +118 -0
  116. autorag/nodes/passagereranker/colbert.py +213 -0
  117. autorag/nodes/passagereranker/flag_embedding.py +112 -0
  118. autorag/nodes/passagereranker/flag_embedding_llm.py +101 -0
  119. autorag/nodes/passagereranker/flashrank.py +245 -0
  120. autorag/nodes/passagereranker/jina.py +115 -0
  121. autorag/nodes/passagereranker/koreranker.py +136 -0
  122. autorag/nodes/passagereranker/mixedbreadai.py +126 -0
  123. autorag/nodes/passagereranker/monot5.py +190 -0
  124. autorag/nodes/passagereranker/openvino.py +191 -0
  125. autorag/nodes/passagereranker/pass_reranker.py +31 -0
  126. autorag/nodes/passagereranker/rankgpt.py +170 -0
  127. autorag/nodes/passagereranker/run.py +145 -0
  128. autorag/nodes/passagereranker/sentence_transformer.py +129 -0
  129. autorag/nodes/passagereranker/tart/__init__.py +1 -0
  130. autorag/nodes/passagereranker/tart/modeling_enc_t5.py +152 -0
  131. autorag/nodes/passagereranker/tart/tart.py +139 -0
  132. autorag/nodes/passagereranker/tart/tokenization_enc_t5.py +112 -0
  133. autorag/nodes/passagereranker/time_reranker.py +72 -0
  134. autorag/nodes/passagereranker/upr.py +160 -0
  135. autorag/nodes/passagereranker/voyageai.py +109 -0
  136. autorag/nodes/promptmaker/__init__.py +12 -0
  137. autorag/nodes/promptmaker/base.py +32 -0
  138. autorag/nodes/promptmaker/chat_fstring.py +73 -0
  139. autorag/nodes/promptmaker/fstring.py +49 -0
  140. autorag/nodes/promptmaker/long_context_reorder.py +83 -0
  141. autorag/nodes/promptmaker/run.py +283 -0
  142. autorag/nodes/promptmaker/window_replacement.py +85 -0
  143. autorag/nodes/queryexpansion/__init__.py +4 -0
  144. autorag/nodes/queryexpansion/base.py +62 -0
  145. autorag/nodes/queryexpansion/hyde.py +43 -0
  146. autorag/nodes/queryexpansion/multi_query_expansion.py +57 -0
  147. autorag/nodes/queryexpansion/pass_query_expansion.py +22 -0
  148. autorag/nodes/queryexpansion/query_decompose.py +111 -0
  149. autorag/nodes/queryexpansion/run.py +308 -0
  150. autorag/nodes/retrieval/__init__.py +0 -0
  151. autorag/nodes/retrieval/base.py +127 -0
  152. autorag/nodes/retrieval/run_util.py +152 -0
  153. autorag/nodes/semanticretrieval/__init__.py +1 -0
  154. autorag/nodes/semanticretrieval/run.py +148 -0
  155. autorag/nodes/semanticretrieval/vectordb.py +339 -0
  156. autorag/nodes/util.py +16 -0
  157. autorag/parser.py +37 -0
  158. autorag/schema/__init__.py +3 -0
  159. autorag/schema/base.py +35 -0
  160. autorag/schema/metricinput.py +99 -0
  161. autorag/schema/module.py +24 -0
  162. autorag/schema/node.py +144 -0
  163. autorag/strategy.py +165 -0
  164. autorag/support.py +235 -0
  165. autorag/utils/__init__.py +8 -0
  166. autorag/utils/cast.py +45 -0
  167. autorag/utils/preprocess.py +149 -0
  168. autorag/utils/util.py +759 -0
  169. autorag/validator.py +98 -0
  170. autorag/vectordb/__init__.py +75 -0
  171. autorag/vectordb/base.py +73 -0
  172. autorag/vectordb/chroma.py +118 -0
  173. autorag/vectordb/couchbase.py +239 -0
  174. autorag/vectordb/milvus.py +169 -0
  175. autorag/vectordb/pinecone.py +121 -0
  176. autorag/vectordb/qdrant.py +155 -0
  177. autorag/vectordb/weaviate.py +184 -0
  178. autorag/web.py +81 -0
  179. autorag-0.0.0.dist-info/METADATA +780 -0
  180. autorag-0.0.0.dist-info/RECORD +184 -0
  181. autorag-0.0.0.dist-info/WHEEL +5 -0
  182. autorag-0.0.0.dist-info/entry_points.txt +2 -0
  183. autorag-0.0.0.dist-info/licenses/LICENSE +201 -0
  184. autorag-0.0.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,84 @@
1
+ import itertools
2
+ from typing import Dict
3
+
4
+ from openai import AsyncClient
5
+ from pydantic import BaseModel
6
+
7
+ from autorag.data.qa.generation_gt.base import add_gen_gt
8
+ from autorag.data.qa.generation_gt.prompt import GEN_GT_SYSTEM_PROMPT
9
+
10
+
11
+ class Response(BaseModel):
12
+ answer: str
13
+
14
+
15
+ async def make_gen_gt_openai(
16
+ row: Dict,
17
+ client: AsyncClient,
18
+ system_prompt: str,
19
+ model_name: str = "gpt-4o-2024-08-06",
20
+ ):
21
+ retrieval_gt_contents = list(
22
+ itertools.chain.from_iterable(row["retrieval_gt_contents"])
23
+ )
24
+ query = row["query"]
25
+ passage_str = "\n".join(retrieval_gt_contents)
26
+ user_prompt = f"Text:\n<|text_start|>\n{passage_str}\n<|text_end|>\n\nQuestion:\n{query}\n\nAnswer:"
27
+
28
+ completion = await client.beta.chat.completions.parse(
29
+ model=model_name,
30
+ messages=[
31
+ {"role": "system", "content": system_prompt},
32
+ {"role": "user", "content": user_prompt},
33
+ ],
34
+ temperature=0.0,
35
+ response_format=Response,
36
+ )
37
+ response: Response = completion.choices[0].message.parsed
38
+ return add_gen_gt(row, response.answer)
39
+
40
+
41
+ async def make_concise_gen_gt(
42
+ row: Dict,
43
+ client: AsyncClient,
44
+ model_name: str = "gpt-4o-2024-08-06",
45
+ lang: str = "en",
46
+ ):
47
+ """
48
+ Generate concise generation_gt using OpenAI Structured Output for preventing errors.
49
+ It generates a concise answer, so it is generally a word or just a phrase.
50
+
51
+ :param row: The input row of the qa dataframe.
52
+ :param client: The OpenAI async client.
53
+ :param model_name: The model name that supports structured output.
54
+ It has to be "gpt-4o-2024-08-06" or "gpt-4o-mini-2024-07-18".
55
+ :param lang: The language code of the prompt.
56
+ Default is "en".
57
+ :return: The output row of the qa dataframe with added "generation_gt" in it.
58
+ """
59
+ return await make_gen_gt_openai(
60
+ row, client, GEN_GT_SYSTEM_PROMPT["concise"][lang], model_name
61
+ )
62
+
63
+
64
+ async def make_basic_gen_gt(
65
+ row: Dict,
66
+ client: AsyncClient,
67
+ model_name: str = "gpt-4o-2024-08-06",
68
+ lang: str = "en",
69
+ ):
70
+ """
71
+ Generate basic generation_gt using OpenAI Structured Output for preventing errors.
72
+ It generates a "basic" answer, and its prompt is simple.
73
+
74
+ :param row: The input row of the qa dataframe.
75
+ :param client: The OpenAI async client.
76
+ :param model_name: The model name that supports structured output.
77
+ It has to be "gpt-4o-2024-08-06" or "gpt-4o-mini-2024-07-18".
78
+ :param lang: The language code of the prompt.
79
+ Default is "en".
80
+ :return: The output row of the qa dataframe with added "generation_gt" in it.
81
+ """
82
+ return await make_gen_gt_openai(
83
+ row, client, GEN_GT_SYSTEM_PROMPT["basic"][lang], model_name
84
+ )
@@ -0,0 +1,27 @@
1
+ GEN_GT_SYSTEM_PROMPT = {
2
+ "concise": {
3
+ "en": """You are an AI assistant to answer the given question in the provide evidence text.
4
+ You can find the evidence from the given text about question, and you have to write a proper answer to the given question.
5
+ Your answer have to be concise and relevant to the question.
6
+ Do not make a verbose answer and make it super clear.
7
+ It doesn't have to be an full sentence. It can be the answer is a word or a paraphrase.""",
8
+ "ko": """당신은 주어진 질문에 대해 제공된 Text 내에서 답을 찾는 AI 비서입니다.
9
+ 질문에 대한 답을 Text에서 찾아 적절한 답변을 작성하세요.
10
+ 답변은 간결하고 질문에 관련된 내용만 포함해야 합니다.
11
+ 불필요하게 길게 답변하지 말고, 명확하게 작성하세요.
12
+ 완전한 문장이 아니어도 되며, 답은 단어나 요약일 수 있습니다.""",
13
+ "ja": """
14
+ あなたは与えられた質問に対して提供されたText内で答えを探すAI秘書です。
15
+ 質問に対する答えをTextで探して適切な答えを作成しましょう。
16
+ 回答は簡潔で、質問に関連する内容のみを含める必要があります。
17
+ 不必要に長く答えず、明確に作成しましょう。
18
+ 完全な文章でなくてもいいし、答えは単語や要約かもしれません。
19
+ """,
20
+ },
21
+ "basic": {
22
+ "en": """You are an AI assistant to answer the given question in the provide evidence text.
23
+ You can find the evidence from the given text about question, and you have to write a proper answer to the given question.""",
24
+ "ko": "당신은 주어진 질문에 대한 답을 제공된 Text 내에서 찾는 AI 비서입니다. 질문과 관련된 증거를 Text에서 찾아 적절한 답변을 작성하세요.",
25
+ "ja": "あなたは与えられた質問に対する答えを提供されたText内で探すAI秘書です。 質問に関する証拠をTextで探して適切な回答を作成しましょう。",
26
+ },
27
+ }
File without changes
@@ -0,0 +1,82 @@
1
+ import itertools
2
+ from typing import Dict, List
3
+
4
+ from llama_index.core.base.llms.base import BaseLLM
5
+ from llama_index.core.base.llms.types import ChatResponse, ChatMessage, MessageRole
6
+
7
+ from autorag.data.qa.query.prompt import QUERY_GEN_PROMPT, QUERY_GEN_PROMPT_EXTRA
8
+
9
+
10
+ async def llama_index_generate_base(
11
+ row: Dict,
12
+ llm: BaseLLM,
13
+ messages: List[ChatMessage],
14
+ ) -> Dict:
15
+ context = list(itertools.chain.from_iterable(row["retrieval_gt_contents"]))
16
+ context_str = "\n".join([f"{i + 1}. {c}" for i, c in enumerate(context)])
17
+ user_prompt = f"Text:\n{context_str}\n\nGenerated Question from the Text:\n"
18
+ user_message = ChatMessage(role=MessageRole.USER, content=user_prompt)
19
+ new_messages = [*messages, user_message]
20
+ chat_response: ChatResponse = await llm.achat(messages=new_messages)
21
+ row["query"] = chat_response.message.content
22
+ return row
23
+
24
+
25
+ async def factoid_query_gen(
26
+ row: Dict,
27
+ llm: BaseLLM,
28
+ lang: str = "en",
29
+ ) -> Dict:
30
+ return await llama_index_generate_base(
31
+ row, llm, QUERY_GEN_PROMPT["factoid_single_hop"][lang]
32
+ )
33
+
34
+
35
+ async def concept_completion_query_gen(
36
+ row: Dict,
37
+ llm: BaseLLM,
38
+ lang: str = "en",
39
+ ) -> Dict:
40
+ return await llama_index_generate_base(
41
+ row, llm, QUERY_GEN_PROMPT["concept_completion"][lang]
42
+ )
43
+
44
+
45
+ async def two_hop_incremental(
46
+ row: Dict,
47
+ llm: BaseLLM,
48
+ lang: str = "en",
49
+ ) -> Dict:
50
+ messages = QUERY_GEN_PROMPT["two_hop_incremental"][lang]
51
+ passages = row["retrieval_gt_contents"]
52
+ assert len(passages) >= 2, (
53
+ "You have to sample more than two passages for making two-hop questions."
54
+ )
55
+ context_str = f"Document 1: {passages[0][0]}\nDocument 2: {passages[1][0]}"
56
+ user_prompt = f"{context_str}\n\nGenerated two-hop Question from two Documents:\n"
57
+ messages.append(ChatMessage(role=MessageRole.USER, content=user_prompt))
58
+
59
+ chat_response: ChatResponse = await llm.achat(messages=messages)
60
+ response = chat_response.message.content
61
+ row["query"] = response.split(":")[-1].strip()
62
+ return row
63
+
64
+
65
+ async def custom_query_gen(
66
+ row: Dict,
67
+ llm: BaseLLM,
68
+ messages: List[ChatMessage],
69
+ ) -> Dict:
70
+ return await llama_index_generate_base(row, llm, messages)
71
+
72
+
73
+ # Experimental feature: can only use factoid_single_hop
74
+ async def multiple_queries_gen(
75
+ row: Dict,
76
+ llm: BaseLLM,
77
+ lang: str = "en",
78
+ n: int = 3,
79
+ ) -> Dict:
80
+ _messages = QUERY_GEN_PROMPT["factoid_single_hop"][lang]
81
+ _messages[0].content += QUERY_GEN_PROMPT_EXTRA["multiple_queries"][lang].format(n=n)
82
+ return await llama_index_generate_base(row, llm, _messages)
@@ -0,0 +1,95 @@
1
+ import itertools
2
+ from typing import Dict, List
3
+
4
+ from llama_index.core.base.llms.types import ChatMessage, MessageRole
5
+ from llama_index.llms.openai.utils import to_openai_message_dicts
6
+ from openai import AsyncClient
7
+ from pydantic import BaseModel
8
+
9
+ from autorag.data.qa.query.prompt import QUERY_GEN_PROMPT
10
+
11
+
12
+ class Response(BaseModel):
13
+ query: str
14
+
15
+
16
+ # Single hop QA generation OpenAI
17
+ async def query_gen_openai_base(
18
+ row: Dict,
19
+ client: AsyncClient,
20
+ messages: List[ChatMessage],
21
+ model_name: str = "gpt-4o-2024-08-06",
22
+ ):
23
+ context = list(itertools.chain.from_iterable(row["retrieval_gt_contents"]))
24
+ context_str = "Text:\n" + "\n".join(
25
+ [f"{i + 1}. {c}" for i, c in enumerate(context)]
26
+ )
27
+ user_prompt = f"{context_str}\n\nGenerated Question from the Text:\n"
28
+ messages.append(ChatMessage(role=MessageRole.USER, content=user_prompt))
29
+
30
+ completion = await client.beta.chat.completions.parse(
31
+ model=model_name,
32
+ messages=to_openai_message_dicts(messages),
33
+ response_format=Response,
34
+ )
35
+ row["query"] = completion.choices[0].message.parsed.query
36
+ return row
37
+
38
+
39
+ async def factoid_query_gen(
40
+ row: Dict,
41
+ client: AsyncClient,
42
+ model_name: str = "gpt-4o-2024-08-06",
43
+ lang: str = "en",
44
+ ) -> Dict:
45
+ return await query_gen_openai_base(
46
+ row, client, QUERY_GEN_PROMPT["factoid_single_hop"][lang], model_name
47
+ )
48
+
49
+
50
+ async def concept_completion_query_gen(
51
+ row: Dict,
52
+ client: AsyncClient,
53
+ model_name: str = "gpt-4o-2024-08-06",
54
+ lang: str = "en",
55
+ ) -> Dict:
56
+ return await query_gen_openai_base(
57
+ row, client, QUERY_GEN_PROMPT["factoid_single_hop"][lang], model_name
58
+ )
59
+
60
+
61
+ class TwoHopIncrementalResponse(BaseModel):
62
+ answer: str
63
+ one_hop_question: str
64
+ two_hop_question: str
65
+
66
+
67
+ async def two_hop_incremental(
68
+ row: Dict,
69
+ client: AsyncClient,
70
+ model_name: str = "gpt-4o-2024-08-06",
71
+ lang: str = "en",
72
+ ) -> Dict:
73
+ """
74
+ Create a two-hop question using incremental prompt.
75
+ Incremental prompt is more effective to create multi-hop question.
76
+ The input retrieval_gt has to include more than one passage.
77
+
78
+ :return: The two-hop question using openai incremental prompt
79
+ """
80
+ messages = QUERY_GEN_PROMPT["two_hop_incremental"][lang]
81
+ passages = row["retrieval_gt_contents"]
82
+ assert len(passages) >= 2, (
83
+ "You have to sample more than two passages for making two-hop questions."
84
+ )
85
+ context_str = f"Document 1: {passages[0][0]}\nDocument 2: {passages[1][0]}"
86
+ user_prompt = f"{context_str}\n\nGenerated two-hop Question from two Documents:\n"
87
+ messages.append(ChatMessage(role=MessageRole.USER, content=user_prompt))
88
+
89
+ completion = await client.beta.chat.completions.parse(
90
+ model=model_name,
91
+ messages=to_openai_message_dicts(messages),
92
+ response_format=TwoHopIncrementalResponse,
93
+ )
94
+ row["query"] = completion.choices[0].message.parsed.two_hop_question
95
+ return row
@@ -0,0 +1,201 @@
1
+ from llama_index.core.base.llms.types import ChatMessage, MessageRole
2
+
3
+ QUERY_GEN_PROMPT = {
4
+ "factoid_single_hop": {
5
+ "en": [
6
+ ChatMessage(
7
+ role=MessageRole.SYSTEM,
8
+ content="""You're an AI tasked to convert Text into a factoid question.
9
+ Factoid questions are those seeking brief, factual information that can be easily verified. They typically require a yes or no answer or a brief explanation and often inquire about specific details such as dates, names, places, or events.
10
+
11
+ Examples of factoid questions include:
12
+
13
+ - What is the capital of France?
14
+ - Who invented the light bulb?
15
+ - When was Wikipedia founded?
16
+
17
+ Instructions:
18
+ 1. Questions MUST BE extracted from given Text
19
+ 2. Questions should be as detailed as possible from Text
20
+ 3. Create questions that ask about factual information from the Text
21
+ 4. Do not mention any of these in the questions: "in the given text", "in the provided information", etc.
22
+ Users do not know the passage source of the question, so it should not be mentioned in the question.
23
+ 5. Do not ask about the file name or the file title. Ask about the content of the file.
24
+ For example, avoid to write questions like `What is the file name of the document?`""",
25
+ )
26
+ ],
27
+ "ko": [
28
+ ChatMessage(
29
+ role=MessageRole.SYSTEM,
30
+ content="""당신은 주어진 Text를 '사실 질문'으로 변환하는 AI입니다.
31
+
32
+ 사실 질문(factoid questions)이란 사실적인 정보를 요구하는 질문으로, 쉽게 검증할 수 있는 답변을 필요로 합니다. 일반적으로 예/아니오 답변이나 간단한 설명을 요구하며, 날짜, 이름, 장소 또는 사건과 같은 구체적인 세부사항에 대해 묻는 질문입니다.
33
+
34
+ 사실 질문의 예는 다음과 같습니다:
35
+
36
+ • 프랑스의 수도는 어디입니까?
37
+ • 전구를 발명한 사람은 누구입니까?
38
+ • 위키피디아는 언제 설립되었습니까?
39
+
40
+ 지침:
41
+ 1. 질문은 반드시 주어진 Text를 기반으로 작성되어야 합니다.
42
+ 2. 질문은 Text를 기반으로 가능한 한 구체적으로 작성되어야 합니다.
43
+ 3. Text에서 사실적 정보를 요구하는 질문을 만들어야 합니다. 즉, Text를 기반으로 사실 질문을 만드세요.
44
+ 4. 질문에 “주어진 Text에서” 또는 “제공된 단락에서”와 같은 표현을 포함해서는 안 됩니다.
45
+ 사용자는 질문의 출처가 Text라는 것을 모르기 때문에 반드시 그 출처를 언급해서는 안 됩니다.
46
+ 5. 파일 이름이나 파일 제목에 대한 질문을 하지 마세요. 파일의 내용에 대해 물어보세요.
47
+ 예를 들어, '문서의 파일 이름은 무엇입니까?'와 같은 질문을 작성하지 마세요.
48
+ 6. 질문을 한국어로 작성하세요.""",
49
+ )
50
+ ],
51
+ "ja": [
52
+ ChatMessage(
53
+ role=MessageRole.SYSTEM,
54
+ content="""あなたは与えられたTextを「実は質問」に変換するAIです。
55
+
56
+ 事実質問(factoid questions)とは、事実的な情報を求める質問であり、容易に検証できる回答を必要とします。 一般的に、「はい/いいえ」の返答や簡単な説明を要求し、日付、名前、場所、または事件のような具体的な詳細事項について尋ねる質問です。
57
+
58
+ 事実質問の例は下記の通りです:
59
+
60
+ • フランスの首都はどこですか?
61
+ • 電球を発明したのは誰ですか?
62
+ • ウィキペディアはいつ設立されましたか?
63
+
64
+ 指針:
65
+ 1. 質問は、必ず与えられたTextに基づいて作成されなければなりません。
66
+ 2. 質問は、Textに基づいて可能な限り具体的に作成されなければなりません。
67
+ 3. Textで事実的な情報を要求する質問を作らなければなりません。 つまり、Textに基づいて事実質問を作成してください。
68
+ 4. 質問に「与えられたTextで」または「提供された段落で」のような表現を含めてはいけません。
69
+ ユーザーは質問の出所がTextだということを知らないので、絶対にその出所を言及してはいけません。
70
+ 5. ファイル名やファイルのタイトルを訊かないでください。代わりに、ファイルの内容について聞いてください。
71
+ 例えば、「このドキュメントのファイル名は何ですか?」や「ファイル名は何ですか?」というような質問は書かないでください。
72
+ 6. 質問を日本語で作成してください。""",
73
+ )
74
+ ],
75
+ },
76
+ "concept_completion": {
77
+ "en": [
78
+ ChatMessage(
79
+ role=MessageRole.SYSTEM,
80
+ content="""You're an AI tasked to convert Text into a "Concept Completion" Question.
81
+ A “concept completion” question asks directly about the essence or identity of a concept.
82
+
83
+ Follow the following instructions.
84
+ Instructions:
85
+ 1. Questions MUST BE extracted from given Text
86
+ 2. Questions should be as detailed as possible from Text
87
+ 3. Create questions that ask about information from the Text
88
+ 4. MUST include specific keywords from the Text.
89
+ 5. Do not mention any of these in the questions: "in the given text", "in the provided information", etc.
90
+ Users do not know the passage source of the question, so it should not be mentioned in the question.
91
+ 6. Do not ask about the file name or the file title. Ask about the content of the file.
92
+ For example, avoid to write questions like `What is the file name of the document?""",
93
+ )
94
+ ],
95
+ "ko": [
96
+ ChatMessage(
97
+ role=MessageRole.SYSTEM,
98
+ content="""당신은 Text를 “개념 완성” 질문으로 변환하는 AI입니다.
99
+ "개념 완성" 질문은 개념의 본질이나 정체성에 대해 직접적으로 묻는 질문입니다.
100
+
101
+ 다음 지시사항을 따르세요.
102
+ 지시사항:
103
+ 1. 질문은 반드시 주어진 Text를 기반으로 작성되어야 합니다.
104
+ 2. 질문은 Text를 기반으로 가능한 한 자세하게 작성되어야 합니다.
105
+ 3. Text에서 제공된 정보를 묻는 질문을 생성하세요.
106
+ 4. Text의 특정 키워드를 반드시 질문에 포함하세요.
107
+ 5. 질문에 “주어진 Text에서” 또는 “제공된 단락에서”와 같은 표현을 포함해서는 안 됩니다.
108
+ 사용자는 질문의 출처가 Text라는 것을 모르기 때문에 반드시 그 출처를 언급해서는 안 됩니다.
109
+ 6. 파일 이름이나 파일 제목에 대한 질문을 하지 마세요. 파일의 내용에 대해 물어보세요.
110
+ 예를 들어, '문서의 파일 이름은 무엇입니까?'와 같은 질문을 작성하지 마세요.
111
+ 7. 질문을 한국어로 작성하세요.""",
112
+ )
113
+ ],
114
+ "ja": [
115
+ ChatMessage(
116
+ role=MessageRole.SYSTEM,
117
+ content="""あなたはTextを「概念完成」の質問に変換するAIです。
118
+ 「概念完成」の質問は概念の本質やアイデンティティについて直接に尋ねる質問です。
119
+
120
+ 次の指示に従ってください。
121
+ 指示事項:
122
+ 1. 質問は、必ず与えられたTextに基づいて作成されなければなりません。
123
+ 2. 質問は、Textに基づいてできるだけ詳しく作成されなければなりません。
124
+ 3. Textで提供された情報を尋ねる質問を作成してください。
125
+ 4. Textの特定のキーワードを必ず質問に含めてください。
126
+ 5. 質問に「与えられたTextで」または「提供された段落で」のような表現を含めてはいけません。
127
+ ユーザーは質問の出所がTextだということを知らないので、絶対にその出所を言及してはいけません。
128
+ 6. ファイル名やファイルのタイトルを訊かないでください。代わりに、ファイルの内容について聞いてください。
129
+ 例えば、「このドキュメントのファイル名は何ですか?」や「ファイル名は何ですか?」というような質問は書かないでください。
130
+ 7. 質問を日本語で作成してください。""",
131
+ )
132
+ ],
133
+ },
134
+ "two_hop_incremental": {
135
+ "en": [
136
+ ChatMessage(
137
+ role=MessageRole.SYSTEM,
138
+ content="Generate a multi-hop question for the given answer which requires reference to all of the given documents.",
139
+ ),
140
+ ChatMessage(
141
+ role=MessageRole.USER,
142
+ content="""Document 1: The Municipality of Nuevo Laredo is located in the Mexican state of Tamaulipas.
143
+ Document 2: The Ciudad Deportiva (Sports City ¨ ¨) is a sports
144
+ complex in Nuevo Laredo, Mexico. It is home to the Tecolotes de
145
+ Nuevo Laredo Mexican Baseball League team and ...""",
146
+ ),
147
+ ChatMessage(
148
+ role=MessageRole.ASSISTANT,
149
+ content="""Answer: Tamaulipas
150
+ One-hop question (using Document 1): In which Mexican state is Nuevo Laredo located?
151
+ Two-hop question (using Document 2): In which Mexican state can one find the Ciudad Deportiva, home to the Tecolotes de Nuevo Laredo?""",
152
+ ),
153
+ ],
154
+ "ko": [
155
+ ChatMessage(
156
+ role=MessageRole.SYSTEM,
157
+ content="Generate a multi-hop question for the given answer which requires reference to all of the given documents.",
158
+ ),
159
+ ChatMessage(
160
+ role=MessageRole.USER,
161
+ content="""Document 1: The Municipality of Nuevo Laredo is located in the Mexican state of Tamaulipas.
162
+ Document 2: The Ciudad Deportiva (Sports City ¨ ¨) is a sports
163
+ complex in Nuevo Laredo, Mexico. It is home to the Tecolotes de
164
+ Nuevo Laredo Mexican Baseball League team and ...""",
165
+ ),
166
+ ChatMessage(
167
+ role=MessageRole.ASSISTANT,
168
+ content="""Answer: Tamaulipas
169
+ One-hop question (using Document 1): In which Mexican state is Nuevo Laredo located?
170
+ Two-hop question (using Document 2): In which Mexican state can one find the Ciudad Deportiva, home to the Tecolotes de Nuevo Laredo?""",
171
+ ),
172
+ ],
173
+ "ja": [
174
+ ChatMessage(
175
+ role=MessageRole.SYSTEM,
176
+ content="与えられた答えに対するマルチホップ質問を生成し、与えられたすべての文書を参照する必要があります。",
177
+ ),
178
+ ChatMessage(
179
+ role=MessageRole.USER,
180
+ content="""Document 1: ヌエヴォ·ラレド自治体はメキシコのタマウリパス州にあります。
181
+ シウダー・デポルティーバ(スポーツ・シティ)は、
182
+ メキシコのヌエボ・ラレドにある複合スポーツ施設です。""",
183
+ ),
184
+ ChatMessage(
185
+ role=MessageRole.ASSISTANT,
186
+ content="""Answer: Tamaulipas
187
+ One-hop question (using Document 1): ヌエヴォ·ラレド自治体はどのメキシコの州にありますか?
188
+ Two-hop question (using Document 2): ヌエヴォ·ラレドのテコロテス·デ·テコロテスの故郷であるメキシコの州はどこですか?""",
189
+ ),
190
+ ],
191
+ },
192
+ }
193
+
194
+ # Experimental feature
195
+ QUERY_GEN_PROMPT_EXTRA = {
196
+ "multiple_queries": {
197
+ "en": "\nAdditional instructions:\n - Please make {n} questions.",
198
+ "ko": "\n추가 지침:\n - 질문은 {n}개를 만드세요.",
199
+ "ja": "\n追加指示:\n - 質問を{n}個作成してください。",
200
+ }
201
+ }
@@ -0,0 +1,26 @@
1
+ import uuid
2
+ from typing import Iterable
3
+
4
+ import pandas as pd
5
+
6
+
7
+ def random_single_hop(
8
+ corpus_df: pd.DataFrame, n: int, random_state: int = 42
9
+ ) -> pd.DataFrame:
10
+ sample_df = corpus_df.sample(n, random_state=random_state)
11
+ return pd.DataFrame(
12
+ {
13
+ "qid": [str(uuid.uuid4()) for _ in range(len(sample_df))],
14
+ "retrieval_gt": [[[id_]] for id_ in sample_df["doc_id"].tolist()],
15
+ }
16
+ )
17
+
18
+
19
+ def range_single_hop(corpus_df: pd.DataFrame, idx_range: Iterable):
20
+ sample_df = corpus_df.iloc[idx_range]
21
+ return pd.DataFrame(
22
+ {
23
+ "qid": [str(uuid.uuid4()) for _ in range(len(sample_df))],
24
+ "retrieval_gt": [[[id_]] for id_ in sample_df["doc_id"].tolist()],
25
+ }
26
+ )