haystack-ai 3.1.1__tar.gz → 3.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (299) hide show
  1. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/PKG-INFO +1 -1
  2. haystack_ai-3.2.0/VERSION.txt +1 -0
  3. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/agents/agent.py +66 -26
  4. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/builders/chat_prompt_builder.py +1 -1
  5. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/converters/csv.py +3 -2
  6. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/converters/image/document_to_image.py +17 -10
  7. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/converters/json.py +1 -1
  8. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/converters/markdown.py +1 -1
  9. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/converters/multi_file_converter.py +2 -2
  10. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/converters/output_adapter.py +5 -1
  11. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/converters/txt.py +3 -2
  12. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/converters/xlsx.py +11 -2
  13. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/embedders/openai_document_embedder.py +1 -1
  14. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/extractors/image/llm_document_content_extractor.py +43 -32
  15. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/extractors/llm_metadata_extractor.py +18 -23
  16. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/fetchers/link_content.py +10 -5
  17. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/generators/chat/azure.py +1 -1
  18. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/generators/chat/azure_responses.py +7 -4
  19. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/generators/chat/fallback.py +19 -13
  20. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/generators/chat/openai.py +2 -3
  21. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/generators/chat/openai_responses.py +44 -14
  22. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/generators/openai_image_generator.py +2 -0
  23. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/generators/utils.py +5 -2
  24. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/joiners/answer_joiner.py +17 -5
  25. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/joiners/document_joiner.py +9 -4
  26. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/preprocessors/document_cleaner.py +18 -0
  27. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/preprocessors/document_preprocessor.py +9 -2
  28. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/preprocessors/document_splitter.py +136 -11
  29. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/preprocessors/embedding_based_document_splitter.py +10 -0
  30. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/preprocessors/markdown_header_splitter.py +94 -59
  31. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/preprocessors/python_code_splitter.py +3 -2
  32. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/preprocessors/recursive_splitter.py +36 -18
  33. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/preprocessors/text_cleaner.py +16 -1
  34. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/query/query_expander.py +8 -8
  35. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/rankers/llm_ranker.py +2 -2
  36. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/rankers/meta_field.py +2 -3
  37. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/rankers/meta_field_grouping_ranker.py +1 -1
  38. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/retrievers/auto_merging_retriever.py +4 -2
  39. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/retrievers/filter_retriever.py +6 -2
  40. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/retrievers/multi_query_embedding_retriever.py +10 -2
  41. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/retrievers/multi_query_text_retriever.py +10 -2
  42. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/retrievers/multi_retriever.py +10 -6
  43. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/routers/conditional_router.py +5 -1
  44. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/routers/document_type_router.py +9 -3
  45. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/routers/file_type_router.py +18 -1
  46. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/routers/llm_messages_router.py +14 -2
  47. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/routers/metadata_router.py +6 -1
  48. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/samplers/top_p.py +6 -2
  49. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/validators/json_schema.py +5 -5
  50. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/core/pipeline/base.py +108 -11
  51. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/core/pipeline/breakpoint.py +12 -2
  52. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/core/pipeline/draw.py +1 -1
  53. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/core/super_component/super_component.py +4 -1
  54. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/core/type_utils.py +19 -11
  55. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/document_stores/in_memory/document_store.py +33 -5
  56. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/document_stores/types/filter_policy.py +10 -6
  57. haystack_ai-3.2.0/haystack/hooks/budget/__init__.py +15 -0
  58. haystack_ai-3.2.0/haystack/hooks/budget/hooks.py +100 -0
  59. haystack_ai-3.2.0/haystack/hooks/compaction/AGENTS.md +10 -0
  60. haystack_ai-3.2.0/haystack/hooks/compaction/CLAUDE.md +1 -0
  61. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/hooks/compaction/__init__.py +2 -0
  62. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/hooks/compaction/hooks.py +35 -7
  63. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/hooks/compaction/sliding_window.py +1 -0
  64. haystack_ai-3.2.0/haystack/hooks/compaction/summarization.py +511 -0
  65. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/hooks/compaction/tool_result_pruning.py +1 -0
  66. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/hooks/compaction/utils.py +0 -22
  67. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/hooks/tool_result_offloading/hooks.py +134 -85
  68. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/hooks/tool_result_offloading/stores.py +25 -8
  69. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/hooks/tool_result_offloading/types/protocol.py +28 -9
  70. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/logging.py +27 -0
  71. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/skill_stores/file_system/skill_store.py +28 -10
  72. haystack_ai-3.2.0/haystack/testing/telemetry.py +77 -0
  73. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/token_counters/utils.py +33 -8
  74. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/tools/agent_tool.py +9 -3
  75. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/tools/from_function.py +12 -7
  76. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/tools/searchable_toolset.py +1 -5
  77. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/tools/skills/skill_toolset.py +1 -8
  78. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/tools/toolset.py +8 -156
  79. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/utils/auth.py +3 -3
  80. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/utils/azure.py +1 -1
  81. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/utils/device.py +13 -4
  82. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/utils/filters.py +3 -1
  83. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/pyproject.toml +1 -1
  84. haystack_ai-3.1.1/VERSION.txt +0 -1
  85. haystack_ai-3.1.1/haystack/components/generators/chat/utils.py +0 -41
  86. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/.gitignore +0 -0
  87. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/LICENSE +0 -0
  88. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/README.md +0 -0
  89. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/__init__.py +0 -0
  90. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/__init__.py +0 -0
  91. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/agents/__init__.py +0 -0
  92. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/agents/state/__init__.py +0 -0
  93. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/agents/state/state.py +0 -0
  94. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/agents/state/state_utils.py +0 -0
  95. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/agents/tool_calling.py +0 -0
  96. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/agents/utils.py +0 -0
  97. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/builders/__init__.py +0 -0
  98. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/builders/answer_builder.py +0 -0
  99. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/builders/prompt_builder.py +0 -0
  100. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/caching/__init__.py +0 -0
  101. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/caching/cache_checker.py +0 -0
  102. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/converters/__init__.py +0 -0
  103. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/converters/docx.py +0 -0
  104. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/converters/file_to_file_content.py +0 -0
  105. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/converters/html.py +0 -0
  106. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/converters/image/__init__.py +0 -0
  107. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/converters/image/file_to_document.py +0 -0
  108. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/converters/image/file_to_image.py +0 -0
  109. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/converters/image/image_utils.py +0 -0
  110. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/converters/image/pdf_to_image.py +0 -0
  111. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/converters/msg.py +0 -0
  112. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/converters/pdfminer.py +0 -0
  113. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/converters/pptx.py +0 -0
  114. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/converters/pypdf.py +0 -0
  115. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/converters/utils.py +0 -0
  116. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/embedders/__init__.py +0 -0
  117. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/embedders/azure_document_embedder.py +0 -0
  118. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/embedders/azure_text_embedder.py +0 -0
  119. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/embedders/mock_document_embedder.py +0 -0
  120. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/embedders/mock_text_embedder.py +0 -0
  121. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/embedders/mock_utils.py +0 -0
  122. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/embedders/openai_text_embedder.py +0 -0
  123. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/embedders/types/__init__.py +0 -0
  124. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/embedders/types/protocol.py +0 -0
  125. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/evaluators/__init__.py +0 -0
  126. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/evaluators/answer_exact_match.py +0 -0
  127. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/evaluators/context_relevance.py +0 -0
  128. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/evaluators/document_map.py +0 -0
  129. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/evaluators/document_mrr.py +0 -0
  130. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/evaluators/document_ndcg.py +0 -0
  131. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/evaluators/document_recall.py +0 -0
  132. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/evaluators/faithfulness.py +0 -0
  133. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/evaluators/llm_evaluator.py +0 -0
  134. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/evaluators/sas_evaluator.py +0 -0
  135. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/extractors/__init__.py +0 -0
  136. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/extractors/image/__init__.py +0 -0
  137. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/extractors/regex_text_extractor.py +0 -0
  138. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/fetchers/__init__.py +0 -0
  139. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/generators/__init__.py +0 -0
  140. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/generators/chat/__init__.py +0 -0
  141. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/generators/chat/llm.py +0 -0
  142. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/generators/chat/mock.py +0 -0
  143. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/generators/chat/types/__init__.py +0 -0
  144. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/generators/chat/types/protocol.py +0 -0
  145. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/joiners/__init__.py +0 -0
  146. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/joiners/branch.py +0 -0
  147. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/joiners/list_joiner.py +0 -0
  148. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/joiners/string_joiner.py +0 -0
  149. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/preprocessors/__init__.py +0 -0
  150. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/preprocessors/csv_document_cleaner.py +0 -0
  151. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/preprocessors/csv_document_splitter.py +0 -0
  152. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/preprocessors/hierarchical_document_splitter.py +0 -0
  153. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/preprocessors/sentence_tokenizer.py +0 -0
  154. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/query/__init__.py +0 -0
  155. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/rankers/__init__.py +0 -0
  156. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/rankers/lost_in_the_middle.py +0 -0
  157. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/retrievers/__init__.py +0 -0
  158. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/retrievers/in_memory/__init__.py +0 -0
  159. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/retrievers/in_memory/bm25_retriever.py +0 -0
  160. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/retrievers/in_memory/embedding_retriever.py +0 -0
  161. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/retrievers/sentence_window_retriever.py +0 -0
  162. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/retrievers/text_embedding_retriever.py +0 -0
  163. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/retrievers/types/__init__.py +0 -0
  164. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/retrievers/types/protocol.py +0 -0
  165. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/routers/__init__.py +0 -0
  166. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/routers/document_length_router.py +0 -0
  167. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/samplers/__init__.py +0 -0
  168. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/validators/__init__.py +0 -0
  169. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/writers/__init__.py +0 -0
  170. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/writers/document_writer.py +0 -0
  171. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/core/__init__.py +0 -0
  172. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/core/component/__init__.py +0 -0
  173. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/core/component/component.py +0 -0
  174. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/core/component/sockets.py +0 -0
  175. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/core/component/types.py +0 -0
  176. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/core/errors.py +0 -0
  177. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/core/pipeline/__init__.py +0 -0
  178. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/core/pipeline/component_checks.py +0 -0
  179. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/core/pipeline/descriptions.py +0 -0
  180. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/core/pipeline/pipeline.py +0 -0
  181. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/core/pipeline/utils.py +0 -0
  182. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/core/serialization.py +0 -0
  183. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/core/serialization_security.py +0 -0
  184. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/core/super_component/__init__.py +0 -0
  185. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/core/super_component/utils.py +0 -0
  186. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/data/abbreviations/de.txt +0 -0
  187. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/data/abbreviations/en.txt +0 -0
  188. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/dataclasses/__init__.py +0 -0
  189. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/dataclasses/answer.py +0 -0
  190. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/dataclasses/breakpoints.py +0 -0
  191. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/dataclasses/byte_stream.py +0 -0
  192. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/dataclasses/chat_message.py +0 -0
  193. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/dataclasses/document.py +0 -0
  194. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/dataclasses/file_content.py +0 -0
  195. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/dataclasses/image_content.py +0 -0
  196. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/dataclasses/skill_info.py +0 -0
  197. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/dataclasses/sparse_embedding.py +0 -0
  198. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/dataclasses/streaming_chunk.py +0 -0
  199. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/document_stores/__init__.py +0 -0
  200. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/document_stores/errors/__init__.py +0 -0
  201. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/document_stores/errors/errors.py +0 -0
  202. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/document_stores/in_memory/__init__.py +0 -0
  203. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/document_stores/types/__init__.py +0 -0
  204. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/document_stores/types/policy.py +0 -0
  205. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/document_stores/types/protocol.py +0 -0
  206. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/errors.py +0 -0
  207. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/evaluation/__init__.py +0 -0
  208. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/evaluation/eval_run_result.py +0 -0
  209. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/hooks/__init__.py +0 -0
  210. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/hooks/compaction/types/__init__.py +0 -0
  211. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/hooks/compaction/types/protocol.py +0 -0
  212. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/hooks/from_function.py +0 -0
  213. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/hooks/human_in_the_loop/__init__.py +0 -0
  214. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/hooks/human_in_the_loop/dataclasses.py +0 -0
  215. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/hooks/human_in_the_loop/hooks.py +0 -0
  216. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/hooks/human_in_the_loop/policies.py +0 -0
  217. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/hooks/human_in_the_loop/strategies.py +0 -0
  218. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/hooks/human_in_the_loop/types/__init__.py +0 -0
  219. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/hooks/human_in_the_loop/types/protocol.py +0 -0
  220. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/hooks/human_in_the_loop/user_interfaces.py +0 -0
  221. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/hooks/invocation.py +0 -0
  222. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/hooks/protocol.py +0 -0
  223. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/hooks/tool_result_offloading/__init__.py +0 -0
  224. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/hooks/tool_result_offloading/policies.py +0 -0
  225. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/hooks/tool_result_offloading/types/__init__.py +0 -0
  226. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/hooks/utils.py +0 -0
  227. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/lazy_imports.py +0 -0
  228. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/marshal/__init__.py +0 -0
  229. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/marshal/protocol.py +0 -0
  230. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/marshal/yaml.py +0 -0
  231. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/py.typed +0 -0
  232. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/skill_stores/__init__.py +0 -0
  233. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/skill_stores/file_system/__init__.py +0 -0
  234. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/skill_stores/types/__init__.py +0 -0
  235. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/skill_stores/types/protocol.py +0 -0
  236. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/telemetry/__init__.py +0 -0
  237. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/telemetry/_environment.py +0 -0
  238. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/telemetry/_telemetry.py +0 -0
  239. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/testing/__init__.py +0 -0
  240. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/testing/callable_serialization/random_callable.py +0 -0
  241. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/testing/document_store.py +0 -0
  242. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/testing/document_store_async.py +0 -0
  243. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/testing/factory.py +0 -0
  244. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/testing/sample_components/__init__.py +0 -0
  245. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/testing/sample_components/accumulate.py +0 -0
  246. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/testing/sample_components/add_value.py +0 -0
  247. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/testing/sample_components/concatenate.py +0 -0
  248. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/testing/sample_components/double.py +0 -0
  249. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/testing/sample_components/fstring.py +0 -0
  250. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/testing/sample_components/future_annotations.py +0 -0
  251. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/testing/sample_components/greet.py +0 -0
  252. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/testing/sample_components/hello.py +0 -0
  253. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/testing/sample_components/joiner.py +0 -0
  254. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/testing/sample_components/parity.py +0 -0
  255. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/testing/sample_components/remainder.py +0 -0
  256. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/testing/sample_components/repeat.py +0 -0
  257. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/testing/sample_components/subtract.py +0 -0
  258. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/testing/sample_components/sum.py +0 -0
  259. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/testing/sample_components/text_splitter.py +0 -0
  260. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/testing/sample_components/threshold.py +0 -0
  261. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/testing/test_utils.py +0 -0
  262. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/token_counters/__init__.py +0 -0
  263. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/token_counters/approximate_counter.py +0 -0
  264. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/token_counters/openai_counter.py +0 -0
  265. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/token_counters/tiktoken_counter.py +0 -0
  266. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/token_counters/types/__init__.py +0 -0
  267. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/token_counters/types/protocol.py +0 -0
  268. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/tools/__init__.py +0 -0
  269. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/tools/component_tool.py +0 -0
  270. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/tools/errors.py +0 -0
  271. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/tools/parameters_schema_utils.py +0 -0
  272. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/tools/pipeline_tool.py +0 -0
  273. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/tools/serde_utils.py +0 -0
  274. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/tools/skills/__init__.py +0 -0
  275. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/tools/tool.py +0 -0
  276. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/tools/tool_types.py +0 -0
  277. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/tools/utils.py +0 -0
  278. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/tracing/__init__.py +0 -0
  279. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/tracing/logging_tracer.py +0 -0
  280. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/tracing/tracer.py +0 -0
  281. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/tracing/utils.py +0 -0
  282. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/utils/__init__.py +0 -0
  283. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/utils/async_utils.py +0 -0
  284. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/utils/base_serialization.py +0 -0
  285. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/utils/callable_serialization.py +0 -0
  286. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/utils/dataclasses.py +0 -0
  287. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/utils/deserialization.py +0 -0
  288. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/utils/experimental.py +0 -0
  289. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/utils/hf.py +0 -0
  290. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/utils/http_client.py +0 -0
  291. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/utils/jinja2_chat_extension.py +0 -0
  292. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/utils/jinja2_extensions.py +0 -0
  293. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/utils/jinja2_sandbox.py +0 -0
  294. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/utils/jupyter.py +0 -0
  295. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/utils/misc.py +0 -0
  296. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/utils/requests_utils.py +0 -0
  297. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/utils/type_serialization.py +0 -0
  298. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/utils/url_validation.py +0 -0
  299. {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/version.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: haystack-ai
3
- Version: 3.1.1
3
+ Version: 3.2.0
4
4
  Summary: LLM framework to build customizable, production-ready LLM applications. Connect components (models, vector DBs, file converters) to pipelines or agents that can interact with your data.
5
5
  Project-URL: CI: GitHub, https://github.com/deepset-ai/haystack/actions
6
6
  Project-URL: Docs: RTD, https://haystack.deepset.ai/overview/intro
@@ -0,0 +1 @@
1
+ 3.2.0
@@ -65,9 +65,11 @@ from haystack.utils.deserialization import deserialize_component_inplace
65
65
 
66
66
  logger = logging.getLogger(__name__)
67
67
 
68
- # `exit_reason` values the Agent sets when it stops without a tool exit condition: a tool-call-free reply, or the
69
- # `max_agent_steps` budget running out. A tool exit condition instead reports the tool's name.
68
+ # `exit_reason` values the Agent sets when it stops without a tool exit condition: a tool-call-free reply, an
69
+ # incomplete model generation, or the `max_agent_steps` budget running out.
70
70
  _EXIT_REASON_TEXT = "text"
71
+ _EXIT_REASON_LENGTH = "length"
72
+ _EXIT_REASON_CONTENT_FILTER = "content_filter"
71
73
  _EXIT_REASON_MAX_STEPS = "max_agent_steps"
72
74
 
73
75
  # Run-metadata state keys the Agent populates automatically during a run. Users may not define them in their own
@@ -82,6 +84,7 @@ _RUN_METADATA_STATE_KEYS: dict[str, dict[str, Any]] = {
82
84
  # Internal state keys the Agent manages for run control and hooks. Like run-metadata keys they are reserved and cannot
83
85
  # be redefined by users, but unlike them they are NOT exposed as Agent inputs or outputs (purely internal state):
84
86
  # - `continue_run`: set by an `on_exit` hook to keep the Agent running instead of stopping (re-read each exit attempt).
87
+ # - `stop_run`: set by a hook to stop the run, read before each LLM call and used as the `exit_reason`.
85
88
  # - `tools`: the flattened tools available in the current step, so a hook can inspect them (e.g. HITL confirmation).
86
89
  # - `hook_context`: per-run request-scoped resources passed to `run`/`run_async` for hooks to read.
87
90
  # - `context_tokens`: approximate current context-window size, refreshed after each LLM call, for hooks to read
@@ -89,6 +92,7 @@ _RUN_METADATA_STATE_KEYS: dict[str, dict[str, Any]] = {
89
92
  # exposed as an output because it is a best-effort snapshot; see `_record_context_tokens`.
90
93
  _INTERNAL_STATE_KEYS: dict[str, dict[str, Any]] = {
91
94
  "continue_run": {"type": bool, "handler": replace_values},
95
+ "stop_run": {"type": str, "handler": replace_values},
92
96
  "tools": {"type": list, "handler": replace_values},
93
97
  "hook_context": {"type": dict[str, Any], "handler": replace_values},
94
98
  "context_tokens": {"type": int, "handler": replace_values},
@@ -147,17 +151,36 @@ def _consume_continue_run(state: State) -> bool:
147
151
  return should_continue
148
152
 
149
153
 
150
- def _is_text_exit(messages: list[ChatMessage]) -> bool:
154
+ def _get_model_exit_reason(messages: list[ChatMessage]) -> str | None:
151
155
  """
152
- Return whether `messages` end in a plain assistant text reply with no tool calls anywhere in the batch.
156
+ Return the exit reason for a terminal assistant reply without tool calls.
153
157
 
154
- This is the "no tool call" exit for the model's own replies. The last message must be a non-empty assistant text
155
- message, so an invalid response (e.g. one with no tool calls and no text) does not trigger an exit.
158
+ Incomplete generation reasons take precedence over text so callers can distinguish a partial response from a
159
+ complete answer. An empty response without a recognized terminal reason does not trigger an exit, preserving the
160
+ Agent's recovery behavior for malformed tool calls that a Chat Generator discarded.
156
161
  """
157
- if not messages:
158
- return False
162
+ # If the messages list is empty or the last message has tool calls, don't exit.
163
+ if not messages or any(message.tool_call for message in messages):
164
+ return None
165
+
159
166
  last = messages[-1]
160
- return not any(m.tool_call for m in messages) and last.is_from(ChatRole.ASSISTANT) and bool(last.text)
167
+
168
+ # If the last message is not from the assistant, don't exit.
169
+ if not last.is_from(ChatRole.ASSISTANT):
170
+ return None
171
+
172
+ # If the finish reason on the last message is length or content_filter, exit with that reason.
173
+ if last.meta.get("finish_reason") == _EXIT_REASON_LENGTH:
174
+ return _EXIT_REASON_LENGTH
175
+ if last.meta.get("finish_reason") == _EXIT_REASON_CONTENT_FILTER:
176
+ return _EXIT_REASON_CONTENT_FILTER
177
+
178
+ # If the last message has text, exit with the text reason.
179
+ if last.text:
180
+ return _EXIT_REASON_TEXT
181
+
182
+ # If we reached here no valid exit reason was found, so don't exit.
183
+ return None
161
184
 
162
185
 
163
186
  def _pending_tool_call_messages_from_state(state: State) -> list[ChatMessage]:
@@ -435,9 +458,12 @@ class Agent:
435
458
  """
436
459
  # --- Validation ---
437
460
  self._chat_generator_supports_tools: bool = "tools" in inspect.signature(chat_generator.run).parameters
438
- # We use an explicit None check for tools b/c testing for truthiness calls __len__, which for SearchableToolset
439
- # would iterate and prematurely warm it up at init.
440
- if tools is not None and not self._chat_generator_supports_tools:
461
+ # An empty list carries no tools, so it must not trip this check: `tools` is normalized to `[]` below, and
462
+ # both `clone()` and `to_dict()` feed that normalized value straight back into `__init__`. This mirrors the
463
+ # equivalent check in `run()`. Only a list is measured; a Toolset is never tested for truthiness here b/c
464
+ # that calls __len__, which for SearchableToolset would iterate and prematurely warm it up at init.
465
+ tools_provided = tools is not None and (not isinstance(tools, list) or len(tools) > 0)
466
+ if tools_provided and not self._chat_generator_supports_tools:
441
467
  raise TypeError(
442
468
  f"{type(chat_generator).__name__} does not accept tools parameter in its run method. "
443
469
  "The Agent component requires a chat generator that supports tools when tools are provided."
@@ -838,9 +864,11 @@ class Agent:
838
864
  `meta["usage"]`.
839
865
  - "tool_call_counts": Mapping of tool name to the number of times that tool was invoked.
840
866
  - "exit_reason": Why the Agent stopped, useful for routing the output downstream (e.g. with a
841
- `ConditionalRouter`). One of: `"text"` (the model returned a reply with no tool calls), the name of
842
- the tool that satisfied a tool exit condition (in which case `last_message` is that tool's result),
843
- or `"max_agent_steps"` (the Agent hit `max_agent_steps` before meeting an exit condition).
867
+ `ConditionalRouter`). One of: `"text"` (the model returned a complete reply with no tool calls),
868
+ `"length"` or `"content_filter"` (the model returned an incomplete reply, which may contain partial
869
+ text), the name of the tool that satisfied a tool exit condition (in which case `last_message` is that
870
+ tool's result), or `"max_agent_steps"` (the Agent hit `max_agent_steps` before meeting an exit
871
+ condition), or a custom reason a hook supplied through the `stop_run` state key.
844
872
  - Any additional keys defined in the `state_schema`.
845
873
  """
846
874
  agent_inputs = {"messages": messages, "streaming_callback": streaming_callback, **kwargs}
@@ -922,9 +950,11 @@ class Agent:
922
950
  `meta["usage"]`.
923
951
  - "tool_call_counts": Mapping of tool name to the number of times that tool was invoked.
924
952
  - "exit_reason": Why the Agent stopped, useful for routing the output downstream (e.g. with a
925
- `ConditionalRouter`). One of: `"text"` (the model returned a reply with no tool calls), the name of
926
- the tool that satisfied a tool exit condition (in which case `last_message` is that tool's result),
927
- or `"max_agent_steps"` (the Agent hit `max_agent_steps` before meeting an exit condition).
953
+ `ConditionalRouter`). One of: `"text"` (the model returned a complete reply with no tool calls),
954
+ `"length"` or `"content_filter"` (the model returned an incomplete reply, which may contain partial
955
+ text), the name of the tool that satisfied a tool exit condition (in which case `last_message` is that
956
+ tool's result), or `"max_agent_steps"` (the Agent hit `max_agent_steps` before meeting an exit
957
+ condition), or a custom reason a hook supplied through the `stop_run` state key.
928
958
  - Any additional keys defined in the `state_schema`.
929
959
  """
930
960
  agent_inputs = {"messages": messages, "streaming_callback": streaming_callback, **kwargs}
@@ -978,6 +1008,10 @@ class Agent:
978
1008
  exe_context.state.set("tools", current_tools, handler_override=replace_values)
979
1009
 
980
1010
  _run_hooks(hooks=self.hooks, hook_point=BEFORE_LLM, state=exe_context.state)
1011
+ # A hook requested a stop: end the run at the step boundary, before spending another LLM call.
1012
+ if (reason := exe_context.state.data.get("stop_run")) is not None:
1013
+ exe_context.state.set("exit_reason", reason)
1014
+ return False
981
1015
  chat_generator_inputs = {
982
1016
  "messages": exe_context.state.data["messages"],
983
1017
  **exe_context.chat_generator_inputs,
@@ -993,17 +1027,18 @@ class Agent:
993
1027
  _record_llm_usage(state=exe_context.state, llm_messages=llm_messages)
994
1028
  _record_context_tokens(state=exe_context.state, llm_messages=llm_messages)
995
1029
 
996
- # Stop on the "no tool call" exit: no tools available, or a plain assistant text reply (see _is_text_exit).
997
- if not current_tools or _is_text_exit(messages=llm_messages):
1030
+ # Stop when there are no tools, or the model produced a terminal reply without tool calls.
1031
+ model_exit_reason = _get_model_exit_reason(messages=llm_messages)
1032
+ if not current_tools or model_exit_reason is not None:
998
1033
  exe_context.counter += 1
999
1034
  exe_context.state.set("step_count", exe_context.counter)
1000
- exe_context.state.set("exit_reason", _EXIT_REASON_TEXT)
1035
+ exe_context.state.set("exit_reason", model_exit_reason or _EXIT_REASON_TEXT)
1001
1036
  return self._continue_after_exit_hooks(exe_context=exe_context)
1002
1037
 
1003
1038
  _run_hooks(hooks=self.hooks, hook_point=BEFORE_TOOL, state=exe_context.state)
1004
1039
  # Re-read the pending tool calls from State so that any rewrites a before_tool hook made (e.g.
1005
1040
  # ConfirmationHook rejecting or modifying calls) are honored by the executor.
1006
- pending_tool_call_messages = _pending_tool_call_messages_from_state(exe_context.state)
1041
+ pending_tool_call_messages = _pending_tool_call_messages_from_state(state=exe_context.state)
1007
1042
 
1008
1043
  tool_execution_inputs = {
1009
1044
  "messages": pending_tool_call_messages,
@@ -1041,6 +1076,10 @@ class Agent:
1041
1076
  exe_context.state.set("tools", current_tools, handler_override=replace_values)
1042
1077
 
1043
1078
  await _run_hooks_async(hooks=self.hooks, hook_point=BEFORE_LLM, state=exe_context.state)
1079
+ # A hook requested a stop: end the run at the step boundary, before spending another LLM call.
1080
+ if (reason := exe_context.state.data.get("stop_run")) is not None:
1081
+ exe_context.state.set("exit_reason", reason)
1082
+ return False
1044
1083
  chat_generator_inputs = {
1045
1084
  "messages": exe_context.state.data["messages"],
1046
1085
  **exe_context.chat_generator_inputs,
@@ -1058,17 +1097,18 @@ class Agent:
1058
1097
  _record_llm_usage(state=exe_context.state, llm_messages=llm_messages)
1059
1098
  _record_context_tokens(state=exe_context.state, llm_messages=llm_messages)
1060
1099
 
1061
- # Stop on the "no tool call" exit: no tools available, or a plain assistant text reply (see _is_text_exit).
1062
- if not current_tools or _is_text_exit(messages=llm_messages):
1100
+ # Stop when there are no tools, or the model produced a terminal reply without tool calls.
1101
+ model_exit_reason = _get_model_exit_reason(messages=llm_messages)
1102
+ if not current_tools or model_exit_reason is not None:
1063
1103
  exe_context.counter += 1
1064
1104
  exe_context.state.set("step_count", exe_context.counter)
1065
- exe_context.state.set("exit_reason", _EXIT_REASON_TEXT)
1105
+ exe_context.state.set("exit_reason", model_exit_reason or _EXIT_REASON_TEXT)
1066
1106
  return await self._continue_after_exit_hooks_async(exe_context=exe_context)
1067
1107
 
1068
1108
  await _run_hooks_async(hooks=self.hooks, hook_point=BEFORE_TOOL, state=exe_context.state)
1069
1109
  # Re-read the pending tool calls from State so that any rewrites a before_tool hook made (e.g.
1070
1110
  # ConfirmationHook rejecting or modifying calls) are honored by the executor.
1071
- pending_tool_call_messages = _pending_tool_call_messages_from_state(exe_context.state)
1111
+ pending_tool_call_messages = _pending_tool_call_messages_from_state(state=exe_context.state)
1072
1112
 
1073
1113
  tool_execution_inputs = {
1074
1114
  "messages": pending_tool_call_messages,
@@ -234,7 +234,7 @@ class ChatPromptBuilder:
234
234
  :returns: A dictionary with the following keys:
235
235
  - `prompt`: The updated list of `ChatMessage` objects after rendering the templates.
236
236
  :raises ValueError:
237
- If `chat_messages` is empty or contains elements that are not instances of `ChatMessage`.
237
+ If `template` is empty or contains elements that are not instances of `ChatMessage`.
238
238
  """
239
239
  kwargs = kwargs or {}
240
240
  template_variables = template_variables or {}
@@ -22,7 +22,8 @@ class CSVToDocument:
22
22
  """
23
23
  Converts CSV files to Documents.
24
24
 
25
- By default, it uses UTF-8 encoding when converting files but
25
+ By default, it uses UTF-8 encoding (`utf-8-sig`, which also strips a byte order mark if
26
+ present) when converting files but
26
27
  you can also set a custom encoding.
27
28
  It can attach metadata to the resulting documents.
28
29
 
@@ -44,7 +45,7 @@ class CSVToDocument:
44
45
 
45
46
  def __init__(
46
47
  self,
47
- encoding: str = "utf-8",
48
+ encoding: str = "utf-8-sig",
48
49
  store_full_path: bool = False,
49
50
  *,
50
51
  conversion_mode: Literal["file", "row"] = "file",
@@ -113,23 +113,30 @@ class DocumentToImageContent:
113
113
  :returns:
114
114
  Dictionary containing one key:
115
115
  - "image_contents": ImageContents created from the processed documents. These contain base64-encoded image
116
- data and metadata. The order corresponds to order of input documents.
117
- :raises ValueError:
118
- If any document is missing the required metadata keys, has an invalid file path, or has an unsupported
119
- MIME type. The error message will specify which document and what information is missing or incorrect.
116
+ data and metadata. The order corresponds to the order of the input documents. A document that is
117
+ missing the required metadata keys, has an invalid file path, or has an unsupported MIME type gets
118
+ None in its position and a logged warning with the reason.
120
119
  """
121
120
  if not documents:
122
121
  return {"image_contents": []}
123
122
 
124
- images_source_info = _extract_image_sources_info(
125
- documents=documents, file_path_meta_field=self.file_path_meta_field, root_path=self.root_path
126
- )
127
-
128
123
  image_contents: list[ImageContent | None] = [None] * len(documents)
129
124
 
130
125
  pdf_page_infos: list[_PDFPageInfo] = []
131
126
 
132
- for doc_idx, image_source_info in enumerate(images_source_info):
127
+ for doc_idx, document in enumerate(documents):
128
+ # Validate each document on its own so one invalid document leaves None in its slot
129
+ # instead of failing the whole batch
130
+ try:
131
+ image_source_info = _extract_image_sources_info(
132
+ documents=[document], file_path_meta_field=self.file_path_meta_field, root_path=self.root_path
133
+ )[0]
134
+ except ValueError as error:
135
+ logger.warning(
136
+ "Skipping document with ID {document_id}: {error}", document_id=document.id, error=str(error)
137
+ )
138
+ continue
139
+
133
140
  mime_type = image_source_info["mime_type"]
134
141
  path = image_source_info["path"]
135
142
  if mime_type == "application/pdf":
@@ -146,7 +153,7 @@ class DocumentToImageContent:
146
153
  base64_image=base64_image,
147
154
  mime_type=mime_type,
148
155
  detail=self.detail,
149
- meta={"file_path": documents[doc_idx].meta[self.file_path_meta_field]},
156
+ meta={"file_path": document.meta[self.file_path_meta_field]},
150
157
  )
151
158
 
152
159
  # efficiently convert PDF pages to images: each PDF is opened and processed only once
@@ -187,7 +187,7 @@ class JSONConverter:
187
187
  to a different document.
188
188
  """
189
189
  try:
190
- file_content = source.data.decode("utf-8")
190
+ file_content = source.data.decode("utf-8-sig")
191
191
  except UnicodeError as exc:
192
192
  logger.warning(
193
193
  "Failed to extract text from {source}. Skipping it. Error: {error}",
@@ -52,7 +52,7 @@ class MarkdownToDocument:
52
52
  table_to_single_line: bool = False,
53
53
  progress_bar: bool = True,
54
54
  store_full_path: bool = False,
55
- encoding: str = "utf-8",
55
+ encoding: str = "utf-8-sig",
56
56
  *,
57
57
  extract_frontmatter: bool = False,
58
58
  ) -> None:
@@ -52,14 +52,14 @@ class MultiFileConverter:
52
52
 
53
53
  Usage example:
54
54
  ```
55
- from haystack.super_components.converters import MultiFileConverter
55
+ from haystack.components.converters import MultiFileConverter
56
56
 
57
57
  converter = MultiFileConverter()
58
58
  converter.run(sources=["test/test_files/txt/doc_1.txt", "test/test_files/pdf/sample_pdf_1.pdf"], meta={})
59
59
  ```
60
60
  """
61
61
 
62
- def __init__(self, encoding: str = "utf-8", json_content_key: str = "content") -> None:
62
+ def __init__(self, encoding: str = "utf-8-sig", json_content_key: str = "content") -> None:
63
63
  """
64
64
  Initialize the MultiFileConverter.
65
65
 
@@ -136,8 +136,12 @@ class OutputAdapter:
136
136
  # we try to evaluate it and would fail.
137
137
  # This must be done cause the output could be different literal structures.
138
138
  # This doesn't support any user types.
139
+ # When the declared output_type is str we skip literal evaluation so that a
140
+ # rendered string that happens to be a valid Python literal (e.g. "1,000" -> (1, 0),
141
+ # "42" -> 42, "None" -> None) is returned unchanged instead of being coerced to
142
+ # another type, which would violate the declared output_type.
139
143
  with contextlib.suppress(Exception):
140
- if not self._unsafe:
144
+ if not self._unsafe and self.output_type is not str:
141
145
  output_result = ast.literal_eval(output_result)
142
146
 
143
147
  adapted_outputs["output"] = output_result
@@ -18,7 +18,8 @@ class TextFileToDocument:
18
18
  """
19
19
  Converts text files to documents your pipeline can query.
20
20
 
21
- By default, it uses UTF-8 encoding when converting files but
21
+ By default, it uses UTF-8 encoding (`utf-8-sig`, which also strips a byte order mark if
22
+ present) when converting files but
22
23
  you can also set custom encoding.
23
24
  It can attach metadata to the resulting documents.
24
25
 
@@ -36,7 +37,7 @@ class TextFileToDocument:
36
37
  ```
37
38
  """
38
39
 
39
- def __init__(self, encoding: str = "utf-8", store_full_path: bool = False) -> None:
40
+ def __init__(self, encoding: str = "utf-8-sig", store_full_path: bool = False) -> None:
40
41
  """
41
42
  Creates a TextFileToDocument component.
42
43
 
@@ -209,6 +209,10 @@ class XLSXToDocument:
209
209
  if row_idx < len(df) and col_idx < len(df.columns):
210
210
  cell_value = df.iat[row_idx, col_idx]
211
211
  text = str(cell_value) if pd.notna(cell_value) else ""
212
+ # Hyperlink text must be assignable to numeric and other typed columns.
213
+ column = df.columns[col_idx]
214
+ if df[column].dtype != object:
215
+ df[column] = df[column].astype(object)
212
216
  if self.link_format == "markdown":
213
217
  df.iat[row_idx, col_idx] = f"[{text}]({url})"
214
218
  else:
@@ -227,10 +231,15 @@ class XLSXToDocument:
227
231
  "index": True,
228
232
  "headers": value.columns,
229
233
  "tablefmt": "pipe",
234
+ "missingval": "",
230
235
  **self.table_format_kwargs,
231
236
  }
232
- # to_markdown uses tabulate
233
- tables.append(value.to_markdown(**resolved_kwargs))
237
+ # to_markdown uses tabulate, whose missingval only covers None: a NaN
238
+ # reaches the formatter as a number and is written out as "nan". Replace
239
+ # the empty cells with None so an empty cell reads as empty, the way
240
+ # to_csv already writes it, and so missingval keeps working.
241
+ filled = value.astype(object).where(value.notna(), None)
242
+ tables.append(filled.to_markdown(**resolved_kwargs))
234
243
  # add sheet_name to metadata
235
244
  metadata.append({"xlsx": {"sheet_name": key}})
236
245
  return tables, metadata
@@ -297,7 +297,7 @@ class OpenAIDocumentEmbedder:
297
297
  batches = async_tqdm(batches, desc="Calculating embeddings")
298
298
 
299
299
  for batch in batches:
300
- args: dict[str, Any] = {"model": self.model, "input": [b[1] for b in batch]}
300
+ args: dict[str, Any] = {"model": self.model, "input": [b[1] for b in batch], "encoding_format": "float"}
301
301
 
302
302
  if self.dimensions is not None:
303
303
  args["dimensions"] = self.dimensions
@@ -6,7 +6,6 @@ import asyncio
6
6
  import json
7
7
  from concurrent.futures import ThreadPoolExecutor
8
8
  from dataclasses import replace
9
- from functools import partial
10
9
  from typing import Any, Literal
11
10
 
12
11
  from jinja2 import meta
@@ -273,19 +272,25 @@ class LLMDocumentContentExtractor:
273
272
  meta_updates = {k: v for k, v in parsed.items() if k != DOCUMENT_CONTENT_KEY}
274
273
  return content, meta_updates, None
275
274
 
275
+ @staticmethod
276
+ def _fail(document: Document, error: str) -> tuple[Document, bool]:
277
+ """Return a copy of ``document`` with ``extraction_error`` set, flagged as failed."""
278
+ return replace(document, meta={**document.meta, "extraction_error": error}), False
279
+
276
280
  def _run_on_thread(
277
- self, image_content: ImageContent | None, parent_span: tracing.Span | None = None
278
- ) -> dict[str, Any]:
281
+ self, document: Document, image_content: ImageContent | None, parent_span: tracing.Span | None = None
282
+ ) -> tuple[Document, bool]:
279
283
  """
280
284
  Execute the LLM inference in a separate thread for each document.
281
285
 
282
- :param image_content: The image content for one document, or None if conversion failed.
286
+ :param document: The document to extract content for.
287
+ :param image_content: The image content for the document, or None if conversion failed.
283
288
  :param parent_span: Span to nest the generator span under, captured on the calling thread.
284
289
  :returns:
285
- The LLM response if successful, or a dictionary with an "error" key on failure.
290
+ The updated document and True on success, or the document with failure metadata and False.
286
291
  """
287
292
  if image_content is None:
288
- return {"error": "Document has no content, skipping LLM call."}
293
+ return self._fail(document, "Document has no content, skipping LLM call.")
289
294
 
290
295
  # the prompt is the same for all documents, so we can set it up once here for each document/thread
291
296
  message = ChatMessage.from_user(content_parts=[TextContent(text=self.prompt), image_content])
@@ -304,23 +309,24 @@ class LLMDocumentContentExtractor:
304
309
  class_name=self._chat_generator.__class__.__name__,
305
310
  error=e,
306
311
  )
307
- result = {"error": "LLM failed with exception: " + str(e)}
312
+ return self._fail(document, "LLM failed with exception: " + str(e))
308
313
 
309
- return result
314
+ return self._process_llm_results(document, result["replies"][0])
310
315
 
311
316
  async def _run_async(
312
- self, image_content: ImageContent | None, parent_span: tracing.Span | None = None
313
- ) -> dict[str, Any]:
317
+ self, document: Document, image_content: ImageContent | None, parent_span: tracing.Span | None = None
318
+ ) -> tuple[Document, bool]:
314
319
  """
315
320
  Execute the LLM inference asynchronously for each document.
316
321
 
317
- :param image_content: The image content for one document, or None if conversion failed.
322
+ :param document: The document to extract content for.
323
+ :param image_content: The image content for the document, or None if conversion failed.
318
324
  :param parent_span: Span to nest the generator span under, captured on the calling task.
319
325
  :returns:
320
- The LLM response if successful, or a dictionary with an "error" key on failure.
326
+ The updated document and True on success, or the document with failure metadata and False.
321
327
  """
322
328
  if image_content is None:
323
- return {"error": "Document has no content, skipping LLM call."}
329
+ return self._fail(document, "Document has no content, skipping LLM call.")
324
330
 
325
331
  # the prompt is the same for all documents, so we can set it up once here for each document
326
332
  message = ChatMessage.from_user(content_parts=[TextContent(text=self.prompt), image_content])
@@ -339,28 +345,23 @@ class LLMDocumentContentExtractor:
339
345
  class_name=self._chat_generator.__class__.__name__,
340
346
  error=e,
341
347
  )
342
- result = {"error": "LLM failed with exception: " + str(e)}
348
+ return self._fail(document, "LLM failed with exception: " + str(e))
343
349
 
344
- return result
350
+ return self._process_llm_results(document, result["replies"][0])
345
351
 
346
352
  @staticmethod
347
- def _process_llm_results(document: Document, result: dict[str, Any]) -> tuple[Document, bool]:
353
+ def _process_llm_results(document: Document, reply: ChatMessage) -> tuple[Document, bool]:
348
354
  """
349
- Process one document's LLM result using the unified response logic.
355
+ Process one document's LLM reply using the unified response logic.
350
356
 
351
357
  Returns (updated_document, True if success else False).
352
358
  """
353
- if "error" in result:
354
- new_meta = {**document.meta, "extraction_error": result["error"]}
355
- return replace(document, meta=new_meta), False
356
-
357
359
  # remove potentially existing error metadata from previous runs
358
360
  new_meta = {**document.meta}
359
361
  new_meta.pop("extraction_error", None)
360
362
 
361
363
  # process the LLM response considering the possible response formats
362
- response_text = result["replies"][0].text
363
- content, meta_updates, error = LLMDocumentContentExtractor._process_response(response_text)
364
+ content, meta_updates, error = LLMDocumentContentExtractor._process_response(reply.text or "")
364
365
 
365
366
  if error:
366
367
  new_meta["extraction_error"] = error
@@ -390,12 +391,15 @@ class LLMDocumentContentExtractor:
390
391
  parent_span = tracing.tracer.current_span()
391
392
 
392
393
  with ThreadPoolExecutor(max_workers=self.max_workers) as executor:
393
- results = executor.map(partial(self._run_on_thread, parent_span=parent_span), image_contents)
394
+ results = executor.map(
395
+ lambda document, image_content: self._run_on_thread(document, image_content, parent_span=parent_span),
396
+ documents,
397
+ image_contents,
398
+ )
394
399
 
395
400
  successful_documents = []
396
401
  failed_documents = []
397
- for document, result in zip(documents, results, strict=True):
398
- doc, success = self._process_llm_results(document, result)
402
+ for doc, success in results:
399
403
  if success:
400
404
  successful_documents.append(doc)
401
405
  else:
@@ -422,7 +426,10 @@ class LLMDocumentContentExtractor:
422
426
 
423
427
  await self.warm_up_async()
424
428
 
425
- image_contents = self._document_to_image_content.run(documents=documents)["image_contents"]
429
+ # Reading the files and rendering PDF pages is blocking work and `DocumentToImageContent` has no
430
+ # `run_async`, so it runs in a thread instead of on the event loop.
431
+ conversion_result = await _execute_component_async(self._document_to_image_content, documents=documents)
432
+ image_contents = conversion_result["image_contents"]
426
433
 
427
434
  # Capture the current span here so concurrent tasks nest their generator spans under the component span.
428
435
  parent_span = tracing.tracer.current_span()
@@ -430,16 +437,20 @@ class LLMDocumentContentExtractor:
430
437
  # Run the LLM on each image content, bounding concurrency per task so max_workers is enforced.
431
438
  sem = asyncio.Semaphore(max(1, self.max_workers))
432
439
 
433
- async def _bounded_run(image_content: ImageContent | None) -> dict[str, Any]:
440
+ async def _bounded_run(document: Document, image_content: ImageContent | None) -> tuple[Document, bool]:
434
441
  async with sem:
435
- return await self._run_async(image_content, parent_span=parent_span)
442
+ return await self._run_async(document, image_content, parent_span=parent_span)
436
443
 
437
- results = await asyncio.gather(*[_bounded_run(image_content) for image_content in image_contents])
444
+ results = await asyncio.gather(
445
+ *[
446
+ _bounded_run(document, image_content)
447
+ for document, image_content in zip(documents, image_contents, strict=True)
448
+ ]
449
+ )
438
450
 
439
451
  successful_documents = []
440
452
  failed_documents = []
441
- for document, result in zip(documents, results, strict=True):
442
- doc, success = self._process_llm_results(document, result)
453
+ for doc, success in results:
443
454
  if success:
444
455
  successful_documents.append(doc)
445
456
  else:
@@ -268,20 +268,6 @@ class LLMMetadataExtractor:
268
268
  deserialize_chatgenerator_inplace(data["init_parameters"], key="chat_generator")
269
269
  return default_from_dict(cls, data)
270
270
 
271
- def _extract_metadata(self, llm_answer: str) -> dict[str, Any]:
272
- try:
273
- parsed_metadata = _parse_dict_from_json(llm_answer, expected_keys=self.expected_keys, raise_on_failure=True)
274
- except (ValueError, json.JSONDecodeError) as e:
275
- logger.warning(
276
- "Response from the LLM is not valid JSON or missing expected keys. Received output: {response}",
277
- response=llm_answer,
278
- )
279
- if self.raise_on_failure:
280
- raise e
281
- return {"error": "Response is not valid JSON or missing keys. Error: " + str(e)}
282
-
283
- return parsed_metadata
284
-
285
271
  def _prepare_prompts(
286
272
  self, documents: list[Document], expanded_range: list[int] | None = None
287
273
  ) -> list[ChatMessage | None]:
@@ -368,18 +354,27 @@ class LLMMetadataExtractor:
368
354
  failed_documents.append(replace(document, meta=new_meta))
369
355
  continue
370
356
 
371
- parsed_metadata = self._extract_metadata(result["replies"][0].text)
372
- if "error" in parsed_metadata:
373
- new_meta["metadata_extraction_error"] = parsed_metadata["error"]
374
- new_meta["metadata_extraction_response"] = result["replies"][0]
357
+ reply = result["replies"][0]
358
+ try:
359
+ parsed_metadata = _parse_dict_from_json(
360
+ reply.text, expected_keys=self.expected_keys, raise_on_failure=True
361
+ )
362
+ except (ValueError, json.JSONDecodeError) as e:
363
+ logger.warning(
364
+ "Response from the LLM is not valid JSON or missing expected keys. Received output: {response}",
365
+ response=reply.text,
366
+ )
367
+ if self.raise_on_failure:
368
+ raise
369
+ new_meta["metadata_extraction_error"] = "Response is not valid JSON or missing keys. Error: " + str(e)
370
+ new_meta["metadata_extraction_response"] = reply
375
371
  failed_documents.append(replace(document, meta=new_meta))
376
372
  continue
377
373
 
378
- for key in parsed_metadata:
379
- new_meta[key] = parsed_metadata[key]
380
- # Remove metadata_extraction_error and metadata_extraction_response if present from previous runs
381
- new_meta.pop("metadata_extraction_error", None)
382
- new_meta.pop("metadata_extraction_response", None)
374
+ new_meta.update(parsed_metadata)
375
+ # Remove metadata_extraction_error and metadata_extraction_response if present from previous runs.
376
+ new_meta.pop("metadata_extraction_error", None)
377
+ new_meta.pop("metadata_extraction_response", None)
383
378
  successful_documents.append(replace(document, meta=new_meta))
384
379
  return successful_documents, failed_documents
385
380