haystack-ai 3.1.1__tar.gz → 3.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/PKG-INFO +1 -1
- haystack_ai-3.2.0/VERSION.txt +1 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/agents/agent.py +66 -26
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/builders/chat_prompt_builder.py +1 -1
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/converters/csv.py +3 -2
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/converters/image/document_to_image.py +17 -10
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/converters/json.py +1 -1
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/converters/markdown.py +1 -1
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/converters/multi_file_converter.py +2 -2
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/converters/output_adapter.py +5 -1
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/converters/txt.py +3 -2
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/converters/xlsx.py +11 -2
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/embedders/openai_document_embedder.py +1 -1
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/extractors/image/llm_document_content_extractor.py +43 -32
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/extractors/llm_metadata_extractor.py +18 -23
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/fetchers/link_content.py +10 -5
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/generators/chat/azure.py +1 -1
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/generators/chat/azure_responses.py +7 -4
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/generators/chat/fallback.py +19 -13
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/generators/chat/openai.py +2 -3
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/generators/chat/openai_responses.py +44 -14
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/generators/openai_image_generator.py +2 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/generators/utils.py +5 -2
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/joiners/answer_joiner.py +17 -5
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/joiners/document_joiner.py +9 -4
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/preprocessors/document_cleaner.py +18 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/preprocessors/document_preprocessor.py +9 -2
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/preprocessors/document_splitter.py +136 -11
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/preprocessors/embedding_based_document_splitter.py +10 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/preprocessors/markdown_header_splitter.py +94 -59
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/preprocessors/python_code_splitter.py +3 -2
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/preprocessors/recursive_splitter.py +36 -18
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/preprocessors/text_cleaner.py +16 -1
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/query/query_expander.py +8 -8
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/rankers/llm_ranker.py +2 -2
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/rankers/meta_field.py +2 -3
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/rankers/meta_field_grouping_ranker.py +1 -1
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/retrievers/auto_merging_retriever.py +4 -2
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/retrievers/filter_retriever.py +6 -2
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/retrievers/multi_query_embedding_retriever.py +10 -2
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/retrievers/multi_query_text_retriever.py +10 -2
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/retrievers/multi_retriever.py +10 -6
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/routers/conditional_router.py +5 -1
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/routers/document_type_router.py +9 -3
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/routers/file_type_router.py +18 -1
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/routers/llm_messages_router.py +14 -2
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/routers/metadata_router.py +6 -1
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/samplers/top_p.py +6 -2
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/validators/json_schema.py +5 -5
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/core/pipeline/base.py +108 -11
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/core/pipeline/breakpoint.py +12 -2
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/core/pipeline/draw.py +1 -1
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/core/super_component/super_component.py +4 -1
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/core/type_utils.py +19 -11
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/document_stores/in_memory/document_store.py +33 -5
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/document_stores/types/filter_policy.py +10 -6
- haystack_ai-3.2.0/haystack/hooks/budget/__init__.py +15 -0
- haystack_ai-3.2.0/haystack/hooks/budget/hooks.py +100 -0
- haystack_ai-3.2.0/haystack/hooks/compaction/AGENTS.md +10 -0
- haystack_ai-3.2.0/haystack/hooks/compaction/CLAUDE.md +1 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/hooks/compaction/__init__.py +2 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/hooks/compaction/hooks.py +35 -7
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/hooks/compaction/sliding_window.py +1 -0
- haystack_ai-3.2.0/haystack/hooks/compaction/summarization.py +511 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/hooks/compaction/tool_result_pruning.py +1 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/hooks/compaction/utils.py +0 -22
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/hooks/tool_result_offloading/hooks.py +134 -85
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/hooks/tool_result_offloading/stores.py +25 -8
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/hooks/tool_result_offloading/types/protocol.py +28 -9
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/logging.py +27 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/skill_stores/file_system/skill_store.py +28 -10
- haystack_ai-3.2.0/haystack/testing/telemetry.py +77 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/token_counters/utils.py +33 -8
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/tools/agent_tool.py +9 -3
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/tools/from_function.py +12 -7
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/tools/searchable_toolset.py +1 -5
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/tools/skills/skill_toolset.py +1 -8
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/tools/toolset.py +8 -156
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/utils/auth.py +3 -3
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/utils/azure.py +1 -1
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/utils/device.py +13 -4
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/utils/filters.py +3 -1
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/pyproject.toml +1 -1
- haystack_ai-3.1.1/VERSION.txt +0 -1
- haystack_ai-3.1.1/haystack/components/generators/chat/utils.py +0 -41
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/.gitignore +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/LICENSE +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/README.md +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/agents/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/agents/state/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/agents/state/state.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/agents/state/state_utils.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/agents/tool_calling.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/agents/utils.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/builders/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/builders/answer_builder.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/builders/prompt_builder.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/caching/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/caching/cache_checker.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/converters/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/converters/docx.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/converters/file_to_file_content.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/converters/html.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/converters/image/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/converters/image/file_to_document.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/converters/image/file_to_image.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/converters/image/image_utils.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/converters/image/pdf_to_image.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/converters/msg.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/converters/pdfminer.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/converters/pptx.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/converters/pypdf.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/converters/utils.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/embedders/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/embedders/azure_document_embedder.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/embedders/azure_text_embedder.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/embedders/mock_document_embedder.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/embedders/mock_text_embedder.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/embedders/mock_utils.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/embedders/openai_text_embedder.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/embedders/types/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/embedders/types/protocol.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/evaluators/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/evaluators/answer_exact_match.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/evaluators/context_relevance.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/evaluators/document_map.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/evaluators/document_mrr.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/evaluators/document_ndcg.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/evaluators/document_recall.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/evaluators/faithfulness.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/evaluators/llm_evaluator.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/evaluators/sas_evaluator.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/extractors/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/extractors/image/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/extractors/regex_text_extractor.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/fetchers/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/generators/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/generators/chat/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/generators/chat/llm.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/generators/chat/mock.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/generators/chat/types/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/generators/chat/types/protocol.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/joiners/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/joiners/branch.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/joiners/list_joiner.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/joiners/string_joiner.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/preprocessors/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/preprocessors/csv_document_cleaner.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/preprocessors/csv_document_splitter.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/preprocessors/hierarchical_document_splitter.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/preprocessors/sentence_tokenizer.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/query/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/rankers/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/rankers/lost_in_the_middle.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/retrievers/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/retrievers/in_memory/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/retrievers/in_memory/bm25_retriever.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/retrievers/in_memory/embedding_retriever.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/retrievers/sentence_window_retriever.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/retrievers/text_embedding_retriever.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/retrievers/types/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/retrievers/types/protocol.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/routers/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/routers/document_length_router.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/samplers/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/validators/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/writers/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/writers/document_writer.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/core/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/core/component/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/core/component/component.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/core/component/sockets.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/core/component/types.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/core/errors.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/core/pipeline/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/core/pipeline/component_checks.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/core/pipeline/descriptions.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/core/pipeline/pipeline.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/core/pipeline/utils.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/core/serialization.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/core/serialization_security.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/core/super_component/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/core/super_component/utils.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/data/abbreviations/de.txt +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/data/abbreviations/en.txt +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/dataclasses/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/dataclasses/answer.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/dataclasses/breakpoints.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/dataclasses/byte_stream.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/dataclasses/chat_message.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/dataclasses/document.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/dataclasses/file_content.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/dataclasses/image_content.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/dataclasses/skill_info.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/dataclasses/sparse_embedding.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/dataclasses/streaming_chunk.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/document_stores/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/document_stores/errors/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/document_stores/errors/errors.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/document_stores/in_memory/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/document_stores/types/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/document_stores/types/policy.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/document_stores/types/protocol.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/errors.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/evaluation/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/evaluation/eval_run_result.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/hooks/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/hooks/compaction/types/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/hooks/compaction/types/protocol.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/hooks/from_function.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/hooks/human_in_the_loop/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/hooks/human_in_the_loop/dataclasses.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/hooks/human_in_the_loop/hooks.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/hooks/human_in_the_loop/policies.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/hooks/human_in_the_loop/strategies.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/hooks/human_in_the_loop/types/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/hooks/human_in_the_loop/types/protocol.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/hooks/human_in_the_loop/user_interfaces.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/hooks/invocation.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/hooks/protocol.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/hooks/tool_result_offloading/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/hooks/tool_result_offloading/policies.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/hooks/tool_result_offloading/types/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/hooks/utils.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/lazy_imports.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/marshal/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/marshal/protocol.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/marshal/yaml.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/py.typed +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/skill_stores/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/skill_stores/file_system/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/skill_stores/types/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/skill_stores/types/protocol.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/telemetry/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/telemetry/_environment.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/telemetry/_telemetry.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/testing/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/testing/callable_serialization/random_callable.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/testing/document_store.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/testing/document_store_async.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/testing/factory.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/testing/sample_components/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/testing/sample_components/accumulate.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/testing/sample_components/add_value.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/testing/sample_components/concatenate.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/testing/sample_components/double.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/testing/sample_components/fstring.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/testing/sample_components/future_annotations.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/testing/sample_components/greet.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/testing/sample_components/hello.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/testing/sample_components/joiner.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/testing/sample_components/parity.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/testing/sample_components/remainder.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/testing/sample_components/repeat.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/testing/sample_components/subtract.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/testing/sample_components/sum.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/testing/sample_components/text_splitter.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/testing/sample_components/threshold.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/testing/test_utils.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/token_counters/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/token_counters/approximate_counter.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/token_counters/openai_counter.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/token_counters/tiktoken_counter.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/token_counters/types/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/token_counters/types/protocol.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/tools/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/tools/component_tool.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/tools/errors.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/tools/parameters_schema_utils.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/tools/pipeline_tool.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/tools/serde_utils.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/tools/skills/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/tools/tool.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/tools/tool_types.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/tools/utils.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/tracing/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/tracing/logging_tracer.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/tracing/tracer.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/tracing/utils.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/utils/__init__.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/utils/async_utils.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/utils/base_serialization.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/utils/callable_serialization.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/utils/dataclasses.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/utils/deserialization.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/utils/experimental.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/utils/hf.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/utils/http_client.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/utils/jinja2_chat_extension.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/utils/jinja2_extensions.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/utils/jinja2_sandbox.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/utils/jupyter.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/utils/misc.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/utils/requests_utils.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/utils/type_serialization.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/utils/url_validation.py +0 -0
- {haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/version.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: haystack-ai
|
|
3
|
-
Version: 3.
|
|
3
|
+
Version: 3.2.0
|
|
4
4
|
Summary: LLM framework to build customizable, production-ready LLM applications. Connect components (models, vector DBs, file converters) to pipelines or agents that can interact with your data.
|
|
5
5
|
Project-URL: CI: GitHub, https://github.com/deepset-ai/haystack/actions
|
|
6
6
|
Project-URL: Docs: RTD, https://haystack.deepset.ai/overview/intro
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
3.2.0
|
|
@@ -65,9 +65,11 @@ from haystack.utils.deserialization import deserialize_component_inplace
|
|
|
65
65
|
|
|
66
66
|
logger = logging.getLogger(__name__)
|
|
67
67
|
|
|
68
|
-
# `exit_reason` values the Agent sets when it stops without a tool exit condition: a tool-call-free reply,
|
|
69
|
-
# `max_agent_steps` budget running out.
|
|
68
|
+
# `exit_reason` values the Agent sets when it stops without a tool exit condition: a tool-call-free reply, an
|
|
69
|
+
# incomplete model generation, or the `max_agent_steps` budget running out.
|
|
70
70
|
_EXIT_REASON_TEXT = "text"
|
|
71
|
+
_EXIT_REASON_LENGTH = "length"
|
|
72
|
+
_EXIT_REASON_CONTENT_FILTER = "content_filter"
|
|
71
73
|
_EXIT_REASON_MAX_STEPS = "max_agent_steps"
|
|
72
74
|
|
|
73
75
|
# Run-metadata state keys the Agent populates automatically during a run. Users may not define them in their own
|
|
@@ -82,6 +84,7 @@ _RUN_METADATA_STATE_KEYS: dict[str, dict[str, Any]] = {
|
|
|
82
84
|
# Internal state keys the Agent manages for run control and hooks. Like run-metadata keys they are reserved and cannot
|
|
83
85
|
# be redefined by users, but unlike them they are NOT exposed as Agent inputs or outputs (purely internal state):
|
|
84
86
|
# - `continue_run`: set by an `on_exit` hook to keep the Agent running instead of stopping (re-read each exit attempt).
|
|
87
|
+
# - `stop_run`: set by a hook to stop the run, read before each LLM call and used as the `exit_reason`.
|
|
85
88
|
# - `tools`: the flattened tools available in the current step, so a hook can inspect them (e.g. HITL confirmation).
|
|
86
89
|
# - `hook_context`: per-run request-scoped resources passed to `run`/`run_async` for hooks to read.
|
|
87
90
|
# - `context_tokens`: approximate current context-window size, refreshed after each LLM call, for hooks to read
|
|
@@ -89,6 +92,7 @@ _RUN_METADATA_STATE_KEYS: dict[str, dict[str, Any]] = {
|
|
|
89
92
|
# exposed as an output because it is a best-effort snapshot; see `_record_context_tokens`.
|
|
90
93
|
_INTERNAL_STATE_KEYS: dict[str, dict[str, Any]] = {
|
|
91
94
|
"continue_run": {"type": bool, "handler": replace_values},
|
|
95
|
+
"stop_run": {"type": str, "handler": replace_values},
|
|
92
96
|
"tools": {"type": list, "handler": replace_values},
|
|
93
97
|
"hook_context": {"type": dict[str, Any], "handler": replace_values},
|
|
94
98
|
"context_tokens": {"type": int, "handler": replace_values},
|
|
@@ -147,17 +151,36 @@ def _consume_continue_run(state: State) -> bool:
|
|
|
147
151
|
return should_continue
|
|
148
152
|
|
|
149
153
|
|
|
150
|
-
def
|
|
154
|
+
def _get_model_exit_reason(messages: list[ChatMessage]) -> str | None:
|
|
151
155
|
"""
|
|
152
|
-
Return
|
|
156
|
+
Return the exit reason for a terminal assistant reply without tool calls.
|
|
153
157
|
|
|
154
|
-
|
|
155
|
-
|
|
158
|
+
Incomplete generation reasons take precedence over text so callers can distinguish a partial response from a
|
|
159
|
+
complete answer. An empty response without a recognized terminal reason does not trigger an exit, preserving the
|
|
160
|
+
Agent's recovery behavior for malformed tool calls that a Chat Generator discarded.
|
|
156
161
|
"""
|
|
157
|
-
|
|
158
|
-
|
|
162
|
+
# If the messages list is empty or the last message has tool calls, don't exit.
|
|
163
|
+
if not messages or any(message.tool_call for message in messages):
|
|
164
|
+
return None
|
|
165
|
+
|
|
159
166
|
last = messages[-1]
|
|
160
|
-
|
|
167
|
+
|
|
168
|
+
# If the last message is not from the assistant, don't exit.
|
|
169
|
+
if not last.is_from(ChatRole.ASSISTANT):
|
|
170
|
+
return None
|
|
171
|
+
|
|
172
|
+
# If the finish reason on the last message is length or content_filter, exit with that reason.
|
|
173
|
+
if last.meta.get("finish_reason") == _EXIT_REASON_LENGTH:
|
|
174
|
+
return _EXIT_REASON_LENGTH
|
|
175
|
+
if last.meta.get("finish_reason") == _EXIT_REASON_CONTENT_FILTER:
|
|
176
|
+
return _EXIT_REASON_CONTENT_FILTER
|
|
177
|
+
|
|
178
|
+
# If the last message has text, exit with the text reason.
|
|
179
|
+
if last.text:
|
|
180
|
+
return _EXIT_REASON_TEXT
|
|
181
|
+
|
|
182
|
+
# If we reached here no valid exit reason was found, so don't exit.
|
|
183
|
+
return None
|
|
161
184
|
|
|
162
185
|
|
|
163
186
|
def _pending_tool_call_messages_from_state(state: State) -> list[ChatMessage]:
|
|
@@ -435,9 +458,12 @@ class Agent:
|
|
|
435
458
|
"""
|
|
436
459
|
# --- Validation ---
|
|
437
460
|
self._chat_generator_supports_tools: bool = "tools" in inspect.signature(chat_generator.run).parameters
|
|
438
|
-
#
|
|
439
|
-
#
|
|
440
|
-
|
|
461
|
+
# An empty list carries no tools, so it must not trip this check: `tools` is normalized to `[]` below, and
|
|
462
|
+
# both `clone()` and `to_dict()` feed that normalized value straight back into `__init__`. This mirrors the
|
|
463
|
+
# equivalent check in `run()`. Only a list is measured; a Toolset is never tested for truthiness here b/c
|
|
464
|
+
# that calls __len__, which for SearchableToolset would iterate and prematurely warm it up at init.
|
|
465
|
+
tools_provided = tools is not None and (not isinstance(tools, list) or len(tools) > 0)
|
|
466
|
+
if tools_provided and not self._chat_generator_supports_tools:
|
|
441
467
|
raise TypeError(
|
|
442
468
|
f"{type(chat_generator).__name__} does not accept tools parameter in its run method. "
|
|
443
469
|
"The Agent component requires a chat generator that supports tools when tools are provided."
|
|
@@ -838,9 +864,11 @@ class Agent:
|
|
|
838
864
|
`meta["usage"]`.
|
|
839
865
|
- "tool_call_counts": Mapping of tool name to the number of times that tool was invoked.
|
|
840
866
|
- "exit_reason": Why the Agent stopped, useful for routing the output downstream (e.g. with a
|
|
841
|
-
`ConditionalRouter`). One of: `"text"` (the model returned a reply with no tool calls),
|
|
842
|
-
|
|
843
|
-
|
|
867
|
+
`ConditionalRouter`). One of: `"text"` (the model returned a complete reply with no tool calls),
|
|
868
|
+
`"length"` or `"content_filter"` (the model returned an incomplete reply, which may contain partial
|
|
869
|
+
text), the name of the tool that satisfied a tool exit condition (in which case `last_message` is that
|
|
870
|
+
tool's result), or `"max_agent_steps"` (the Agent hit `max_agent_steps` before meeting an exit
|
|
871
|
+
condition), or a custom reason a hook supplied through the `stop_run` state key.
|
|
844
872
|
- Any additional keys defined in the `state_schema`.
|
|
845
873
|
"""
|
|
846
874
|
agent_inputs = {"messages": messages, "streaming_callback": streaming_callback, **kwargs}
|
|
@@ -922,9 +950,11 @@ class Agent:
|
|
|
922
950
|
`meta["usage"]`.
|
|
923
951
|
- "tool_call_counts": Mapping of tool name to the number of times that tool was invoked.
|
|
924
952
|
- "exit_reason": Why the Agent stopped, useful for routing the output downstream (e.g. with a
|
|
925
|
-
`ConditionalRouter`). One of: `"text"` (the model returned a reply with no tool calls),
|
|
926
|
-
|
|
927
|
-
|
|
953
|
+
`ConditionalRouter`). One of: `"text"` (the model returned a complete reply with no tool calls),
|
|
954
|
+
`"length"` or `"content_filter"` (the model returned an incomplete reply, which may contain partial
|
|
955
|
+
text), the name of the tool that satisfied a tool exit condition (in which case `last_message` is that
|
|
956
|
+
tool's result), or `"max_agent_steps"` (the Agent hit `max_agent_steps` before meeting an exit
|
|
957
|
+
condition), or a custom reason a hook supplied through the `stop_run` state key.
|
|
928
958
|
- Any additional keys defined in the `state_schema`.
|
|
929
959
|
"""
|
|
930
960
|
agent_inputs = {"messages": messages, "streaming_callback": streaming_callback, **kwargs}
|
|
@@ -978,6 +1008,10 @@ class Agent:
|
|
|
978
1008
|
exe_context.state.set("tools", current_tools, handler_override=replace_values)
|
|
979
1009
|
|
|
980
1010
|
_run_hooks(hooks=self.hooks, hook_point=BEFORE_LLM, state=exe_context.state)
|
|
1011
|
+
# A hook requested a stop: end the run at the step boundary, before spending another LLM call.
|
|
1012
|
+
if (reason := exe_context.state.data.get("stop_run")) is not None:
|
|
1013
|
+
exe_context.state.set("exit_reason", reason)
|
|
1014
|
+
return False
|
|
981
1015
|
chat_generator_inputs = {
|
|
982
1016
|
"messages": exe_context.state.data["messages"],
|
|
983
1017
|
**exe_context.chat_generator_inputs,
|
|
@@ -993,17 +1027,18 @@ class Agent:
|
|
|
993
1027
|
_record_llm_usage(state=exe_context.state, llm_messages=llm_messages)
|
|
994
1028
|
_record_context_tokens(state=exe_context.state, llm_messages=llm_messages)
|
|
995
1029
|
|
|
996
|
-
# Stop
|
|
997
|
-
|
|
1030
|
+
# Stop when there are no tools, or the model produced a terminal reply without tool calls.
|
|
1031
|
+
model_exit_reason = _get_model_exit_reason(messages=llm_messages)
|
|
1032
|
+
if not current_tools or model_exit_reason is not None:
|
|
998
1033
|
exe_context.counter += 1
|
|
999
1034
|
exe_context.state.set("step_count", exe_context.counter)
|
|
1000
|
-
exe_context.state.set("exit_reason", _EXIT_REASON_TEXT)
|
|
1035
|
+
exe_context.state.set("exit_reason", model_exit_reason or _EXIT_REASON_TEXT)
|
|
1001
1036
|
return self._continue_after_exit_hooks(exe_context=exe_context)
|
|
1002
1037
|
|
|
1003
1038
|
_run_hooks(hooks=self.hooks, hook_point=BEFORE_TOOL, state=exe_context.state)
|
|
1004
1039
|
# Re-read the pending tool calls from State so that any rewrites a before_tool hook made (e.g.
|
|
1005
1040
|
# ConfirmationHook rejecting or modifying calls) are honored by the executor.
|
|
1006
|
-
pending_tool_call_messages = _pending_tool_call_messages_from_state(exe_context.state)
|
|
1041
|
+
pending_tool_call_messages = _pending_tool_call_messages_from_state(state=exe_context.state)
|
|
1007
1042
|
|
|
1008
1043
|
tool_execution_inputs = {
|
|
1009
1044
|
"messages": pending_tool_call_messages,
|
|
@@ -1041,6 +1076,10 @@ class Agent:
|
|
|
1041
1076
|
exe_context.state.set("tools", current_tools, handler_override=replace_values)
|
|
1042
1077
|
|
|
1043
1078
|
await _run_hooks_async(hooks=self.hooks, hook_point=BEFORE_LLM, state=exe_context.state)
|
|
1079
|
+
# A hook requested a stop: end the run at the step boundary, before spending another LLM call.
|
|
1080
|
+
if (reason := exe_context.state.data.get("stop_run")) is not None:
|
|
1081
|
+
exe_context.state.set("exit_reason", reason)
|
|
1082
|
+
return False
|
|
1044
1083
|
chat_generator_inputs = {
|
|
1045
1084
|
"messages": exe_context.state.data["messages"],
|
|
1046
1085
|
**exe_context.chat_generator_inputs,
|
|
@@ -1058,17 +1097,18 @@ class Agent:
|
|
|
1058
1097
|
_record_llm_usage(state=exe_context.state, llm_messages=llm_messages)
|
|
1059
1098
|
_record_context_tokens(state=exe_context.state, llm_messages=llm_messages)
|
|
1060
1099
|
|
|
1061
|
-
# Stop
|
|
1062
|
-
|
|
1100
|
+
# Stop when there are no tools, or the model produced a terminal reply without tool calls.
|
|
1101
|
+
model_exit_reason = _get_model_exit_reason(messages=llm_messages)
|
|
1102
|
+
if not current_tools or model_exit_reason is not None:
|
|
1063
1103
|
exe_context.counter += 1
|
|
1064
1104
|
exe_context.state.set("step_count", exe_context.counter)
|
|
1065
|
-
exe_context.state.set("exit_reason", _EXIT_REASON_TEXT)
|
|
1105
|
+
exe_context.state.set("exit_reason", model_exit_reason or _EXIT_REASON_TEXT)
|
|
1066
1106
|
return await self._continue_after_exit_hooks_async(exe_context=exe_context)
|
|
1067
1107
|
|
|
1068
1108
|
await _run_hooks_async(hooks=self.hooks, hook_point=BEFORE_TOOL, state=exe_context.state)
|
|
1069
1109
|
# Re-read the pending tool calls from State so that any rewrites a before_tool hook made (e.g.
|
|
1070
1110
|
# ConfirmationHook rejecting or modifying calls) are honored by the executor.
|
|
1071
|
-
pending_tool_call_messages = _pending_tool_call_messages_from_state(exe_context.state)
|
|
1111
|
+
pending_tool_call_messages = _pending_tool_call_messages_from_state(state=exe_context.state)
|
|
1072
1112
|
|
|
1073
1113
|
tool_execution_inputs = {
|
|
1074
1114
|
"messages": pending_tool_call_messages,
|
|
@@ -234,7 +234,7 @@ class ChatPromptBuilder:
|
|
|
234
234
|
:returns: A dictionary with the following keys:
|
|
235
235
|
- `prompt`: The updated list of `ChatMessage` objects after rendering the templates.
|
|
236
236
|
:raises ValueError:
|
|
237
|
-
If `
|
|
237
|
+
If `template` is empty or contains elements that are not instances of `ChatMessage`.
|
|
238
238
|
"""
|
|
239
239
|
kwargs = kwargs or {}
|
|
240
240
|
template_variables = template_variables or {}
|
|
@@ -22,7 +22,8 @@ class CSVToDocument:
|
|
|
22
22
|
"""
|
|
23
23
|
Converts CSV files to Documents.
|
|
24
24
|
|
|
25
|
-
By default, it uses UTF-8 encoding
|
|
25
|
+
By default, it uses UTF-8 encoding (`utf-8-sig`, which also strips a byte order mark if
|
|
26
|
+
present) when converting files but
|
|
26
27
|
you can also set a custom encoding.
|
|
27
28
|
It can attach metadata to the resulting documents.
|
|
28
29
|
|
|
@@ -44,7 +45,7 @@ class CSVToDocument:
|
|
|
44
45
|
|
|
45
46
|
def __init__(
|
|
46
47
|
self,
|
|
47
|
-
encoding: str = "utf-8",
|
|
48
|
+
encoding: str = "utf-8-sig",
|
|
48
49
|
store_full_path: bool = False,
|
|
49
50
|
*,
|
|
50
51
|
conversion_mode: Literal["file", "row"] = "file",
|
{haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/converters/image/document_to_image.py
RENAMED
|
@@ -113,23 +113,30 @@ class DocumentToImageContent:
|
|
|
113
113
|
:returns:
|
|
114
114
|
Dictionary containing one key:
|
|
115
115
|
- "image_contents": ImageContents created from the processed documents. These contain base64-encoded image
|
|
116
|
-
data and metadata. The order corresponds to order of input documents.
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
MIME type. The error message will specify which document and what information is missing or incorrect.
|
|
116
|
+
data and metadata. The order corresponds to the order of the input documents. A document that is
|
|
117
|
+
missing the required metadata keys, has an invalid file path, or has an unsupported MIME type gets
|
|
118
|
+
None in its position and a logged warning with the reason.
|
|
120
119
|
"""
|
|
121
120
|
if not documents:
|
|
122
121
|
return {"image_contents": []}
|
|
123
122
|
|
|
124
|
-
images_source_info = _extract_image_sources_info(
|
|
125
|
-
documents=documents, file_path_meta_field=self.file_path_meta_field, root_path=self.root_path
|
|
126
|
-
)
|
|
127
|
-
|
|
128
123
|
image_contents: list[ImageContent | None] = [None] * len(documents)
|
|
129
124
|
|
|
130
125
|
pdf_page_infos: list[_PDFPageInfo] = []
|
|
131
126
|
|
|
132
|
-
for doc_idx,
|
|
127
|
+
for doc_idx, document in enumerate(documents):
|
|
128
|
+
# Validate each document on its own so one invalid document leaves None in its slot
|
|
129
|
+
# instead of failing the whole batch
|
|
130
|
+
try:
|
|
131
|
+
image_source_info = _extract_image_sources_info(
|
|
132
|
+
documents=[document], file_path_meta_field=self.file_path_meta_field, root_path=self.root_path
|
|
133
|
+
)[0]
|
|
134
|
+
except ValueError as error:
|
|
135
|
+
logger.warning(
|
|
136
|
+
"Skipping document with ID {document_id}: {error}", document_id=document.id, error=str(error)
|
|
137
|
+
)
|
|
138
|
+
continue
|
|
139
|
+
|
|
133
140
|
mime_type = image_source_info["mime_type"]
|
|
134
141
|
path = image_source_info["path"]
|
|
135
142
|
if mime_type == "application/pdf":
|
|
@@ -146,7 +153,7 @@ class DocumentToImageContent:
|
|
|
146
153
|
base64_image=base64_image,
|
|
147
154
|
mime_type=mime_type,
|
|
148
155
|
detail=self.detail,
|
|
149
|
-
meta={"file_path":
|
|
156
|
+
meta={"file_path": document.meta[self.file_path_meta_field]},
|
|
150
157
|
)
|
|
151
158
|
|
|
152
159
|
# efficiently convert PDF pages to images: each PDF is opened and processed only once
|
|
@@ -187,7 +187,7 @@ class JSONConverter:
|
|
|
187
187
|
to a different document.
|
|
188
188
|
"""
|
|
189
189
|
try:
|
|
190
|
-
file_content = source.data.decode("utf-8")
|
|
190
|
+
file_content = source.data.decode("utf-8-sig")
|
|
191
191
|
except UnicodeError as exc:
|
|
192
192
|
logger.warning(
|
|
193
193
|
"Failed to extract text from {source}. Skipping it. Error: {error}",
|
{haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/converters/multi_file_converter.py
RENAMED
|
@@ -52,14 +52,14 @@ class MultiFileConverter:
|
|
|
52
52
|
|
|
53
53
|
Usage example:
|
|
54
54
|
```
|
|
55
|
-
from haystack.
|
|
55
|
+
from haystack.components.converters import MultiFileConverter
|
|
56
56
|
|
|
57
57
|
converter = MultiFileConverter()
|
|
58
58
|
converter.run(sources=["test/test_files/txt/doc_1.txt", "test/test_files/pdf/sample_pdf_1.pdf"], meta={})
|
|
59
59
|
```
|
|
60
60
|
"""
|
|
61
61
|
|
|
62
|
-
def __init__(self, encoding: str = "utf-8", json_content_key: str = "content") -> None:
|
|
62
|
+
def __init__(self, encoding: str = "utf-8-sig", json_content_key: str = "content") -> None:
|
|
63
63
|
"""
|
|
64
64
|
Initialize the MultiFileConverter.
|
|
65
65
|
|
|
@@ -136,8 +136,12 @@ class OutputAdapter:
|
|
|
136
136
|
# we try to evaluate it and would fail.
|
|
137
137
|
# This must be done cause the output could be different literal structures.
|
|
138
138
|
# This doesn't support any user types.
|
|
139
|
+
# When the declared output_type is str we skip literal evaluation so that a
|
|
140
|
+
# rendered string that happens to be a valid Python literal (e.g. "1,000" -> (1, 0),
|
|
141
|
+
# "42" -> 42, "None" -> None) is returned unchanged instead of being coerced to
|
|
142
|
+
# another type, which would violate the declared output_type.
|
|
139
143
|
with contextlib.suppress(Exception):
|
|
140
|
-
if not self._unsafe:
|
|
144
|
+
if not self._unsafe and self.output_type is not str:
|
|
141
145
|
output_result = ast.literal_eval(output_result)
|
|
142
146
|
|
|
143
147
|
adapted_outputs["output"] = output_result
|
|
@@ -18,7 +18,8 @@ class TextFileToDocument:
|
|
|
18
18
|
"""
|
|
19
19
|
Converts text files to documents your pipeline can query.
|
|
20
20
|
|
|
21
|
-
By default, it uses UTF-8 encoding
|
|
21
|
+
By default, it uses UTF-8 encoding (`utf-8-sig`, which also strips a byte order mark if
|
|
22
|
+
present) when converting files but
|
|
22
23
|
you can also set custom encoding.
|
|
23
24
|
It can attach metadata to the resulting documents.
|
|
24
25
|
|
|
@@ -36,7 +37,7 @@ class TextFileToDocument:
|
|
|
36
37
|
```
|
|
37
38
|
"""
|
|
38
39
|
|
|
39
|
-
def __init__(self, encoding: str = "utf-8", store_full_path: bool = False) -> None:
|
|
40
|
+
def __init__(self, encoding: str = "utf-8-sig", store_full_path: bool = False) -> None:
|
|
40
41
|
"""
|
|
41
42
|
Creates a TextFileToDocument component.
|
|
42
43
|
|
|
@@ -209,6 +209,10 @@ class XLSXToDocument:
|
|
|
209
209
|
if row_idx < len(df) and col_idx < len(df.columns):
|
|
210
210
|
cell_value = df.iat[row_idx, col_idx]
|
|
211
211
|
text = str(cell_value) if pd.notna(cell_value) else ""
|
|
212
|
+
# Hyperlink text must be assignable to numeric and other typed columns.
|
|
213
|
+
column = df.columns[col_idx]
|
|
214
|
+
if df[column].dtype != object:
|
|
215
|
+
df[column] = df[column].astype(object)
|
|
212
216
|
if self.link_format == "markdown":
|
|
213
217
|
df.iat[row_idx, col_idx] = f"[{text}]({url})"
|
|
214
218
|
else:
|
|
@@ -227,10 +231,15 @@ class XLSXToDocument:
|
|
|
227
231
|
"index": True,
|
|
228
232
|
"headers": value.columns,
|
|
229
233
|
"tablefmt": "pipe",
|
|
234
|
+
"missingval": "",
|
|
230
235
|
**self.table_format_kwargs,
|
|
231
236
|
}
|
|
232
|
-
# to_markdown uses tabulate
|
|
233
|
-
|
|
237
|
+
# to_markdown uses tabulate, whose missingval only covers None: a NaN
|
|
238
|
+
# reaches the formatter as a number and is written out as "nan". Replace
|
|
239
|
+
# the empty cells with None so an empty cell reads as empty, the way
|
|
240
|
+
# to_csv already writes it, and so missingval keeps working.
|
|
241
|
+
filled = value.astype(object).where(value.notna(), None)
|
|
242
|
+
tables.append(filled.to_markdown(**resolved_kwargs))
|
|
234
243
|
# add sheet_name to metadata
|
|
235
244
|
metadata.append({"xlsx": {"sheet_name": key}})
|
|
236
245
|
return tables, metadata
|
{haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/embedders/openai_document_embedder.py
RENAMED
|
@@ -297,7 +297,7 @@ class OpenAIDocumentEmbedder:
|
|
|
297
297
|
batches = async_tqdm(batches, desc="Calculating embeddings")
|
|
298
298
|
|
|
299
299
|
for batch in batches:
|
|
300
|
-
args: dict[str, Any] = {"model": self.model, "input": [b[1] for b in batch]}
|
|
300
|
+
args: dict[str, Any] = {"model": self.model, "input": [b[1] for b in batch], "encoding_format": "float"}
|
|
301
301
|
|
|
302
302
|
if self.dimensions is not None:
|
|
303
303
|
args["dimensions"] = self.dimensions
|
|
@@ -6,7 +6,6 @@ import asyncio
|
|
|
6
6
|
import json
|
|
7
7
|
from concurrent.futures import ThreadPoolExecutor
|
|
8
8
|
from dataclasses import replace
|
|
9
|
-
from functools import partial
|
|
10
9
|
from typing import Any, Literal
|
|
11
10
|
|
|
12
11
|
from jinja2 import meta
|
|
@@ -273,19 +272,25 @@ class LLMDocumentContentExtractor:
|
|
|
273
272
|
meta_updates = {k: v for k, v in parsed.items() if k != DOCUMENT_CONTENT_KEY}
|
|
274
273
|
return content, meta_updates, None
|
|
275
274
|
|
|
275
|
+
@staticmethod
|
|
276
|
+
def _fail(document: Document, error: str) -> tuple[Document, bool]:
|
|
277
|
+
"""Return a copy of ``document`` with ``extraction_error`` set, flagged as failed."""
|
|
278
|
+
return replace(document, meta={**document.meta, "extraction_error": error}), False
|
|
279
|
+
|
|
276
280
|
def _run_on_thread(
|
|
277
|
-
self, image_content: ImageContent | None, parent_span: tracing.Span | None = None
|
|
278
|
-
) ->
|
|
281
|
+
self, document: Document, image_content: ImageContent | None, parent_span: tracing.Span | None = None
|
|
282
|
+
) -> tuple[Document, bool]:
|
|
279
283
|
"""
|
|
280
284
|
Execute the LLM inference in a separate thread for each document.
|
|
281
285
|
|
|
282
|
-
:param
|
|
286
|
+
:param document: The document to extract content for.
|
|
287
|
+
:param image_content: The image content for the document, or None if conversion failed.
|
|
283
288
|
:param parent_span: Span to nest the generator span under, captured on the calling thread.
|
|
284
289
|
:returns:
|
|
285
|
-
The
|
|
290
|
+
The updated document and True on success, or the document with failure metadata and False.
|
|
286
291
|
"""
|
|
287
292
|
if image_content is None:
|
|
288
|
-
return
|
|
293
|
+
return self._fail(document, "Document has no content, skipping LLM call.")
|
|
289
294
|
|
|
290
295
|
# the prompt is the same for all documents, so we can set it up once here for each document/thread
|
|
291
296
|
message = ChatMessage.from_user(content_parts=[TextContent(text=self.prompt), image_content])
|
|
@@ -304,23 +309,24 @@ class LLMDocumentContentExtractor:
|
|
|
304
309
|
class_name=self._chat_generator.__class__.__name__,
|
|
305
310
|
error=e,
|
|
306
311
|
)
|
|
307
|
-
|
|
312
|
+
return self._fail(document, "LLM failed with exception: " + str(e))
|
|
308
313
|
|
|
309
|
-
return result
|
|
314
|
+
return self._process_llm_results(document, result["replies"][0])
|
|
310
315
|
|
|
311
316
|
async def _run_async(
|
|
312
|
-
self, image_content: ImageContent | None, parent_span: tracing.Span | None = None
|
|
313
|
-
) ->
|
|
317
|
+
self, document: Document, image_content: ImageContent | None, parent_span: tracing.Span | None = None
|
|
318
|
+
) -> tuple[Document, bool]:
|
|
314
319
|
"""
|
|
315
320
|
Execute the LLM inference asynchronously for each document.
|
|
316
321
|
|
|
317
|
-
:param
|
|
322
|
+
:param document: The document to extract content for.
|
|
323
|
+
:param image_content: The image content for the document, or None if conversion failed.
|
|
318
324
|
:param parent_span: Span to nest the generator span under, captured on the calling task.
|
|
319
325
|
:returns:
|
|
320
|
-
The
|
|
326
|
+
The updated document and True on success, or the document with failure metadata and False.
|
|
321
327
|
"""
|
|
322
328
|
if image_content is None:
|
|
323
|
-
return
|
|
329
|
+
return self._fail(document, "Document has no content, skipping LLM call.")
|
|
324
330
|
|
|
325
331
|
# the prompt is the same for all documents, so we can set it up once here for each document
|
|
326
332
|
message = ChatMessage.from_user(content_parts=[TextContent(text=self.prompt), image_content])
|
|
@@ -339,28 +345,23 @@ class LLMDocumentContentExtractor:
|
|
|
339
345
|
class_name=self._chat_generator.__class__.__name__,
|
|
340
346
|
error=e,
|
|
341
347
|
)
|
|
342
|
-
|
|
348
|
+
return self._fail(document, "LLM failed with exception: " + str(e))
|
|
343
349
|
|
|
344
|
-
return result
|
|
350
|
+
return self._process_llm_results(document, result["replies"][0])
|
|
345
351
|
|
|
346
352
|
@staticmethod
|
|
347
|
-
def _process_llm_results(document: Document,
|
|
353
|
+
def _process_llm_results(document: Document, reply: ChatMessage) -> tuple[Document, bool]:
|
|
348
354
|
"""
|
|
349
|
-
Process one document's LLM
|
|
355
|
+
Process one document's LLM reply using the unified response logic.
|
|
350
356
|
|
|
351
357
|
Returns (updated_document, True if success else False).
|
|
352
358
|
"""
|
|
353
|
-
if "error" in result:
|
|
354
|
-
new_meta = {**document.meta, "extraction_error": result["error"]}
|
|
355
|
-
return replace(document, meta=new_meta), False
|
|
356
|
-
|
|
357
359
|
# remove potentially existing error metadata from previous runs
|
|
358
360
|
new_meta = {**document.meta}
|
|
359
361
|
new_meta.pop("extraction_error", None)
|
|
360
362
|
|
|
361
363
|
# process the LLM response considering the possible response formats
|
|
362
|
-
|
|
363
|
-
content, meta_updates, error = LLMDocumentContentExtractor._process_response(response_text)
|
|
364
|
+
content, meta_updates, error = LLMDocumentContentExtractor._process_response(reply.text or "")
|
|
364
365
|
|
|
365
366
|
if error:
|
|
366
367
|
new_meta["extraction_error"] = error
|
|
@@ -390,12 +391,15 @@ class LLMDocumentContentExtractor:
|
|
|
390
391
|
parent_span = tracing.tracer.current_span()
|
|
391
392
|
|
|
392
393
|
with ThreadPoolExecutor(max_workers=self.max_workers) as executor:
|
|
393
|
-
results = executor.map(
|
|
394
|
+
results = executor.map(
|
|
395
|
+
lambda document, image_content: self._run_on_thread(document, image_content, parent_span=parent_span),
|
|
396
|
+
documents,
|
|
397
|
+
image_contents,
|
|
398
|
+
)
|
|
394
399
|
|
|
395
400
|
successful_documents = []
|
|
396
401
|
failed_documents = []
|
|
397
|
-
for
|
|
398
|
-
doc, success = self._process_llm_results(document, result)
|
|
402
|
+
for doc, success in results:
|
|
399
403
|
if success:
|
|
400
404
|
successful_documents.append(doc)
|
|
401
405
|
else:
|
|
@@ -422,7 +426,10 @@ class LLMDocumentContentExtractor:
|
|
|
422
426
|
|
|
423
427
|
await self.warm_up_async()
|
|
424
428
|
|
|
425
|
-
|
|
429
|
+
# Reading the files and rendering PDF pages is blocking work and `DocumentToImageContent` has no
|
|
430
|
+
# `run_async`, so it runs in a thread instead of on the event loop.
|
|
431
|
+
conversion_result = await _execute_component_async(self._document_to_image_content, documents=documents)
|
|
432
|
+
image_contents = conversion_result["image_contents"]
|
|
426
433
|
|
|
427
434
|
# Capture the current span here so concurrent tasks nest their generator spans under the component span.
|
|
428
435
|
parent_span = tracing.tracer.current_span()
|
|
@@ -430,16 +437,20 @@ class LLMDocumentContentExtractor:
|
|
|
430
437
|
# Run the LLM on each image content, bounding concurrency per task so max_workers is enforced.
|
|
431
438
|
sem = asyncio.Semaphore(max(1, self.max_workers))
|
|
432
439
|
|
|
433
|
-
async def _bounded_run(image_content: ImageContent | None) ->
|
|
440
|
+
async def _bounded_run(document: Document, image_content: ImageContent | None) -> tuple[Document, bool]:
|
|
434
441
|
async with sem:
|
|
435
|
-
return await self._run_async(image_content, parent_span=parent_span)
|
|
442
|
+
return await self._run_async(document, image_content, parent_span=parent_span)
|
|
436
443
|
|
|
437
|
-
results = await asyncio.gather(
|
|
444
|
+
results = await asyncio.gather(
|
|
445
|
+
*[
|
|
446
|
+
_bounded_run(document, image_content)
|
|
447
|
+
for document, image_content in zip(documents, image_contents, strict=True)
|
|
448
|
+
]
|
|
449
|
+
)
|
|
438
450
|
|
|
439
451
|
successful_documents = []
|
|
440
452
|
failed_documents = []
|
|
441
|
-
for
|
|
442
|
-
doc, success = self._process_llm_results(document, result)
|
|
453
|
+
for doc, success in results:
|
|
443
454
|
if success:
|
|
444
455
|
successful_documents.append(doc)
|
|
445
456
|
else:
|
{haystack_ai-3.1.1 → haystack_ai-3.2.0}/haystack/components/extractors/llm_metadata_extractor.py
RENAMED
|
@@ -268,20 +268,6 @@ class LLMMetadataExtractor:
|
|
|
268
268
|
deserialize_chatgenerator_inplace(data["init_parameters"], key="chat_generator")
|
|
269
269
|
return default_from_dict(cls, data)
|
|
270
270
|
|
|
271
|
-
def _extract_metadata(self, llm_answer: str) -> dict[str, Any]:
|
|
272
|
-
try:
|
|
273
|
-
parsed_metadata = _parse_dict_from_json(llm_answer, expected_keys=self.expected_keys, raise_on_failure=True)
|
|
274
|
-
except (ValueError, json.JSONDecodeError) as e:
|
|
275
|
-
logger.warning(
|
|
276
|
-
"Response from the LLM is not valid JSON or missing expected keys. Received output: {response}",
|
|
277
|
-
response=llm_answer,
|
|
278
|
-
)
|
|
279
|
-
if self.raise_on_failure:
|
|
280
|
-
raise e
|
|
281
|
-
return {"error": "Response is not valid JSON or missing keys. Error: " + str(e)}
|
|
282
|
-
|
|
283
|
-
return parsed_metadata
|
|
284
|
-
|
|
285
271
|
def _prepare_prompts(
|
|
286
272
|
self, documents: list[Document], expanded_range: list[int] | None = None
|
|
287
273
|
) -> list[ChatMessage | None]:
|
|
@@ -368,18 +354,27 @@ class LLMMetadataExtractor:
|
|
|
368
354
|
failed_documents.append(replace(document, meta=new_meta))
|
|
369
355
|
continue
|
|
370
356
|
|
|
371
|
-
|
|
372
|
-
|
|
373
|
-
|
|
374
|
-
|
|
357
|
+
reply = result["replies"][0]
|
|
358
|
+
try:
|
|
359
|
+
parsed_metadata = _parse_dict_from_json(
|
|
360
|
+
reply.text, expected_keys=self.expected_keys, raise_on_failure=True
|
|
361
|
+
)
|
|
362
|
+
except (ValueError, json.JSONDecodeError) as e:
|
|
363
|
+
logger.warning(
|
|
364
|
+
"Response from the LLM is not valid JSON or missing expected keys. Received output: {response}",
|
|
365
|
+
response=reply.text,
|
|
366
|
+
)
|
|
367
|
+
if self.raise_on_failure:
|
|
368
|
+
raise
|
|
369
|
+
new_meta["metadata_extraction_error"] = "Response is not valid JSON or missing keys. Error: " + str(e)
|
|
370
|
+
new_meta["metadata_extraction_response"] = reply
|
|
375
371
|
failed_documents.append(replace(document, meta=new_meta))
|
|
376
372
|
continue
|
|
377
373
|
|
|
378
|
-
|
|
379
|
-
|
|
380
|
-
|
|
381
|
-
|
|
382
|
-
new_meta.pop("metadata_extraction_response", None)
|
|
374
|
+
new_meta.update(parsed_metadata)
|
|
375
|
+
# Remove metadata_extraction_error and metadata_extraction_response if present from previous runs.
|
|
376
|
+
new_meta.pop("metadata_extraction_error", None)
|
|
377
|
+
new_meta.pop("metadata_extraction_response", None)
|
|
383
378
|
successful_documents.append(replace(document, meta=new_meta))
|
|
384
379
|
return successful_documents, failed_documents
|
|
385
380
|
|