markitdown-pro 2.0.0__tar.gz → 2.1.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/.github/workflows/test.yml +2 -4
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/.gitignore +2 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/PKG-INFO +1 -1
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/conversion_pipeline.py +21 -2
- markitdown_pro-2.1.1/markitdown_pro/converters/gotenberg_converter.py +200 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/converters/gpt_vision_converter.py +3 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/handlers/office_handler.py +32 -1
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/handlers/page_pdf_converter.py +1 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/services/openai_services.py +10 -2
- markitdown_pro-2.1.1/tests/data/powerpoint-legacy/geometry-colors.ppt +0 -0
- markitdown_pro-2.1.1/tests/data/powerpoint-legacy/lorem-ipsum-only-text.ppt +0 -0
- markitdown_pro-2.1.1/tests/data/powerpoint-legacy/lorem-ipsum-scanned-text.ppt +0 -0
- markitdown_pro-2.1.1/tests/data/powerpoint-legacy/lorem-ipsum-text-with-image.ppt +0 -0
- markitdown_pro-2.1.1/tests/data/powerpoint-legacy/lorem-ipsum-text-with-scanned-text.ppt +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/test_expectations.yaml +56 -0
- markitdown_pro-2.1.1/tests/data/word-legacy/geometry-colors.doc +0 -0
- markitdown_pro-2.1.1/tests/data/word-legacy/geometry-colors.odt +0 -0
- markitdown_pro-2.1.1/tests/data/word-legacy/geometry-colors.rtf +2197 -0
- markitdown_pro-2.1.1/tests/data/word-legacy/lorem-ipsum-only-text.doc +0 -0
- markitdown_pro-2.1.1/tests/data/word-legacy/lorem-ipsum-only-text.odt +0 -0
- markitdown_pro-2.1.1/tests/data/word-legacy/lorem-ipsum-only-text.rtf +258 -0
- markitdown_pro-2.1.1/tests/data/word-legacy/lorem-ipsum-scanned-text.doc +0 -0
- markitdown_pro-2.1.1/tests/data/word-legacy/lorem-ipsum-scanned-text.odt +0 -0
- markitdown_pro-2.1.1/tests/data/word-legacy/lorem-ipsum-scanned-text.rtf +18373 -0
- markitdown_pro-2.1.1/tests/data/word-legacy/lorem-ipsum-text-with-image.doc +0 -0
- markitdown_pro-2.1.1/tests/data/word-legacy/lorem-ipsum-text-with-image.odt +0 -0
- markitdown_pro-2.1.1/tests/data/word-legacy/lorem-ipsum-text-with-image.rtf +2204 -0
- markitdown_pro-2.1.1/tests/data/word-legacy/lorem-ipsum-text-with-scanned-text.doc +0 -0
- markitdown_pro-2.1.1/tests/data/word-legacy/lorem-ipsum-text-with-scanned-text.odt +0 -0
- markitdown_pro-2.1.1/tests/data/word-legacy/lorem-ipsum-text-with-scanned-text.rtf +7208 -0
- markitdown_pro-2.1.1/tests/integration/test_legacy_office_formats.py +108 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/unit/test_conversion_pipeline.py +74 -2
- markitdown_pro-2.1.1/tests/unit/test_gotenberg_converter.py +69 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/unit/test_office_handler.py +62 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/uv.lock +0 -1
- markitdown_pro-2.0.0/markitdown_pro/converters/gotenberg_converter.py +0 -118
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/.github/workflows/lint.yaml +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/.github/workflows/publish.yaml +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/.vscode/settings.json +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/README.md +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/__init__.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/__init__.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/common/__init__.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/common/config.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/common/isolated_worker.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/common/logger.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/common/schemas.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/common/utils.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/converters/__init__.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/converters/base.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/converters/doc_intel_converter.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/converters/markitdown_converter.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/converters/pymupdf_converter.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/converters/speech_converter.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/converters/tabular_converter.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/converters/unstructured_converter.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/converters/youtube_converter.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/handlers/__init__.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/handlers/audio_handler.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/handlers/base_handler.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/handlers/email_handler.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/handlers/epub_handler.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/handlers/image_handler.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/handlers/ipynb_handler.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/handlers/markup_handler.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/handlers/pdf_handler.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/handlers/pst_handler.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/handlers/tabular_handler.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/handlers/text_handler.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/services/__init__.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/services/azure_doc_intelligence.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/services/azure_speech.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/pyproject.toml +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/__init__.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/common/__init__.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/common/test_config.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/conftest.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/conversion_pipeline/__init__.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/conversion_pipeline/test_conversion_pipeline.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/converters/test_azure_doc_intel_wrapper.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/converters/test_azure_speech_wrapper.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/converters/test_doc_intel_lifecycle.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/converters/test_gpt_vision_config.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/converters/test_gpt_vision_wrapper.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/converters/test_pymupdf_wrapper.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/converters/test_pymupdf_wrapper_async.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/converters/test_unstructured_io_wrapper.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/audio/the-power-of-small-steps_cn.wav +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/audio/the-power-of-small-steps_en.mp3 +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/audio/the-power-of-small-steps_en.wav +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/audio/the-power-of-small-steps_es.mp3 +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/audio/the-power-of-small-steps_es.wav +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/audio/the-power-of-small-steps_it.wav +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/audio/the-power-of-small-steps_jp.mp3 +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/audio/the-power-of-small-steps_jp.wav +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/eml/email-with-image.eml +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/eml/mime-different-plain-html.eml +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/epub/sample1.epub +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/images/flower.jpg +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/images/geometry_colors.bmp +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/images/handwriting.jpg +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/images/scanned_text.heic +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/images/scanned_text.png +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/images/text_with_image.jpg +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/markup/blog.html +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/markup/factbook.xml +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/markup/simple.json +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/markup/simple.yaml +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/msg/fake-email-multiple-attachments.msg +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/msg/fake-email.msg +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/pdf/geometry-colors.pdf +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/pdf/lorem-ipsum-only-images.pdf +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/pdf/lorem-ipsum-only-text.pdf +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/pdf/lorem-ipsum-scanned-text.pdf +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/pdf/lorem-ipsum-text-with-image.pdf +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/powerpoint/geometry-colors.pptx +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/powerpoint/lorem-ipsum-only-text.pptx +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/powerpoint/lorem-ipsum-scanned-text.pptx +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/powerpoint/lorem-ipsum-text-with-image.pptx +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/powerpoint/lorem-ipsum-text-with-scanned-text.pptx +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/tabular/2023-half-year-analyses-by-segment.xlsx +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/tabular/stanley-cups.csv +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/tabular/stanley-cups.tsv +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/tabular/stanley-cups.xlsx +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/text/book-war-and-peace-1p.txt +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/text/fake-email.txt +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/text/logger.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/word/geometry-colors.docx +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/word/lorem-ipsum-only-text.docx +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/word/lorem-ipsum-scanned-text.docx +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/word/lorem-ipsum-text-with-image.docx +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/word/lorem-ipsum-text-with-scanned-text.docx +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/fixtures.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/handlers/__init__.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/handlers/test_email_handler.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/handlers/test_epub_handler.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/handlers/test_image_handler.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/handlers/test_ipynb_handler.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/handlers/test_markitdown_handler.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/handlers/test_markitdown_validation.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/handlers/test_markup_handler.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/handlers/test_page_pdf_converter.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/handlers/test_pdf_handler.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/handlers/test_pst_handler.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/handlers/test_tabular_handler.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/handlers/test_text_handler.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/integration/__init__.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/integration/cheat_sheet.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/integration/runner.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/integration/test_audio_handler.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/integration/test_doc_intelligence_converter.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/integration/test_email_handler.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/integration/test_gotenberg_converter.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/integration/test_gpt_vision_converter.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/integration/test_markitdown_converter.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/integration/test_markup_handler.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/integration/test_pdf_handler.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/integration/test_pymupdf_converter.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/integration/test_tabular_handler.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/integration/test_text_handler.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/page_count/__init__.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/page_count/test_pdf_page_count.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/unit/__init__.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/unit/test_audio_handler.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/unit/test_base_converter.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/unit/test_doc_intel_converter.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/unit/test_gpt_vision_converter.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/unit/test_image_handler.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/unit/test_markitdown_converter.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/unit/test_page_pdf_converter.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/unit/test_pdf_handler.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/unit/test_pymupdf_converter.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/unit/test_speech_converter.py +0 -0
- {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/utils.py +0 -0
|
@@ -1,16 +1,14 @@
|
|
|
1
1
|
name: '🧪 Test package'
|
|
2
2
|
|
|
3
3
|
on:
|
|
4
|
-
|
|
4
|
+
pull_request:
|
|
5
5
|
branches:
|
|
6
6
|
- main
|
|
7
7
|
paths:
|
|
8
8
|
- 'markitdown_pro/**'
|
|
9
9
|
- 'tests/**'
|
|
10
10
|
- 'pyproject.toml'
|
|
11
|
-
|
|
12
|
-
branches:
|
|
13
|
-
- main
|
|
11
|
+
workflow_dispatch:
|
|
14
12
|
|
|
15
13
|
permissions:
|
|
16
14
|
contents: read
|
|
@@ -152,21 +152,40 @@ class ConversionPipeline:
|
|
|
152
152
|
".pdf": self.pdf_handler,
|
|
153
153
|
".xls": self.tabular_handler,
|
|
154
154
|
".xlsx": self.tabular_handler,
|
|
155
|
-
".docx": self.office_handler,
|
|
156
|
-
".pptx": self.office_handler,
|
|
157
155
|
},
|
|
158
156
|
}
|
|
159
157
|
|
|
158
|
+
# Register Office extensions dynamically — the handler decides which
|
|
159
|
+
# legacy formats it supports based on Gotenberg availability. This
|
|
160
|
+
# keeps ``handlers_mapping`` (what consumers read) truthful at runtime.
|
|
161
|
+
for ext in self.office_handler.supported_extensions:
|
|
162
|
+
self.handlers_categories[ExtensionCategory.FORMATTED_DOCS][ext] = self.office_handler
|
|
163
|
+
|
|
160
164
|
# Normalize map of extensions to handlers
|
|
161
165
|
self.handlers_mapping: dict[str, BaseHandler] = {}
|
|
162
166
|
for handlers in self.handlers_categories.values():
|
|
163
167
|
for ext, handler in handlers.items():
|
|
164
168
|
self.handlers_mapping[ext] = handler
|
|
165
169
|
|
|
170
|
+
# Cache the read-only view so consumers get a stable object (and we
|
|
171
|
+
# don't rebuild the frozenset on every ``.supported_extensions`` access).
|
|
172
|
+
# ``handlers_mapping`` is read-only by convention.
|
|
173
|
+
self._supported_extensions: frozenset[str] = frozenset(self.handlers_mapping)
|
|
174
|
+
|
|
166
175
|
# ---------------------------------------------------------------------
|
|
167
176
|
# Public APIs
|
|
168
177
|
# ---------------------------------------------------------------------
|
|
169
178
|
|
|
179
|
+
@property
|
|
180
|
+
def supported_extensions(self) -> frozenset[str]:
|
|
181
|
+
"""Canonical set of extensions this pipeline can route.
|
|
182
|
+
|
|
183
|
+
This is the source of truth for consumers advertising supported
|
|
184
|
+
formats (e.g., a ``/extensions`` endpoint). ``handlers_mapping`` is
|
|
185
|
+
the routing table; ``supported_extensions`` is its read-only view.
|
|
186
|
+
"""
|
|
187
|
+
return self._supported_extensions
|
|
188
|
+
|
|
170
189
|
async def convert_document_to_md(
|
|
171
190
|
self,
|
|
172
191
|
file_path: str | Path,
|
|
@@ -0,0 +1,200 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Convert Office files to PDF via Gotenberg HTTP API, then process through PDFHandler.
|
|
3
|
+
|
|
4
|
+
Gotenberg (https://gotenberg.dev) is a Docker-based API that wraps LibreOffice
|
|
5
|
+
for document-to-PDF conversion. It handles concurrency internally and scales
|
|
6
|
+
horizontally via container replicas.
|
|
7
|
+
|
|
8
|
+
Configure via the ``GOTENBERG_URL`` environment variable or pass
|
|
9
|
+
``gotenberg_url=...`` explicitly. If neither is provided, the converter is
|
|
10
|
+
disabled and ``convert()`` returns ``None`` without any network I/O — this lets
|
|
11
|
+
downstream consumers advertise supported formats accurately.
|
|
12
|
+
|
|
13
|
+
If the Gotenberg service is configured but unreachable at call time,
|
|
14
|
+
``convert()`` returns ``None`` and the caller should fall through to the next
|
|
15
|
+
converter.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
import asyncio
|
|
19
|
+
import contextlib
|
|
20
|
+
import functools
|
|
21
|
+
import os
|
|
22
|
+
import tempfile
|
|
23
|
+
from collections.abc import Awaitable, Callable
|
|
24
|
+
from pathlib import Path
|
|
25
|
+
from typing import TypeVar
|
|
26
|
+
|
|
27
|
+
import httpx
|
|
28
|
+
|
|
29
|
+
from ..common.logger import logger
|
|
30
|
+
from .base import Converter
|
|
31
|
+
|
|
32
|
+
_COMPONENT = "GotenbergConverter"
|
|
33
|
+
_CONVERT_ENDPOINT = "/forms/libreoffice/convert"
|
|
34
|
+
|
|
35
|
+
# Transient failures worth retrying. ConnectError is intentionally excluded —
|
|
36
|
+
# "service unreachable" is handled as a graceful skip, not a retry.
|
|
37
|
+
_RETRYABLE_EXC: tuple[type[BaseException], ...] = (
|
|
38
|
+
httpx.TimeoutException,
|
|
39
|
+
httpx.RemoteProtocolError,
|
|
40
|
+
httpx.ReadError,
|
|
41
|
+
httpx.WriteError,
|
|
42
|
+
)
|
|
43
|
+
|
|
44
|
+
T = TypeVar("T")
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def _retry_transient(
|
|
48
|
+
attempts: int = 3, delays: tuple[float, ...] = (1.0, 3.0)
|
|
49
|
+
) -> Callable[[Callable[..., Awaitable[T | None]]], Callable[..., Awaitable[T | None]]]:
|
|
50
|
+
"""Retry an async method on transient httpx errors. Sleeps between attempts
|
|
51
|
+
are taken from ``delays`` (short, fixed backoff — enough to ride out a
|
|
52
|
+
worker restart or a brief connection hiccup)."""
|
|
53
|
+
|
|
54
|
+
def decorator(
|
|
55
|
+
fn: Callable[..., Awaitable[T | None]],
|
|
56
|
+
) -> Callable[..., Awaitable[T | None]]:
|
|
57
|
+
@functools.wraps(fn)
|
|
58
|
+
async def wrapper(self, *args, **kwargs) -> T | None:
|
|
59
|
+
tag = args[1] if len(args) >= 2 else kwargs.get("tag", _COMPONENT)
|
|
60
|
+
for i in range(attempts):
|
|
61
|
+
try:
|
|
62
|
+
return await fn(self, *args, **kwargs)
|
|
63
|
+
except _RETRYABLE_EXC as e:
|
|
64
|
+
if i == attempts - 1:
|
|
65
|
+
raise
|
|
66
|
+
sleep_for = delays[min(i, len(delays) - 1)]
|
|
67
|
+
logger.warning(
|
|
68
|
+
f"{tag} | transient {type(e).__name__}: {e}; "
|
|
69
|
+
f"retry {i + 1}/{attempts - 1} in {sleep_for:.1f}s"
|
|
70
|
+
)
|
|
71
|
+
await asyncio.sleep(sleep_for)
|
|
72
|
+
return None
|
|
73
|
+
|
|
74
|
+
return wrapper
|
|
75
|
+
|
|
76
|
+
return decorator
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
class GotenbergConverter(Converter):
|
|
80
|
+
"""
|
|
81
|
+
Convert Office files to PDF via Gotenberg, then delegate to PDFHandler
|
|
82
|
+
for per-page text extraction and OCR.
|
|
83
|
+
|
|
84
|
+
This captures embedded images via OCR that neither MarkItDown nor
|
|
85
|
+
DocIntelligence can extract from Office files directly.
|
|
86
|
+
|
|
87
|
+
Requires a running Gotenberg instance. Gracefully returns None if
|
|
88
|
+
the service is unavailable or if the converter is disabled, allowing
|
|
89
|
+
fallback to other converters.
|
|
90
|
+
"""
|
|
91
|
+
|
|
92
|
+
SUPPORTED_EXTENSIONS = frozenset({".docx", ".pptx", ".doc", ".ppt", ".odt", ".rtf"})
|
|
93
|
+
|
|
94
|
+
def __init__(self, gotenberg_url: str | None = None) -> None:
|
|
95
|
+
super().__init__(_COMPONENT)
|
|
96
|
+
# gotenberg_url is the single source of truth; ``enabled`` derives from it.
|
|
97
|
+
# Treat empty/whitespace values as unconfigured to catch a common
|
|
98
|
+
# ``docker-compose`` misconfiguration (``environment: GOTENBERG_URL=``).
|
|
99
|
+
raw = (gotenberg_url if gotenberg_url is not None else os.getenv("GOTENBERG_URL")) or ""
|
|
100
|
+
self.gotenberg_url: str | None = raw.strip() or None
|
|
101
|
+
|
|
102
|
+
@property
|
|
103
|
+
def enabled(self) -> bool:
|
|
104
|
+
return bool(self.gotenberg_url)
|
|
105
|
+
|
|
106
|
+
@property
|
|
107
|
+
def _convert_url(self) -> str:
|
|
108
|
+
"""The fully-qualified Gotenberg endpoint URL.
|
|
109
|
+
|
|
110
|
+
Only safe to access when ``self.enabled`` is True. Callers in this
|
|
111
|
+
class are already guarded by the ``if not self.enabled`` short-circuit
|
|
112
|
+
in ``convert()``; raising here makes the invariant explicit instead
|
|
113
|
+
of letting a ``None`` propagate into ``httpx`` or the format string.
|
|
114
|
+
"""
|
|
115
|
+
if not self.gotenberg_url:
|
|
116
|
+
raise RuntimeError(
|
|
117
|
+
"GotenbergConverter is not configured (set GOTENBERG_URL "
|
|
118
|
+
"or pass gotenberg_url=...). Check ``enabled`` before calling."
|
|
119
|
+
)
|
|
120
|
+
return f"{self.gotenberg_url.rstrip('/')}{_CONVERT_ENDPOINT}"
|
|
121
|
+
|
|
122
|
+
async def convert(self, file_path: str) -> str | None:
|
|
123
|
+
if not self.enabled:
|
|
124
|
+
return None
|
|
125
|
+
|
|
126
|
+
file_name = Path(file_path).name
|
|
127
|
+
tag = f"{_COMPONENT} | {file_name}"
|
|
128
|
+
|
|
129
|
+
logger.info(f"{tag} | converting to PDF via Gotenberg ({self.gotenberg_url})")
|
|
130
|
+
|
|
131
|
+
try:
|
|
132
|
+
pdf_bytes = await self._convert_to_pdf(file_path, tag)
|
|
133
|
+
except _RETRYABLE_EXC as e:
|
|
134
|
+
logger.error(f"{tag} | Gotenberg error after retries: {type(e).__name__}: {e}")
|
|
135
|
+
return None
|
|
136
|
+
if not pdf_bytes:
|
|
137
|
+
return None
|
|
138
|
+
|
|
139
|
+
# Write PDF to temp file and process through PDFHandler
|
|
140
|
+
tmp_pdf = None
|
|
141
|
+
try:
|
|
142
|
+
fd, tmp_pdf = tempfile.mkstemp(suffix=".pdf", prefix="mkdpro_")
|
|
143
|
+
os.close(fd)
|
|
144
|
+
Path(tmp_pdf).write_bytes(pdf_bytes)
|
|
145
|
+
|
|
146
|
+
logger.info(f"{tag} | PDF received ({len(pdf_bytes)} bytes), delegating to PDFHandler")
|
|
147
|
+
|
|
148
|
+
# Import here to avoid circular import
|
|
149
|
+
from ..handlers.pdf_handler import PDFHandler
|
|
150
|
+
|
|
151
|
+
pdf_handler = PDFHandler()
|
|
152
|
+
try:
|
|
153
|
+
result = await pdf_handler.handle(tmp_pdf, force_ocr=True)
|
|
154
|
+
if result:
|
|
155
|
+
logger.info(f"{tag} | PDFHandler succeeded ({len(result)} chars)")
|
|
156
|
+
return result
|
|
157
|
+
finally:
|
|
158
|
+
await pdf_handler.aclose()
|
|
159
|
+
|
|
160
|
+
except Exception as e:
|
|
161
|
+
logger.error(f"{tag} | error processing PDF: {e}")
|
|
162
|
+
return None
|
|
163
|
+
finally:
|
|
164
|
+
if tmp_pdf and Path(tmp_pdf).exists():
|
|
165
|
+
with contextlib.suppress(Exception):
|
|
166
|
+
os.unlink(tmp_pdf)
|
|
167
|
+
|
|
168
|
+
@_retry_transient(attempts=3, delays=(1.0, 3.0))
|
|
169
|
+
async def _convert_to_pdf(self, file_path: str, tag: str) -> bytes | None:
|
|
170
|
+
"""Send file to Gotenberg and return PDF bytes."""
|
|
171
|
+
try:
|
|
172
|
+
async with httpx.AsyncClient(timeout=120) as client:
|
|
173
|
+
with open(file_path, "rb") as f:
|
|
174
|
+
response = await client.post(
|
|
175
|
+
self._convert_url,
|
|
176
|
+
files={"files": (Path(file_path).name, f)},
|
|
177
|
+
data={
|
|
178
|
+
"losslessImageCompression": "true",
|
|
179
|
+
"quality": "100",
|
|
180
|
+
},
|
|
181
|
+
)
|
|
182
|
+
|
|
183
|
+
if response.status_code == 200:
|
|
184
|
+
return response.content
|
|
185
|
+
|
|
186
|
+
logger.error(
|
|
187
|
+
f"{tag} | Gotenberg returned {response.status_code}: {response.text[:200]}"
|
|
188
|
+
)
|
|
189
|
+
return None
|
|
190
|
+
|
|
191
|
+
except httpx.ConnectError:
|
|
192
|
+
logger.info(f"{tag} | Gotenberg not reachable at {self.gotenberg_url}, skipping")
|
|
193
|
+
return None
|
|
194
|
+
except _RETRYABLE_EXC:
|
|
195
|
+
# Let the retry decorator handle transient errors. If retries are
|
|
196
|
+
# exhausted it re-raises, and the caller in ``convert()`` logs.
|
|
197
|
+
raise
|
|
198
|
+
except Exception as e:
|
|
199
|
+
logger.error(f"{tag} | Gotenberg error: {e}")
|
|
200
|
+
return None
|
{markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/converters/gpt_vision_converter.py
RENAMED
|
@@ -11,6 +11,8 @@ Returns Markdown when successful, or `None` if conversion produced
|
|
|
11
11
|
insufficient content (as determined by `ensure_minimum_content` in the service).
|
|
12
12
|
"""
|
|
13
13
|
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
|
|
14
16
|
from ..common import config
|
|
15
17
|
from ..common.utils import is_pdf
|
|
16
18
|
from ..services.openai_services import GPTVision
|
|
@@ -115,6 +117,7 @@ class GPTVisionConverter(Converter):
|
|
|
115
117
|
max_retries=_retries,
|
|
116
118
|
base_delay=base_delay,
|
|
117
119
|
max_delay=max_delay,
|
|
120
|
+
context_label=Path(file_path).name,
|
|
118
121
|
)
|
|
119
122
|
|
|
120
123
|
async def aclose(self) -> None:
|
|
@@ -38,11 +38,22 @@ class OfficeHandler(BaseHandler):
|
|
|
38
38
|
the pipeline order when initializing the handler.
|
|
39
39
|
|
|
40
40
|
Supported formats: .docx, .pptx (modern Office XML formats).
|
|
41
|
-
|
|
41
|
+
Legacy binary formats (.doc, .ppt, .odt, .rtf) are supported at runtime
|
|
42
|
+
only when the configured pipeline contains an enabled ``GotenbergConverter``.
|
|
43
|
+
Callers should inspect the instance attribute ``supported_extensions`` (not
|
|
44
|
+
the class constant) to learn which formats this handler will actually
|
|
45
|
+
process.
|
|
42
46
|
"""
|
|
43
47
|
|
|
48
|
+
# Class constant stays narrow (the always-supported modern formats) so that
|
|
49
|
+
# ``BaseHandler.is_valid`` and any class-level introspection never falsely
|
|
50
|
+
# advertise legacy formats that would fail without Gotenberg. Dynamic
|
|
51
|
+
# runtime capability lives on the instance attribute ``supported_extensions``
|
|
52
|
+
# set in ``__init__``.
|
|
44
53
|
SUPPORTED_EXTENSIONS = frozenset({".docx", ".pptx"})
|
|
45
54
|
|
|
55
|
+
_GOTENBERG_ONLY_EXTS = frozenset({".doc", ".ppt", ".odt", ".rtf"})
|
|
56
|
+
|
|
46
57
|
def __init__(
|
|
47
58
|
self,
|
|
48
59
|
pipeline: list[tuple[Converter, str]] | None = None,
|
|
@@ -62,6 +73,26 @@ class OfficeHandler(BaseHandler):
|
|
|
62
73
|
(self.markitdown, "MarkItDown"),
|
|
63
74
|
]
|
|
64
75
|
|
|
76
|
+
# isinstance catches subclasses/proxies users might pass in custom
|
|
77
|
+
# pipelines. We explicitly check that the imported ``GotenbergConverter``
|
|
78
|
+
# is still a real class before using it as an isinstance second arg —
|
|
79
|
+
# if tests patch the name to a MagicMock *instance* (not a class),
|
|
80
|
+
# we fall back to ``gotenberg_enabled = False`` instead of wrapping
|
|
81
|
+
# the whole generator in try/except TypeError, which would otherwise
|
|
82
|
+
# silently swallow a real bug inside an ``enabled`` property.
|
|
83
|
+
if isinstance(GotenbergConverter, type):
|
|
84
|
+
gotenberg_enabled = any(
|
|
85
|
+
isinstance(c, GotenbergConverter) and getattr(c, "enabled", False)
|
|
86
|
+
for c, _ in self._pipeline
|
|
87
|
+
)
|
|
88
|
+
else:
|
|
89
|
+
gotenberg_enabled = False
|
|
90
|
+
self.supported_extensions: frozenset[str] = (
|
|
91
|
+
self.SUPPORTED_EXTENSIONS | self._GOTENBERG_ONLY_EXTS
|
|
92
|
+
if gotenberg_enabled
|
|
93
|
+
else self.SUPPORTED_EXTENSIONS
|
|
94
|
+
)
|
|
95
|
+
|
|
65
96
|
async def handle(self, file_path: str, *args, **kwargs) -> str | None:
|
|
66
97
|
"""
|
|
67
98
|
Convert an Office document to Markdown by trying the configured pipeline in order.
|
|
@@ -158,6 +158,7 @@ class PagePDFConverter:
|
|
|
158
158
|
png_path = await asyncio.to_thread(self._render_page_to_png, file_path, page_idx)
|
|
159
159
|
content = await self.gpt_vision.gpt_vision.process_image(
|
|
160
160
|
file_or_url=png_path,
|
|
161
|
+
context_label=f"{Path(file_path).name} page {page_idx + 1}",
|
|
161
162
|
)
|
|
162
163
|
if content:
|
|
163
164
|
logger.info(f"{tag} | succeeded ({len(content)} chars)")
|
|
@@ -240,6 +240,7 @@ class GPTVision:
|
|
|
240
240
|
base_delay: float = 0.5,
|
|
241
241
|
max_delay: float = 20.0,
|
|
242
242
|
retry_http_statuses: tuple[int, ...] = _RETRY_HTTP_STATUSES,
|
|
243
|
+
context_label: str | None = None,
|
|
243
244
|
):
|
|
244
245
|
"""
|
|
245
246
|
Call self.lang_client.ainvoke with:
|
|
@@ -275,8 +276,9 @@ class GPTVision:
|
|
|
275
276
|
raise last_err
|
|
276
277
|
cap = min(max_delay, base_delay * (2 ** (attempt - 1)))
|
|
277
278
|
sleep_for = random.uniform(0, cap)
|
|
279
|
+
label = context_label or file_or_url
|
|
278
280
|
logger.warning(
|
|
279
|
-
f"GPTVision: ainvoke transient error {
|
|
281
|
+
f"GPTVision: ainvoke transient error {label}: attempt {attempt}/{max_retries}: "
|
|
280
282
|
f"{type(last_err).__name__}: {last_err}. Retrying in {sleep_for:.2f}s"
|
|
281
283
|
)
|
|
282
284
|
await asyncio.sleep(sleep_for)
|
|
@@ -331,6 +333,7 @@ class GPTVision:
|
|
|
331
333
|
max_retries: int = 6,
|
|
332
334
|
base_delay: float = 0.5,
|
|
333
335
|
max_delay: float = 20.0,
|
|
336
|
+
context_label: str | None = None,
|
|
334
337
|
) -> str | None:
|
|
335
338
|
"""
|
|
336
339
|
OCR a single image (local path or URL) with the vision model.
|
|
@@ -373,6 +376,7 @@ class GPTVision:
|
|
|
373
376
|
max_retries=max_retries,
|
|
374
377
|
base_delay=base_delay,
|
|
375
378
|
max_delay=max_delay,
|
|
379
|
+
context_label=context_label,
|
|
376
380
|
)
|
|
377
381
|
finally:
|
|
378
382
|
with suppress(Exception):
|
|
@@ -469,7 +473,11 @@ class GPTVision:
|
|
|
469
473
|
# Per-page hard timeout so one slow page cannot stall the run
|
|
470
474
|
try:
|
|
471
475
|
partial_md = await asyncio.wait_for(
|
|
472
|
-
self.process_image(
|
|
476
|
+
self.process_image(
|
|
477
|
+
png_path,
|
|
478
|
+
context_label=f"{file_stem} page {page_index + 1}/{num_pages}",
|
|
479
|
+
),
|
|
480
|
+
timeout=self.page_timeout_s,
|
|
473
481
|
)
|
|
474
482
|
except TimeoutError:
|
|
475
483
|
logger.error(
|
|
Binary file
|
|
Binary file
|
|
@@ -74,6 +74,62 @@ gotenberg_converter:
|
|
|
74
74
|
expect: ["what is lorem ipsum?", "where does it come from"]
|
|
75
75
|
|
|
76
76
|
|
|
77
|
+
# =============================================================================
|
|
78
|
+
# LEGACY OFFICE FORMATS — Gotenberg-gated (.doc / .ppt / .odt / .rtf)
|
|
79
|
+
# Same fixtures as above re-saved / pandoc-converted into legacy formats.
|
|
80
|
+
# Routed through OfficeHandler → Gotenberg → PDF → PDFHandler per-page OCR.
|
|
81
|
+
# Only available iff GOTENBERG_URL is configured on the pipeline.
|
|
82
|
+
# =============================================================================
|
|
83
|
+
legacy_office_formats:
|
|
84
|
+
# .doc (re-saved from .docx via Word)
|
|
85
|
+
- file: tests/data/word-legacy/lorem-ipsum-only-text.doc
|
|
86
|
+
expect: ["what is lorem ipsum?", "where does it come from"]
|
|
87
|
+
- file: tests/data/word-legacy/lorem-ipsum-text-with-image.doc
|
|
88
|
+
expect: ["what is lorem ipsum?", "red", "green", "blue"]
|
|
89
|
+
- file: tests/data/word-legacy/geometry-colors.doc
|
|
90
|
+
expect: ["red", "green", "blue"]
|
|
91
|
+
- file: tests/data/word-legacy/lorem-ipsum-scanned-text.doc
|
|
92
|
+
expect: ["what is lorem ipsum?", "why do we use it?"]
|
|
93
|
+
- file: tests/data/word-legacy/lorem-ipsum-text-with-scanned-text.doc
|
|
94
|
+
expect: ["what is lorem ipsum?", "where does it come from"]
|
|
95
|
+
|
|
96
|
+
# .ppt (re-saved from .pptx via PowerPoint)
|
|
97
|
+
- file: tests/data/powerpoint-legacy/lorem-ipsum-only-text.ppt
|
|
98
|
+
expect: ["what is lorem ipsum?", "where does it come from"]
|
|
99
|
+
- file: tests/data/powerpoint-legacy/lorem-ipsum-text-with-image.ppt
|
|
100
|
+
expect: ["what is lorem ipsum", "red", "green", "blue"]
|
|
101
|
+
- file: tests/data/powerpoint-legacy/geometry-colors.ppt
|
|
102
|
+
expect: ["red", "green", "blue"]
|
|
103
|
+
- file: tests/data/powerpoint-legacy/lorem-ipsum-scanned-text.ppt
|
|
104
|
+
expect: ["what is lorem ipsum", "why do we use it"]
|
|
105
|
+
- file: tests/data/powerpoint-legacy/lorem-ipsum-text-with-scanned-text.ppt
|
|
106
|
+
expect: ["what is lorem ipsum?", "where does it come from"]
|
|
107
|
+
|
|
108
|
+
# .odt (re-saved from .docx via Word)
|
|
109
|
+
- file: tests/data/word-legacy/lorem-ipsum-only-text.odt
|
|
110
|
+
expect: ["what is lorem ipsum?", "where does it come from"]
|
|
111
|
+
- file: tests/data/word-legacy/lorem-ipsum-text-with-image.odt
|
|
112
|
+
expect: ["what is lorem ipsum?", "red", "green", "blue"]
|
|
113
|
+
- file: tests/data/word-legacy/geometry-colors.odt
|
|
114
|
+
expect: ["red", "green", "blue"]
|
|
115
|
+
- file: tests/data/word-legacy/lorem-ipsum-scanned-text.odt
|
|
116
|
+
expect: ["what is lorem ipsum?", "why do we use it?"]
|
|
117
|
+
- file: tests/data/word-legacy/lorem-ipsum-text-with-scanned-text.odt
|
|
118
|
+
expect: ["what is lorem ipsum?", "where does it come from"]
|
|
119
|
+
|
|
120
|
+
# .rtf (re-saved from .docx via Word)
|
|
121
|
+
- file: tests/data/word-legacy/lorem-ipsum-only-text.rtf
|
|
122
|
+
expect: ["what is lorem ipsum?", "where does it come from"]
|
|
123
|
+
- file: tests/data/word-legacy/lorem-ipsum-text-with-image.rtf
|
|
124
|
+
expect: ["what is lorem ipsum?", "red", "green", "blue"]
|
|
125
|
+
- file: tests/data/word-legacy/geometry-colors.rtf
|
|
126
|
+
expect: ["red", "green", "blue"]
|
|
127
|
+
- file: tests/data/word-legacy/lorem-ipsum-scanned-text.rtf
|
|
128
|
+
expect: ["what is lorem ipsum?", "why do we use it?"]
|
|
129
|
+
- file: tests/data/word-legacy/lorem-ipsum-text-with-scanned-text.rtf
|
|
130
|
+
expect: ["what is lorem ipsum?", "where does it come from"]
|
|
131
|
+
|
|
132
|
+
|
|
77
133
|
# =============================================================================
|
|
78
134
|
# MARKITDOWN CONVERTER — text extraction only, NO OCR, NO image content
|
|
79
135
|
# =============================================================================
|
|
Binary file
|
|
Binary file
|