markitdown-pro 2.0.0__tar.gz → 2.1.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (174) hide show
  1. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/.github/workflows/test.yml +2 -4
  2. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/.gitignore +2 -0
  3. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/PKG-INFO +1 -1
  4. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/conversion_pipeline.py +21 -2
  5. markitdown_pro-2.1.1/markitdown_pro/converters/gotenberg_converter.py +200 -0
  6. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/converters/gpt_vision_converter.py +3 -0
  7. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/handlers/office_handler.py +32 -1
  8. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/handlers/page_pdf_converter.py +1 -0
  9. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/services/openai_services.py +10 -2
  10. markitdown_pro-2.1.1/tests/data/powerpoint-legacy/geometry-colors.ppt +0 -0
  11. markitdown_pro-2.1.1/tests/data/powerpoint-legacy/lorem-ipsum-only-text.ppt +0 -0
  12. markitdown_pro-2.1.1/tests/data/powerpoint-legacy/lorem-ipsum-scanned-text.ppt +0 -0
  13. markitdown_pro-2.1.1/tests/data/powerpoint-legacy/lorem-ipsum-text-with-image.ppt +0 -0
  14. markitdown_pro-2.1.1/tests/data/powerpoint-legacy/lorem-ipsum-text-with-scanned-text.ppt +0 -0
  15. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/test_expectations.yaml +56 -0
  16. markitdown_pro-2.1.1/tests/data/word-legacy/geometry-colors.doc +0 -0
  17. markitdown_pro-2.1.1/tests/data/word-legacy/geometry-colors.odt +0 -0
  18. markitdown_pro-2.1.1/tests/data/word-legacy/geometry-colors.rtf +2197 -0
  19. markitdown_pro-2.1.1/tests/data/word-legacy/lorem-ipsum-only-text.doc +0 -0
  20. markitdown_pro-2.1.1/tests/data/word-legacy/lorem-ipsum-only-text.odt +0 -0
  21. markitdown_pro-2.1.1/tests/data/word-legacy/lorem-ipsum-only-text.rtf +258 -0
  22. markitdown_pro-2.1.1/tests/data/word-legacy/lorem-ipsum-scanned-text.doc +0 -0
  23. markitdown_pro-2.1.1/tests/data/word-legacy/lorem-ipsum-scanned-text.odt +0 -0
  24. markitdown_pro-2.1.1/tests/data/word-legacy/lorem-ipsum-scanned-text.rtf +18373 -0
  25. markitdown_pro-2.1.1/tests/data/word-legacy/lorem-ipsum-text-with-image.doc +0 -0
  26. markitdown_pro-2.1.1/tests/data/word-legacy/lorem-ipsum-text-with-image.odt +0 -0
  27. markitdown_pro-2.1.1/tests/data/word-legacy/lorem-ipsum-text-with-image.rtf +2204 -0
  28. markitdown_pro-2.1.1/tests/data/word-legacy/lorem-ipsum-text-with-scanned-text.doc +0 -0
  29. markitdown_pro-2.1.1/tests/data/word-legacy/lorem-ipsum-text-with-scanned-text.odt +0 -0
  30. markitdown_pro-2.1.1/tests/data/word-legacy/lorem-ipsum-text-with-scanned-text.rtf +7208 -0
  31. markitdown_pro-2.1.1/tests/integration/test_legacy_office_formats.py +108 -0
  32. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/unit/test_conversion_pipeline.py +74 -2
  33. markitdown_pro-2.1.1/tests/unit/test_gotenberg_converter.py +69 -0
  34. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/unit/test_office_handler.py +62 -0
  35. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/uv.lock +0 -1
  36. markitdown_pro-2.0.0/markitdown_pro/converters/gotenberg_converter.py +0 -118
  37. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/.github/workflows/lint.yaml +0 -0
  38. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/.github/workflows/publish.yaml +0 -0
  39. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/.vscode/settings.json +0 -0
  40. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/README.md +0 -0
  41. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/__init__.py +0 -0
  42. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/__init__.py +0 -0
  43. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/common/__init__.py +0 -0
  44. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/common/config.py +0 -0
  45. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/common/isolated_worker.py +0 -0
  46. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/common/logger.py +0 -0
  47. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/common/schemas.py +0 -0
  48. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/common/utils.py +0 -0
  49. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/converters/__init__.py +0 -0
  50. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/converters/base.py +0 -0
  51. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/converters/doc_intel_converter.py +0 -0
  52. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/converters/markitdown_converter.py +0 -0
  53. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/converters/pymupdf_converter.py +0 -0
  54. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/converters/speech_converter.py +0 -0
  55. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/converters/tabular_converter.py +0 -0
  56. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/converters/unstructured_converter.py +0 -0
  57. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/converters/youtube_converter.py +0 -0
  58. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/handlers/__init__.py +0 -0
  59. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/handlers/audio_handler.py +0 -0
  60. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/handlers/base_handler.py +0 -0
  61. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/handlers/email_handler.py +0 -0
  62. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/handlers/epub_handler.py +0 -0
  63. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/handlers/image_handler.py +0 -0
  64. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/handlers/ipynb_handler.py +0 -0
  65. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/handlers/markup_handler.py +0 -0
  66. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/handlers/pdf_handler.py +0 -0
  67. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/handlers/pst_handler.py +0 -0
  68. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/handlers/tabular_handler.py +0 -0
  69. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/handlers/text_handler.py +0 -0
  70. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/services/__init__.py +0 -0
  71. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/services/azure_doc_intelligence.py +0 -0
  72. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/markitdown_pro/services/azure_speech.py +0 -0
  73. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/pyproject.toml +0 -0
  74. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/__init__.py +0 -0
  75. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/common/__init__.py +0 -0
  76. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/common/test_config.py +0 -0
  77. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/conftest.py +0 -0
  78. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/conversion_pipeline/__init__.py +0 -0
  79. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/conversion_pipeline/test_conversion_pipeline.py +0 -0
  80. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/converters/test_azure_doc_intel_wrapper.py +0 -0
  81. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/converters/test_azure_speech_wrapper.py +0 -0
  82. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/converters/test_doc_intel_lifecycle.py +0 -0
  83. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/converters/test_gpt_vision_config.py +0 -0
  84. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/converters/test_gpt_vision_wrapper.py +0 -0
  85. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/converters/test_pymupdf_wrapper.py +0 -0
  86. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/converters/test_pymupdf_wrapper_async.py +0 -0
  87. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/converters/test_unstructured_io_wrapper.py +0 -0
  88. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/audio/the-power-of-small-steps_cn.wav +0 -0
  89. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/audio/the-power-of-small-steps_en.mp3 +0 -0
  90. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/audio/the-power-of-small-steps_en.wav +0 -0
  91. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/audio/the-power-of-small-steps_es.mp3 +0 -0
  92. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/audio/the-power-of-small-steps_es.wav +0 -0
  93. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/audio/the-power-of-small-steps_it.wav +0 -0
  94. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/audio/the-power-of-small-steps_jp.mp3 +0 -0
  95. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/audio/the-power-of-small-steps_jp.wav +0 -0
  96. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/eml/email-with-image.eml +0 -0
  97. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/eml/mime-different-plain-html.eml +0 -0
  98. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/epub/sample1.epub +0 -0
  99. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/images/flower.jpg +0 -0
  100. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/images/geometry_colors.bmp +0 -0
  101. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/images/handwriting.jpg +0 -0
  102. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/images/scanned_text.heic +0 -0
  103. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/images/scanned_text.png +0 -0
  104. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/images/text_with_image.jpg +0 -0
  105. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/markup/blog.html +0 -0
  106. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/markup/factbook.xml +0 -0
  107. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/markup/simple.json +0 -0
  108. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/markup/simple.yaml +0 -0
  109. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/msg/fake-email-multiple-attachments.msg +0 -0
  110. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/msg/fake-email.msg +0 -0
  111. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/pdf/geometry-colors.pdf +0 -0
  112. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/pdf/lorem-ipsum-only-images.pdf +0 -0
  113. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/pdf/lorem-ipsum-only-text.pdf +0 -0
  114. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/pdf/lorem-ipsum-scanned-text.pdf +0 -0
  115. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/pdf/lorem-ipsum-text-with-image.pdf +0 -0
  116. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/powerpoint/geometry-colors.pptx +0 -0
  117. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/powerpoint/lorem-ipsum-only-text.pptx +0 -0
  118. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/powerpoint/lorem-ipsum-scanned-text.pptx +0 -0
  119. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/powerpoint/lorem-ipsum-text-with-image.pptx +0 -0
  120. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/powerpoint/lorem-ipsum-text-with-scanned-text.pptx +0 -0
  121. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/tabular/2023-half-year-analyses-by-segment.xlsx +0 -0
  122. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/tabular/stanley-cups.csv +0 -0
  123. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/tabular/stanley-cups.tsv +0 -0
  124. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/tabular/stanley-cups.xlsx +0 -0
  125. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/text/book-war-and-peace-1p.txt +0 -0
  126. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/text/fake-email.txt +0 -0
  127. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/text/logger.py +0 -0
  128. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/word/geometry-colors.docx +0 -0
  129. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/word/lorem-ipsum-only-text.docx +0 -0
  130. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/word/lorem-ipsum-scanned-text.docx +0 -0
  131. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/word/lorem-ipsum-text-with-image.docx +0 -0
  132. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/data/word/lorem-ipsum-text-with-scanned-text.docx +0 -0
  133. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/fixtures.py +0 -0
  134. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/handlers/__init__.py +0 -0
  135. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/handlers/test_email_handler.py +0 -0
  136. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/handlers/test_epub_handler.py +0 -0
  137. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/handlers/test_image_handler.py +0 -0
  138. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/handlers/test_ipynb_handler.py +0 -0
  139. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/handlers/test_markitdown_handler.py +0 -0
  140. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/handlers/test_markitdown_validation.py +0 -0
  141. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/handlers/test_markup_handler.py +0 -0
  142. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/handlers/test_page_pdf_converter.py +0 -0
  143. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/handlers/test_pdf_handler.py +0 -0
  144. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/handlers/test_pst_handler.py +0 -0
  145. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/handlers/test_tabular_handler.py +0 -0
  146. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/handlers/test_text_handler.py +0 -0
  147. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/integration/__init__.py +0 -0
  148. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/integration/cheat_sheet.py +0 -0
  149. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/integration/runner.py +0 -0
  150. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/integration/test_audio_handler.py +0 -0
  151. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/integration/test_doc_intelligence_converter.py +0 -0
  152. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/integration/test_email_handler.py +0 -0
  153. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/integration/test_gotenberg_converter.py +0 -0
  154. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/integration/test_gpt_vision_converter.py +0 -0
  155. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/integration/test_markitdown_converter.py +0 -0
  156. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/integration/test_markup_handler.py +0 -0
  157. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/integration/test_pdf_handler.py +0 -0
  158. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/integration/test_pymupdf_converter.py +0 -0
  159. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/integration/test_tabular_handler.py +0 -0
  160. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/integration/test_text_handler.py +0 -0
  161. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/page_count/__init__.py +0 -0
  162. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/page_count/test_pdf_page_count.py +0 -0
  163. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/unit/__init__.py +0 -0
  164. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/unit/test_audio_handler.py +0 -0
  165. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/unit/test_base_converter.py +0 -0
  166. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/unit/test_doc_intel_converter.py +0 -0
  167. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/unit/test_gpt_vision_converter.py +0 -0
  168. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/unit/test_image_handler.py +0 -0
  169. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/unit/test_markitdown_converter.py +0 -0
  170. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/unit/test_page_pdf_converter.py +0 -0
  171. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/unit/test_pdf_handler.py +0 -0
  172. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/unit/test_pymupdf_converter.py +0 -0
  173. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/unit/test_speech_converter.py +0 -0
  174. {markitdown_pro-2.0.0 → markitdown_pro-2.1.1}/tests/utils.py +0 -0
@@ -1,16 +1,14 @@
1
1
  name: '🧪 Test package'
2
2
 
3
3
  on:
4
- push:
4
+ pull_request:
5
5
  branches:
6
6
  - main
7
7
  paths:
8
8
  - 'markitdown_pro/**'
9
9
  - 'tests/**'
10
10
  - 'pyproject.toml'
11
- pull_request:
12
- branches:
13
- - main
11
+ workflow_dispatch:
14
12
 
15
13
  permissions:
16
14
  contents: read
@@ -24,3 +24,5 @@ docs/
24
24
  !./CLAUDE.md
25
25
  **/CONTINUITY.md
26
26
  **/CHANGELOG.md
27
+ .mcp.json
28
+ .worktrees/
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: markitdown-pro
3
- Version: 2.0.0
3
+ Version: 2.1.1
4
4
  Summary: A package that converts almost any file format to Markdown.
5
5
  Author: Developer
6
6
  License-Expression: MIT
@@ -152,21 +152,40 @@ class ConversionPipeline:
152
152
  ".pdf": self.pdf_handler,
153
153
  ".xls": self.tabular_handler,
154
154
  ".xlsx": self.tabular_handler,
155
- ".docx": self.office_handler,
156
- ".pptx": self.office_handler,
157
155
  },
158
156
  }
159
157
 
158
+ # Register Office extensions dynamically — the handler decides which
159
+ # legacy formats it supports based on Gotenberg availability. This
160
+ # keeps ``handlers_mapping`` (what consumers read) truthful at runtime.
161
+ for ext in self.office_handler.supported_extensions:
162
+ self.handlers_categories[ExtensionCategory.FORMATTED_DOCS][ext] = self.office_handler
163
+
160
164
  # Normalize map of extensions to handlers
161
165
  self.handlers_mapping: dict[str, BaseHandler] = {}
162
166
  for handlers in self.handlers_categories.values():
163
167
  for ext, handler in handlers.items():
164
168
  self.handlers_mapping[ext] = handler
165
169
 
170
+ # Cache the read-only view so consumers get a stable object (and we
171
+ # don't rebuild the frozenset on every ``.supported_extensions`` access).
172
+ # ``handlers_mapping`` is read-only by convention.
173
+ self._supported_extensions: frozenset[str] = frozenset(self.handlers_mapping)
174
+
166
175
  # ---------------------------------------------------------------------
167
176
  # Public APIs
168
177
  # ---------------------------------------------------------------------
169
178
 
179
+ @property
180
+ def supported_extensions(self) -> frozenset[str]:
181
+ """Canonical set of extensions this pipeline can route.
182
+
183
+ This is the source of truth for consumers advertising supported
184
+ formats (e.g., a ``/extensions`` endpoint). ``handlers_mapping`` is
185
+ the routing table; ``supported_extensions`` is its read-only view.
186
+ """
187
+ return self._supported_extensions
188
+
170
189
  async def convert_document_to_md(
171
190
  self,
172
191
  file_path: str | Path,
@@ -0,0 +1,200 @@
1
+ """
2
+ Convert Office files to PDF via Gotenberg HTTP API, then process through PDFHandler.
3
+
4
+ Gotenberg (https://gotenberg.dev) is a Docker-based API that wraps LibreOffice
5
+ for document-to-PDF conversion. It handles concurrency internally and scales
6
+ horizontally via container replicas.
7
+
8
+ Configure via the ``GOTENBERG_URL`` environment variable or pass
9
+ ``gotenberg_url=...`` explicitly. If neither is provided, the converter is
10
+ disabled and ``convert()`` returns ``None`` without any network I/O — this lets
11
+ downstream consumers advertise supported formats accurately.
12
+
13
+ If the Gotenberg service is configured but unreachable at call time,
14
+ ``convert()`` returns ``None`` and the caller should fall through to the next
15
+ converter.
16
+ """
17
+
18
+ import asyncio
19
+ import contextlib
20
+ import functools
21
+ import os
22
+ import tempfile
23
+ from collections.abc import Awaitable, Callable
24
+ from pathlib import Path
25
+ from typing import TypeVar
26
+
27
+ import httpx
28
+
29
+ from ..common.logger import logger
30
+ from .base import Converter
31
+
32
+ _COMPONENT = "GotenbergConverter"
33
+ _CONVERT_ENDPOINT = "/forms/libreoffice/convert"
34
+
35
+ # Transient failures worth retrying. ConnectError is intentionally excluded —
36
+ # "service unreachable" is handled as a graceful skip, not a retry.
37
+ _RETRYABLE_EXC: tuple[type[BaseException], ...] = (
38
+ httpx.TimeoutException,
39
+ httpx.RemoteProtocolError,
40
+ httpx.ReadError,
41
+ httpx.WriteError,
42
+ )
43
+
44
+ T = TypeVar("T")
45
+
46
+
47
+ def _retry_transient(
48
+ attempts: int = 3, delays: tuple[float, ...] = (1.0, 3.0)
49
+ ) -> Callable[[Callable[..., Awaitable[T | None]]], Callable[..., Awaitable[T | None]]]:
50
+ """Retry an async method on transient httpx errors. Sleeps between attempts
51
+ are taken from ``delays`` (short, fixed backoff — enough to ride out a
52
+ worker restart or a brief connection hiccup)."""
53
+
54
+ def decorator(
55
+ fn: Callable[..., Awaitable[T | None]],
56
+ ) -> Callable[..., Awaitable[T | None]]:
57
+ @functools.wraps(fn)
58
+ async def wrapper(self, *args, **kwargs) -> T | None:
59
+ tag = args[1] if len(args) >= 2 else kwargs.get("tag", _COMPONENT)
60
+ for i in range(attempts):
61
+ try:
62
+ return await fn(self, *args, **kwargs)
63
+ except _RETRYABLE_EXC as e:
64
+ if i == attempts - 1:
65
+ raise
66
+ sleep_for = delays[min(i, len(delays) - 1)]
67
+ logger.warning(
68
+ f"{tag} | transient {type(e).__name__}: {e}; "
69
+ f"retry {i + 1}/{attempts - 1} in {sleep_for:.1f}s"
70
+ )
71
+ await asyncio.sleep(sleep_for)
72
+ return None
73
+
74
+ return wrapper
75
+
76
+ return decorator
77
+
78
+
79
+ class GotenbergConverter(Converter):
80
+ """
81
+ Convert Office files to PDF via Gotenberg, then delegate to PDFHandler
82
+ for per-page text extraction and OCR.
83
+
84
+ This captures embedded images via OCR that neither MarkItDown nor
85
+ DocIntelligence can extract from Office files directly.
86
+
87
+ Requires a running Gotenberg instance. Gracefully returns None if
88
+ the service is unavailable or if the converter is disabled, allowing
89
+ fallback to other converters.
90
+ """
91
+
92
+ SUPPORTED_EXTENSIONS = frozenset({".docx", ".pptx", ".doc", ".ppt", ".odt", ".rtf"})
93
+
94
+ def __init__(self, gotenberg_url: str | None = None) -> None:
95
+ super().__init__(_COMPONENT)
96
+ # gotenberg_url is the single source of truth; ``enabled`` derives from it.
97
+ # Treat empty/whitespace values as unconfigured to catch a common
98
+ # ``docker-compose`` misconfiguration (``environment: GOTENBERG_URL=``).
99
+ raw = (gotenberg_url if gotenberg_url is not None else os.getenv("GOTENBERG_URL")) or ""
100
+ self.gotenberg_url: str | None = raw.strip() or None
101
+
102
+ @property
103
+ def enabled(self) -> bool:
104
+ return bool(self.gotenberg_url)
105
+
106
+ @property
107
+ def _convert_url(self) -> str:
108
+ """The fully-qualified Gotenberg endpoint URL.
109
+
110
+ Only safe to access when ``self.enabled`` is True. Callers in this
111
+ class are already guarded by the ``if not self.enabled`` short-circuit
112
+ in ``convert()``; raising here makes the invariant explicit instead
113
+ of letting a ``None`` propagate into ``httpx`` or the format string.
114
+ """
115
+ if not self.gotenberg_url:
116
+ raise RuntimeError(
117
+ "GotenbergConverter is not configured (set GOTENBERG_URL "
118
+ "or pass gotenberg_url=...). Check ``enabled`` before calling."
119
+ )
120
+ return f"{self.gotenberg_url.rstrip('/')}{_CONVERT_ENDPOINT}"
121
+
122
+ async def convert(self, file_path: str) -> str | None:
123
+ if not self.enabled:
124
+ return None
125
+
126
+ file_name = Path(file_path).name
127
+ tag = f"{_COMPONENT} | {file_name}"
128
+
129
+ logger.info(f"{tag} | converting to PDF via Gotenberg ({self.gotenberg_url})")
130
+
131
+ try:
132
+ pdf_bytes = await self._convert_to_pdf(file_path, tag)
133
+ except _RETRYABLE_EXC as e:
134
+ logger.error(f"{tag} | Gotenberg error after retries: {type(e).__name__}: {e}")
135
+ return None
136
+ if not pdf_bytes:
137
+ return None
138
+
139
+ # Write PDF to temp file and process through PDFHandler
140
+ tmp_pdf = None
141
+ try:
142
+ fd, tmp_pdf = tempfile.mkstemp(suffix=".pdf", prefix="mkdpro_")
143
+ os.close(fd)
144
+ Path(tmp_pdf).write_bytes(pdf_bytes)
145
+
146
+ logger.info(f"{tag} | PDF received ({len(pdf_bytes)} bytes), delegating to PDFHandler")
147
+
148
+ # Import here to avoid circular import
149
+ from ..handlers.pdf_handler import PDFHandler
150
+
151
+ pdf_handler = PDFHandler()
152
+ try:
153
+ result = await pdf_handler.handle(tmp_pdf, force_ocr=True)
154
+ if result:
155
+ logger.info(f"{tag} | PDFHandler succeeded ({len(result)} chars)")
156
+ return result
157
+ finally:
158
+ await pdf_handler.aclose()
159
+
160
+ except Exception as e:
161
+ logger.error(f"{tag} | error processing PDF: {e}")
162
+ return None
163
+ finally:
164
+ if tmp_pdf and Path(tmp_pdf).exists():
165
+ with contextlib.suppress(Exception):
166
+ os.unlink(tmp_pdf)
167
+
168
+ @_retry_transient(attempts=3, delays=(1.0, 3.0))
169
+ async def _convert_to_pdf(self, file_path: str, tag: str) -> bytes | None:
170
+ """Send file to Gotenberg and return PDF bytes."""
171
+ try:
172
+ async with httpx.AsyncClient(timeout=120) as client:
173
+ with open(file_path, "rb") as f:
174
+ response = await client.post(
175
+ self._convert_url,
176
+ files={"files": (Path(file_path).name, f)},
177
+ data={
178
+ "losslessImageCompression": "true",
179
+ "quality": "100",
180
+ },
181
+ )
182
+
183
+ if response.status_code == 200:
184
+ return response.content
185
+
186
+ logger.error(
187
+ f"{tag} | Gotenberg returned {response.status_code}: {response.text[:200]}"
188
+ )
189
+ return None
190
+
191
+ except httpx.ConnectError:
192
+ logger.info(f"{tag} | Gotenberg not reachable at {self.gotenberg_url}, skipping")
193
+ return None
194
+ except _RETRYABLE_EXC:
195
+ # Let the retry decorator handle transient errors. If retries are
196
+ # exhausted it re-raises, and the caller in ``convert()`` logs.
197
+ raise
198
+ except Exception as e:
199
+ logger.error(f"{tag} | Gotenberg error: {e}")
200
+ return None
@@ -11,6 +11,8 @@ Returns Markdown when successful, or `None` if conversion produced
11
11
  insufficient content (as determined by `ensure_minimum_content` in the service).
12
12
  """
13
13
 
14
+ from pathlib import Path
15
+
14
16
  from ..common import config
15
17
  from ..common.utils import is_pdf
16
18
  from ..services.openai_services import GPTVision
@@ -115,6 +117,7 @@ class GPTVisionConverter(Converter):
115
117
  max_retries=_retries,
116
118
  base_delay=base_delay,
117
119
  max_delay=max_delay,
120
+ context_label=Path(file_path).name,
118
121
  )
119
122
 
120
123
  async def aclose(self) -> None:
@@ -38,11 +38,22 @@ class OfficeHandler(BaseHandler):
38
38
  the pipeline order when initializing the handler.
39
39
 
40
40
  Supported formats: .docx, .pptx (modern Office XML formats).
41
- Old binary formats (.doc, .ppt, .odt, .rtf) supported via Gotenberg only.
41
+ Legacy binary formats (.doc, .ppt, .odt, .rtf) are supported at runtime
42
+ only when the configured pipeline contains an enabled ``GotenbergConverter``.
43
+ Callers should inspect the instance attribute ``supported_extensions`` (not
44
+ the class constant) to learn which formats this handler will actually
45
+ process.
42
46
  """
43
47
 
48
+ # Class constant stays narrow (the always-supported modern formats) so that
49
+ # ``BaseHandler.is_valid`` and any class-level introspection never falsely
50
+ # advertise legacy formats that would fail without Gotenberg. Dynamic
51
+ # runtime capability lives on the instance attribute ``supported_extensions``
52
+ # set in ``__init__``.
44
53
  SUPPORTED_EXTENSIONS = frozenset({".docx", ".pptx"})
45
54
 
55
+ _GOTENBERG_ONLY_EXTS = frozenset({".doc", ".ppt", ".odt", ".rtf"})
56
+
46
57
  def __init__(
47
58
  self,
48
59
  pipeline: list[tuple[Converter, str]] | None = None,
@@ -62,6 +73,26 @@ class OfficeHandler(BaseHandler):
62
73
  (self.markitdown, "MarkItDown"),
63
74
  ]
64
75
 
76
+ # isinstance catches subclasses/proxies users might pass in custom
77
+ # pipelines. We explicitly check that the imported ``GotenbergConverter``
78
+ # is still a real class before using it as an isinstance second arg —
79
+ # if tests patch the name to a MagicMock *instance* (not a class),
80
+ # we fall back to ``gotenberg_enabled = False`` instead of wrapping
81
+ # the whole generator in try/except TypeError, which would otherwise
82
+ # silently swallow a real bug inside an ``enabled`` property.
83
+ if isinstance(GotenbergConverter, type):
84
+ gotenberg_enabled = any(
85
+ isinstance(c, GotenbergConverter) and getattr(c, "enabled", False)
86
+ for c, _ in self._pipeline
87
+ )
88
+ else:
89
+ gotenberg_enabled = False
90
+ self.supported_extensions: frozenset[str] = (
91
+ self.SUPPORTED_EXTENSIONS | self._GOTENBERG_ONLY_EXTS
92
+ if gotenberg_enabled
93
+ else self.SUPPORTED_EXTENSIONS
94
+ )
95
+
65
96
  async def handle(self, file_path: str, *args, **kwargs) -> str | None:
66
97
  """
67
98
  Convert an Office document to Markdown by trying the configured pipeline in order.
@@ -158,6 +158,7 @@ class PagePDFConverter:
158
158
  png_path = await asyncio.to_thread(self._render_page_to_png, file_path, page_idx)
159
159
  content = await self.gpt_vision.gpt_vision.process_image(
160
160
  file_or_url=png_path,
161
+ context_label=f"{Path(file_path).name} page {page_idx + 1}",
161
162
  )
162
163
  if content:
163
164
  logger.info(f"{tag} | succeeded ({len(content)} chars)")
@@ -240,6 +240,7 @@ class GPTVision:
240
240
  base_delay: float = 0.5,
241
241
  max_delay: float = 20.0,
242
242
  retry_http_statuses: tuple[int, ...] = _RETRY_HTTP_STATUSES,
243
+ context_label: str | None = None,
243
244
  ):
244
245
  """
245
246
  Call self.lang_client.ainvoke with:
@@ -275,8 +276,9 @@ class GPTVision:
275
276
  raise last_err
276
277
  cap = min(max_delay, base_delay * (2 ** (attempt - 1)))
277
278
  sleep_for = random.uniform(0, cap)
279
+ label = context_label or file_or_url
278
280
  logger.warning(
279
- f"GPTVision: ainvoke transient error {file_or_url}: attempt {attempt}/{max_retries}: "
281
+ f"GPTVision: ainvoke transient error {label}: attempt {attempt}/{max_retries}: "
280
282
  f"{type(last_err).__name__}: {last_err}. Retrying in {sleep_for:.2f}s"
281
283
  )
282
284
  await asyncio.sleep(sleep_for)
@@ -331,6 +333,7 @@ class GPTVision:
331
333
  max_retries: int = 6,
332
334
  base_delay: float = 0.5,
333
335
  max_delay: float = 20.0,
336
+ context_label: str | None = None,
334
337
  ) -> str | None:
335
338
  """
336
339
  OCR a single image (local path or URL) with the vision model.
@@ -373,6 +376,7 @@ class GPTVision:
373
376
  max_retries=max_retries,
374
377
  base_delay=base_delay,
375
378
  max_delay=max_delay,
379
+ context_label=context_label,
376
380
  )
377
381
  finally:
378
382
  with suppress(Exception):
@@ -469,7 +473,11 @@ class GPTVision:
469
473
  # Per-page hard timeout so one slow page cannot stall the run
470
474
  try:
471
475
  partial_md = await asyncio.wait_for(
472
- self.process_image(png_path), timeout=self.page_timeout_s
476
+ self.process_image(
477
+ png_path,
478
+ context_label=f"{file_stem} page {page_index + 1}/{num_pages}",
479
+ ),
480
+ timeout=self.page_timeout_s,
473
481
  )
474
482
  except TimeoutError:
475
483
  logger.error(
@@ -74,6 +74,62 @@ gotenberg_converter:
74
74
  expect: ["what is lorem ipsum?", "where does it come from"]
75
75
 
76
76
 
77
+ # =============================================================================
78
+ # LEGACY OFFICE FORMATS — Gotenberg-gated (.doc / .ppt / .odt / .rtf)
79
+ # Same fixtures as above re-saved / pandoc-converted into legacy formats.
80
+ # Routed through OfficeHandler → Gotenberg → PDF → PDFHandler per-page OCR.
81
+ # Only available iff GOTENBERG_URL is configured on the pipeline.
82
+ # =============================================================================
83
+ legacy_office_formats:
84
+ # .doc (re-saved from .docx via Word)
85
+ - file: tests/data/word-legacy/lorem-ipsum-only-text.doc
86
+ expect: ["what is lorem ipsum?", "where does it come from"]
87
+ - file: tests/data/word-legacy/lorem-ipsum-text-with-image.doc
88
+ expect: ["what is lorem ipsum?", "red", "green", "blue"]
89
+ - file: tests/data/word-legacy/geometry-colors.doc
90
+ expect: ["red", "green", "blue"]
91
+ - file: tests/data/word-legacy/lorem-ipsum-scanned-text.doc
92
+ expect: ["what is lorem ipsum?", "why do we use it?"]
93
+ - file: tests/data/word-legacy/lorem-ipsum-text-with-scanned-text.doc
94
+ expect: ["what is lorem ipsum?", "where does it come from"]
95
+
96
+ # .ppt (re-saved from .pptx via PowerPoint)
97
+ - file: tests/data/powerpoint-legacy/lorem-ipsum-only-text.ppt
98
+ expect: ["what is lorem ipsum?", "where does it come from"]
99
+ - file: tests/data/powerpoint-legacy/lorem-ipsum-text-with-image.ppt
100
+ expect: ["what is lorem ipsum", "red", "green", "blue"]
101
+ - file: tests/data/powerpoint-legacy/geometry-colors.ppt
102
+ expect: ["red", "green", "blue"]
103
+ - file: tests/data/powerpoint-legacy/lorem-ipsum-scanned-text.ppt
104
+ expect: ["what is lorem ipsum", "why do we use it"]
105
+ - file: tests/data/powerpoint-legacy/lorem-ipsum-text-with-scanned-text.ppt
106
+ expect: ["what is lorem ipsum?", "where does it come from"]
107
+
108
+ # .odt (re-saved from .docx via Word)
109
+ - file: tests/data/word-legacy/lorem-ipsum-only-text.odt
110
+ expect: ["what is lorem ipsum?", "where does it come from"]
111
+ - file: tests/data/word-legacy/lorem-ipsum-text-with-image.odt
112
+ expect: ["what is lorem ipsum?", "red", "green", "blue"]
113
+ - file: tests/data/word-legacy/geometry-colors.odt
114
+ expect: ["red", "green", "blue"]
115
+ - file: tests/data/word-legacy/lorem-ipsum-scanned-text.odt
116
+ expect: ["what is lorem ipsum?", "why do we use it?"]
117
+ - file: tests/data/word-legacy/lorem-ipsum-text-with-scanned-text.odt
118
+ expect: ["what is lorem ipsum?", "where does it come from"]
119
+
120
+ # .rtf (re-saved from .docx via Word)
121
+ - file: tests/data/word-legacy/lorem-ipsum-only-text.rtf
122
+ expect: ["what is lorem ipsum?", "where does it come from"]
123
+ - file: tests/data/word-legacy/lorem-ipsum-text-with-image.rtf
124
+ expect: ["what is lorem ipsum?", "red", "green", "blue"]
125
+ - file: tests/data/word-legacy/geometry-colors.rtf
126
+ expect: ["red", "green", "blue"]
127
+ - file: tests/data/word-legacy/lorem-ipsum-scanned-text.rtf
128
+ expect: ["what is lorem ipsum?", "why do we use it?"]
129
+ - file: tests/data/word-legacy/lorem-ipsum-text-with-scanned-text.rtf
130
+ expect: ["what is lorem ipsum?", "where does it come from"]
131
+
132
+
77
133
  # =============================================================================
78
134
  # MARKITDOWN CONVERTER — text extraction only, NO OCR, NO image content
79
135
  # =============================================================================