docvortex 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (364) hide show
  1. docvortex/__init__.py +21 -0
  2. docvortex/analyzers/__init__.py +3 -0
  3. docvortex/analyzers/native/__init__.py +37 -0
  4. docvortex/analyzers/native/_shared/__init__.py +3 -0
  5. docvortex/analyzers/native/_shared/hyperlink.py +16 -0
  6. docvortex/analyzers/native/_shared/image.py +8 -0
  7. docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
  8. docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
  9. docvortex/analyzers/native/_shared/markup/formula.py +37 -0
  10. docvortex/analyzers/native/_shared/markup/projector.py +57 -0
  11. docvortex/analyzers/native/_shared/markup/styles.py +19 -0
  12. docvortex/analyzers/native/_shared/mathml.py +17 -0
  13. docvortex/analyzers/native/_shared/names.py +7 -0
  14. docvortex/analyzers/native/_shared/xycut.py +414 -0
  15. docvortex/analyzers/native/contracts.py +44 -0
  16. docvortex/analyzers/native/csv.py +351 -0
  17. docvortex/analyzers/native/epub/__init__.py +18 -0
  18. docvortex/analyzers/native/epub/constants.py +45 -0
  19. docvortex/analyzers/native/epub/converter.py +80 -0
  20. docvortex/analyzers/native/epub/errors.py +20 -0
  21. docvortex/analyzers/native/epub/metadata.py +31 -0
  22. docvortex/analyzers/native/epub/package.py +649 -0
  23. docvortex/analyzers/native/epub/xhtml.py +306 -0
  24. docvortex/analyzers/native/html/__init__.py +6 -0
  25. docvortex/analyzers/native/html/anchors.py +271 -0
  26. docvortex/analyzers/native/html/constants.py +25 -0
  27. docvortex/analyzers/native/html/contracts.py +7 -0
  28. docvortex/analyzers/native/html/converter.py +117 -0
  29. docvortex/analyzers/native/html/document.py +389 -0
  30. docvortex/analyzers/native/html/errors.py +12 -0
  31. docvortex/analyzers/native/html/resources.py +365 -0
  32. docvortex/analyzers/native/html/selector.py +421 -0
  33. docvortex/analyzers/native/models.py +209 -0
  34. docvortex/analyzers/native/ofd/__init__.py +14 -0
  35. docvortex/analyzers/native/ofd/constants.py +59 -0
  36. docvortex/analyzers/native/ofd/converter.py +31 -0
  37. docvortex/analyzers/native/ofd/errors.py +16 -0
  38. docvortex/analyzers/native/ofd/geometry.py +215 -0
  39. docvortex/analyzers/native/ofd/images.py +124 -0
  40. docvortex/analyzers/native/ofd/metadata.py +53 -0
  41. docvortex/analyzers/native/ofd/models.py +170 -0
  42. docvortex/analyzers/native/ofd/package.py +346 -0
  43. docvortex/analyzers/native/ofd/path.py +201 -0
  44. docvortex/analyzers/native/ofd/reading_order.py +293 -0
  45. docvortex/analyzers/native/ofd/resources.py +122 -0
  46. docvortex/analyzers/native/ofd/scene.py +347 -0
  47. docvortex/analyzers/native/ofd/table.py +240 -0
  48. docvortex/analyzers/native/ofd/text.py +521 -0
  49. docvortex/analyzers/native/office/__init__.py +3 -0
  50. docvortex/analyzers/native/office/doc/__init__.py +3 -0
  51. docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
  52. docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
  53. docvortex/analyzers/native/office/doc/fib.py +261 -0
  54. docvortex/analyzers/native/office/doc/fields.py +101 -0
  55. docvortex/analyzers/native/office/doc/formatting.py +171 -0
  56. docvortex/analyzers/native/office/doc/images.py +162 -0
  57. docvortex/analyzers/native/office/doc/lists.py +356 -0
  58. docvortex/analyzers/native/office/doc/models.py +164 -0
  59. docvortex/analyzers/native/office/doc/parser.py +851 -0
  60. docvortex/analyzers/native/office/doc/pieces.py +256 -0
  61. docvortex/analyzers/native/office/doc/records.py +40 -0
  62. docvortex/analyzers/native/office/doc/sprm.py +269 -0
  63. docvortex/analyzers/native/office/doc/styles.py +215 -0
  64. docvortex/analyzers/native/office/docx/__init__.py +3 -0
  65. docvortex/analyzers/native/office/docx/context.py +71 -0
  66. docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
  67. docvortex/analyzers/native/office/docx/equationxml.py +119 -0
  68. docvortex/analyzers/native/office/docx/fields.py +816 -0
  69. docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
  70. docvortex/analyzers/native/office/docx/main.py +50 -0
  71. docvortex/analyzers/native/office/docx/numbering.py +491 -0
  72. docvortex/analyzers/native/office/docx/office_xml.py +57 -0
  73. docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
  74. docvortex/analyzers/native/office/docx/resources.py +526 -0
  75. docvortex/analyzers/native/office/docx/styles.py +693 -0
  76. docvortex/analyzers/native/office/docx/tables.py +515 -0
  77. docvortex/analyzers/native/office/equation/__init__.py +3 -0
  78. docvortex/analyzers/native/office/equation/image.py +470 -0
  79. docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
  80. docvortex/analyzers/native/office/equation/mtef.py +885 -0
  81. docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
  82. docvortex/analyzers/native/office/equation/omml.py +561 -0
  83. docvortex/analyzers/native/office/equation/ooxml.py +62 -0
  84. docvortex/analyzers/native/office/errors.py +33 -0
  85. docvortex/analyzers/native/office/image.py +307 -0
  86. docvortex/analyzers/native/office/legacy/__init__.py +3 -0
  87. docvortex/analyzers/native/office/legacy/binary.py +43 -0
  88. docvortex/analyzers/native/office/legacy/officeart.py +362 -0
  89. docvortex/analyzers/native/office/legacy/ole.py +110 -0
  90. docvortex/analyzers/native/office/limits.py +12 -0
  91. docvortex/analyzers/native/office/odf/__init__.py +3 -0
  92. docvortex/analyzers/native/office/odf/chart.py +76 -0
  93. docvortex/analyzers/native/office/odf/constants.py +70 -0
  94. docvortex/analyzers/native/office/odf/converters.py +403 -0
  95. docvortex/analyzers/native/office/odf/errors.py +18 -0
  96. docvortex/analyzers/native/office/odf/metadata.py +91 -0
  97. docvortex/analyzers/native/office/odf/models.py +176 -0
  98. docvortex/analyzers/native/office/odf/package.py +270 -0
  99. docvortex/analyzers/native/office/odf/styles.py +329 -0
  100. docvortex/analyzers/native/office/odf/table.py +469 -0
  101. docvortex/analyzers/native/office/odf/text.py +1002 -0
  102. docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
  103. docvortex/analyzers/native/office/opc.py +38 -0
  104. docvortex/analyzers/native/office/ppt/__init__.py +3 -0
  105. docvortex/analyzers/native/office/ppt/models.py +118 -0
  106. docvortex/analyzers/native/office/ppt/parser.py +1895 -0
  107. docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
  108. docvortex/analyzers/native/office/ppt/records.py +131 -0
  109. docvortex/analyzers/native/office/ppt/style_text.py +247 -0
  110. docvortex/analyzers/native/office/pptx/__init__.py +3 -0
  111. docvortex/analyzers/native/office/pptx/context.py +113 -0
  112. docvortex/analyzers/native/office/pptx/lists.py +558 -0
  113. docvortex/analyzers/native/office/pptx/main.py +20 -0
  114. docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
  115. docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
  116. docvortex/analyzers/native/office/pptx/resources.py +329 -0
  117. docvortex/analyzers/native/office/pptx/shapes.py +393 -0
  118. docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
  119. docvortex/analyzers/native/office/pptx/titles.py +178 -0
  120. docvortex/analyzers/native/office/rich_text.py +420 -0
  121. docvortex/analyzers/native/office/rtf/__init__.py +3 -0
  122. docvortex/analyzers/native/office/rtf/converter.py +708 -0
  123. docvortex/analyzers/native/office/rtf/lexer.py +213 -0
  124. docvortex/analyzers/native/office/rtf/math.py +339 -0
  125. docvortex/analyzers/native/office/rtf/models.py +185 -0
  126. docvortex/analyzers/native/office/rtf/parser.py +1553 -0
  127. docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
  128. docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
  129. docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
  130. docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
  131. docvortex/analyzers/native/office/streams.py +18 -0
  132. docvortex/analyzers/native/office/xls/__init__.py +3 -0
  133. docvortex/analyzers/native/office/xls/chart.py +132 -0
  134. docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
  135. docvortex/analyzers/native/office/xls/models.py +109 -0
  136. docvortex/analyzers/native/office/xls/number_format.py +521 -0
  137. docvortex/analyzers/native/office/xls/parser.py +1145 -0
  138. docvortex/analyzers/native/office/xls/records.py +201 -0
  139. docvortex/analyzers/native/office/xls/strings.py +205 -0
  140. docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
  141. docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
  142. docvortex/analyzers/native/office/xlsx/main.py +20 -0
  143. docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
  144. docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
  145. docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
  146. docvortex/analyzers/native/pdf/__init__.py +3 -0
  147. docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
  148. docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
  149. docvortex/analyzers/native/pdf/code_blocks.py +535 -0
  150. docvortex/analyzers/native/pdf/formulas.py +1985 -0
  151. docvortex/analyzers/native/pdf/geometry.py +281 -0
  152. docvortex/analyzers/native/pdf/graphics.py +1501 -0
  153. docvortex/analyzers/native/pdf/index_blocks.py +268 -0
  154. docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
  155. docvortex/analyzers/native/pdf/inline/common.py +112 -0
  156. docvortex/analyzers/native/pdf/inline/detection.py +590 -0
  157. docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
  158. docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
  159. docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
  160. docvortex/analyzers/native/pdf/inline/types.py +385 -0
  161. docvortex/analyzers/native/pdf/line_layout.py +1106 -0
  162. docvortex/analyzers/native/pdf/line_merging.py +1223 -0
  163. docvortex/analyzers/native/pdf/models.py +246 -0
  164. docvortex/analyzers/native/pdf/native_text.py +1004 -0
  165. docvortex/analyzers/native/pdf/pipeline.py +1390 -0
  166. docvortex/analyzers/native/pdf/script_geometry.py +636 -0
  167. docvortex/analyzers/native/pdf/shared.py +30 -0
  168. docvortex/analyzers/native/pdf/spatial_text.py +383 -0
  169. docvortex/analyzers/native/pdf/table_annotations.py +446 -0
  170. docvortex/analyzers/native/pdf/table_constants.py +48 -0
  171. docvortex/analyzers/native/pdf/table_detection.py +227 -0
  172. docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
  173. docvortex/analyzers/native/pdf/table_geometry.py +40 -0
  174. docvortex/analyzers/native/pdf/table_materialization.py +444 -0
  175. docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
  176. docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
  177. docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
  178. docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
  179. docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
  180. docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
  181. docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
  182. docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
  183. docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
  184. docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
  185. docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
  186. docvortex/analyzers/native/pdf/table_rows.py +34 -0
  187. docvortex/analyzers/native/pdf/table_rules.py +1129 -0
  188. docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
  189. docvortex/analyzers/native/pdf/tables.py +147 -0
  190. docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
  191. docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
  192. docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
  193. docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
  194. docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
  195. docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
  196. docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
  197. docvortex/analyzers/native/pdf/text_blocks.py +82 -0
  198. docvortex/analyzers/native/pdf/text_styles.py +55 -0
  199. docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
  200. docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
  201. docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
  202. docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
  203. docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
  204. docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
  205. docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
  206. docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
  207. docvortex/analyzers/native/pdf/titles.py +75 -0
  208. docvortex/analyzers/native/pdf/typography.py +19 -0
  209. docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
  210. docvortex/api.py +180 -0
  211. docvortex/assets/__init__.py +5 -0
  212. docvortex/assets/store.py +51 -0
  213. docvortex/cli.py +54 -0
  214. docvortex/codecs/__init__.py +3 -0
  215. docvortex/codecs/html/__init__.py +22 -0
  216. docvortex/codecs/html/contracts.py +236 -0
  217. docvortex/codecs/html/materializer.py +331 -0
  218. docvortex/codecs/html/parser.py +763 -0
  219. docvortex/codecs/html/resources.py +34 -0
  220. docvortex/codecs/json.py +17 -0
  221. docvortex/content/__init__.py +5 -0
  222. docvortex/content/inline.py +248 -0
  223. docvortex/content/markup/__init__.py +44 -0
  224. docvortex/content/markup/anchors.py +188 -0
  225. docvortex/content/markup/formula.py +280 -0
  226. docvortex/content/markup/projector.py +1237 -0
  227. docvortex/content/markup/styles.py +327 -0
  228. docvortex/content/mathml.py +167 -0
  229. docvortex/content/normalization.py +188 -0
  230. docvortex/content/spans.py +183 -0
  231. docvortex/content/table/__init__.py +18 -0
  232. docvortex/content/table/blocks.py +152 -0
  233. docvortex/content/table/content.py +425 -0
  234. docvortex/content/table/document.py +104 -0
  235. docvortex/content/table/html.py +399 -0
  236. docvortex/content/table/models.py +76 -0
  237. docvortex/content/table/rules.py +42 -0
  238. docvortex/content/table/structure.py +221 -0
  239. docvortex/content/tree.py +5 -0
  240. docvortex/document/__init__.py +3 -0
  241. docvortex/document/contracts.py +22 -0
  242. docvortex/document/detection.py +389 -0
  243. docvortex/document/filetypes.py +175 -0
  244. docvortex/document/page_range.py +167 -0
  245. docvortex/document/pdf/__init__.py +19 -0
  246. docvortex/document/pdf/classify.py +1138 -0
  247. docvortex/document/pdf/constants.py +132 -0
  248. docvortex/document/pdf/diagnostics.py +350 -0
  249. docvortex/document/pdf/document.py +582 -0
  250. docvortex/document/pdf/font_runtime.py +335 -0
  251. docvortex/document/pdf/geometry.py +31 -0
  252. docvortex/document/pdf/images.py +654 -0
  253. docvortex/document/pdf/native_annotations.py +367 -0
  254. docvortex/document/pdf/native_contracts.py +169 -0
  255. docvortex/document/pdf/native_coordinates.py +216 -0
  256. docvortex/document/pdf/native_lifecycle.py +16 -0
  257. docvortex/document/pdf/native_objects.py +902 -0
  258. docvortex/document/pdf/native_text_geometry.py +315 -0
  259. docvortex/document/pdf/pdfium.py +325 -0
  260. docvortex/document/pdf/raster.py +46 -0
  261. docvortex/document/pdf/text/__init__.py +62 -0
  262. docvortex/document/pdf/text/contracts.py +211 -0
  263. docvortex/document/pdf/text/extract.py +165 -0
  264. docvortex/document/pdf/text/geometry.py +16 -0
  265. docvortex/document/pdf/text/groups.py +162 -0
  266. docvortex/document/pdf/visual_geometry.py +201 -0
  267. docvortex/document/pdf/visuals.py +343 -0
  268. docvortex/document/source.py +92 -0
  269. docvortex/errors.py +21 -0
  270. docvortex/export/__init__.py +3 -0
  271. docvortex/export/bundle.py +98 -0
  272. docvortex/export/files.py +66 -0
  273. docvortex/export/middle.py +208 -0
  274. docvortex/foundation/__init__.py +3 -0
  275. docvortex/foundation/geometry.py +125 -0
  276. docvortex/foundation/hyperlink.py +65 -0
  277. docvortex/foundation/image.py +48 -0
  278. docvortex/foundation/image_encoding.py +30 -0
  279. docvortex/foundation/image_payload.py +280 -0
  280. docvortex/foundation/language.py +92 -0
  281. docvortex/foundation/platform.py +38 -0
  282. docvortex/foundation/text.py +153 -0
  283. docvortex/foundation/type_identity.py +20 -0
  284. docvortex/foundation/xml_names.py +20 -0
  285. docvortex/options.py +30 -0
  286. docvortex/postprocess/__init__.py +3 -0
  287. docvortex/postprocess/content.py +53 -0
  288. docvortex/postprocess/document.py +19 -0
  289. docvortex/postprocess/lists.py +236 -0
  290. docvortex/postprocess/page_blocks.py +214 -0
  291. docvortex/postprocess/pages.py +95 -0
  292. docvortex/postprocess/paragraphs.py +580 -0
  293. docvortex/postprocess/visual.py +715 -0
  294. docvortex/render/__init__.py +48 -0
  295. docvortex/render/_internal/__init__.py +3 -0
  296. docvortex/render/_internal/common/__init__.py +3 -0
  297. docvortex/render/_internal/common/context.py +43 -0
  298. docvortex/render/_internal/common/html_table.py +178 -0
  299. docvortex/render/_internal/common/index.py +33 -0
  300. docvortex/render/_internal/common/list_items.py +158 -0
  301. docvortex/render/_internal/common/planner.py +140 -0
  302. docvortex/render/_internal/docx/__init__.py +3 -0
  303. docvortex/render/_internal/docx/assets.py +202 -0
  304. docvortex/render/_internal/docx/inline.py +434 -0
  305. docvortex/render/_internal/docx/math.py +220 -0
  306. docvortex/render/_internal/docx/renderer.py +905 -0
  307. docvortex/render/_internal/docx/styles.py +195 -0
  308. docvortex/render/_internal/docx/table.py +442 -0
  309. docvortex/render/_internal/epub/__init__.py +5 -0
  310. docvortex/render/_internal/epub/assets.py +173 -0
  311. docvortex/render/_internal/epub/package.py +249 -0
  312. docvortex/render/_internal/epub/renderer.py +1156 -0
  313. docvortex/render/_internal/html/__init__.py +3 -0
  314. docvortex/render/_internal/html/inline.py +346 -0
  315. docvortex/render/_internal/html/renderer.py +1041 -0
  316. docvortex/render/_internal/html/sanitizer.py +478 -0
  317. docvortex/render/_internal/html/table.py +121 -0
  318. docvortex/render/_internal/latex/__init__.py +1 -0
  319. docvortex/render/_internal/latex/assets.py +85 -0
  320. docvortex/render/_internal/latex/inline.py +145 -0
  321. docvortex/render/_internal/latex/renderer.py +506 -0
  322. docvortex/render/_internal/latex/table.py +347 -0
  323. docvortex/render/_internal/markdown/__init__.py +3 -0
  324. docvortex/render/_internal/markdown/assets.py +78 -0
  325. docvortex/render/_internal/markdown/blocks.py +635 -0
  326. docvortex/render/_internal/markdown/escaping.py +51 -0
  327. docvortex/render/_internal/markdown/inline.py +260 -0
  328. docvortex/render/_internal/markdown/renderer.py +93 -0
  329. docvortex/render/_internal/markdown/table.py +281 -0
  330. docvortex/render/_internal/pdf/__init__.py +3 -0
  331. docvortex/render/_internal/pdf/assets.py +197 -0
  332. docvortex/render/_internal/pdf/formula.py +417 -0
  333. docvortex/render/_internal/pdf/inline.py +343 -0
  334. docvortex/render/_internal/pdf/renderer.py +734 -0
  335. docvortex/render/_internal/pdf/styles.py +206 -0
  336. docvortex/render/_internal/pdf/table.py +272 -0
  337. docvortex/render/_internal/structured_content/__init__.py +3 -0
  338. docvortex/render/_internal/structured_content/renderer.py +193 -0
  339. docvortex/render/api.py +199 -0
  340. docvortex/render/contracts.py +205 -0
  341. docvortex/render/docx.py +38 -0
  342. docvortex/render/epub.py +35 -0
  343. docvortex/render/fragments.py +70 -0
  344. docvortex/render/html.py +29 -0
  345. docvortex/render/latex.py +24 -0
  346. docvortex/render/markdown.py +50 -0
  347. docvortex/render/pdf.py +25 -0
  348. docvortex/render/structured_content.py +23 -0
  349. docvortex/resources/epub/docvortex.css +91 -0
  350. docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
  351. docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
  352. docvortex/resources/fonts/NOTICE +190 -0
  353. docvortex/resources/fonts/manifest.json +9 -0
  354. docvortex/resources/html/docvortex.css +601 -0
  355. docvortex/resources/html/docvortex.min.css +1 -0
  356. docvortex/result.py +91 -0
  357. docvortex/schema.py +1141 -0
  358. docvortex/version.py +3 -0
  359. docvortex-0.2.1.dist-info/METADATA +193 -0
  360. docvortex-0.2.1.dist-info/RECORD +364 -0
  361. docvortex-0.2.1.dist-info/WHEEL +5 -0
  362. docvortex-0.2.1.dist-info/entry_points.txt +2 -0
  363. docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
  364. docvortex-0.2.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,1274 @@
1
+ """合并正文、公式上下文和列表引导块的空间组件。"""
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+ import statistics
7
+ from typing import Any
8
+
9
+ from .....schema import BBox
10
+ from ..geometry import _bbox_axis_overlap_ratio, _bbox_center_y, _bbox_union_many, _rotate_bbox_from_upright
11
+ from .common import (
12
+ _FIGURE_CAPTION_MARKER_RE,
13
+ _INLINE_MATH_RECOVERY_MARKER,
14
+ _LABELLED_METADATA_RE,
15
+ _LIST_ITEM_RE,
16
+ _PARAGRAPH_FORMULA_CONTEXT_MARKER,
17
+ _SHORT_SAME_BASELINE_PREFIX_RE,
18
+ _URL_LINE_RE,
19
+ _block_starts_with_short_wide_rows,
20
+ _compatible_component_lane_width,
21
+ _component_connection_skips_block,
22
+ _component_declared_lane_interval,
23
+ _component_lane_interval,
24
+ _components_share_lane_role,
25
+ _find_short_opener_pairs,
26
+ _has_parallel_text_component,
27
+ _merge_internal_text_block_group,
28
+ _merge_text_line_content,
29
+ _nearest_following_text_component,
30
+ _nearest_tapered_tail_component,
31
+ _text_component_sort_key,
32
+ )
33
+
34
+
35
+ def _merge_short_same_baseline_prefix_blocks(
36
+ blocks: list[dict[str, Any]],
37
+ page_size: tuple[float, float],
38
+ ) -> list[dict[str, Any]]:
39
+ """合并括号序号或时刻等短前缀与右侧同基线正文。"""
40
+
41
+ replacements: dict[int, dict[str, Any]] = {}
42
+ consumed: set[int] = set()
43
+ for prefix_index, prefix in enumerate(blocks):
44
+ prefix_rows = prefix.get("_local_line_bboxes")
45
+ prefix_content = str(prefix.get("content") or "").strip()
46
+ if (
47
+ prefix_index in consumed
48
+ or prefix.get("type") != "text"
49
+ or not isinstance(prefix_rows, list)
50
+ or len(prefix_rows) != 1
51
+ or _SHORT_SAME_BASELINE_PREFIX_RE.match(prefix_content) is None
52
+ ):
53
+ continue
54
+ prefix_bbox = prefix_rows[0]
55
+ prefix_heights = [
56
+ float(height)
57
+ for height in prefix.get("_line_heights", [])
58
+ if isinstance(height, (int, float)) and float(height) > 0
59
+ ]
60
+ prefix_height = statistics.median(prefix_heights) if prefix_heights else max(0.1, prefix_bbox[3] - prefix_bbox[1])
61
+ angle = int(prefix.get("angle", 0) or 0) % 360
62
+ local_page_width = page_size[1] if angle in {90, 270} else page_size[0]
63
+ matches: list[tuple[float, int]] = []
64
+ for host_index, host in enumerate(blocks):
65
+ host_rows = host.get("_local_line_bboxes")
66
+ if (
67
+ host_index == prefix_index
68
+ or host_index in consumed
69
+ or host.get("type") != "text"
70
+ or int(host.get("angle", 0) or 0) % 360 != angle
71
+ or not isinstance(host_rows, list)
72
+ or not host_rows
73
+ ):
74
+ continue
75
+ host_bbox = host_rows[0]
76
+ host_width = host_bbox[2] - host_bbox[0]
77
+ horizontal_gap = host_bbox[0] - prefix_bbox[2]
78
+ if (
79
+ host_bbox[0] < prefix_bbox[2]
80
+ or horizontal_gap > 1.25 * prefix_height
81
+ or host_width < 0.15 * local_page_width
82
+ or _bbox_axis_overlap_ratio(
83
+ prefix_bbox,
84
+ host_bbox,
85
+ axis="y",
86
+ )
87
+ < 0.5
88
+ ):
89
+ continue
90
+ matches.append((horizontal_gap, host_index))
91
+ if not matches:
92
+ continue
93
+ _gap, host_index = min(matches)
94
+ replacement_index = min(prefix_index, host_index)
95
+ replacement = _merge_internal_text_block_group(
96
+ blocks,
97
+ [prefix_index, host_index],
98
+ preserve_visual_spaces=True,
99
+ )
100
+ replacement["type"] = "text"
101
+ replacements[replacement_index] = replacement
102
+ consumed.update({prefix_index, host_index})
103
+ return [
104
+ replacements.get(index, block) for index, block in enumerate(blocks) if index not in consumed or index in replacements
105
+ ]
106
+
107
+
108
+ def _blocks_share_boundary_visual_row(
109
+ first: dict[str, Any],
110
+ second: dict[str, Any],
111
+ pair_height: float,
112
+ ) -> bool:
113
+ """检查前块末行与后块首行是否为被错误切开的同一视觉行。"""
114
+
115
+ first_rows = first.get("_local_line_bboxes")
116
+ second_rows = second.get("_local_line_bboxes")
117
+ if (
118
+ not isinstance(first_rows, list)
119
+ or not isinstance(second_rows, list)
120
+ or max(len(first_rows), len(second_rows)) < 3
121
+ or len(first_rows) + len(second_rows) > 6
122
+ or not _components_share_lane_role(first, second, pair_height)
123
+ ):
124
+ return False
125
+ if first["bbox"][1] <= second["bbox"][1]:
126
+ upper_rows, lower_rows = first_rows, second_rows
127
+ else:
128
+ upper_rows, lower_rows = second_rows, first_rows
129
+ upper_boundary = max(upper_rows, key=lambda bbox: (_bbox_center_y(bbox), bbox[0]))
130
+ lower_boundary = min(lower_rows, key=lambda bbox: (_bbox_center_y(bbox), bbox[0]))
131
+ vertical_overlap = max(
132
+ 0.0,
133
+ min(upper_boundary[3], lower_boundary[3]) - max(upper_boundary[1], lower_boundary[1]),
134
+ )
135
+ shorter_height = max(
136
+ 0.1,
137
+ min(
138
+ upper_boundary[3] - upper_boundary[1],
139
+ lower_boundary[3] - lower_boundary[1],
140
+ ),
141
+ )
142
+ horizontal_gap = lower_boundary[0] - upper_boundary[2]
143
+ union_bbox = _bbox_union_many([first["bbox"], second["bbox"]])
144
+ return (
145
+ vertical_overlap / shorter_height >= 0.7
146
+ and -0.2 * pair_height <= horizontal_gap <= 0.75 * pair_height
147
+ and union_bbox[3] - union_bbox[1] <= 6.0 * pair_height
148
+ )
149
+
150
+
151
+ def _merge_overlapping_same_line_text_blocks(
152
+ blocks: list[dict[str, Any]],
153
+ page_size: tuple[float, float],
154
+ ) -> list[dict[str, Any]]:
155
+ """合并块体或边界视觉行重叠的宽正文块,修复错误分栏。"""
156
+
157
+ consumed: set[int] = set()
158
+ replacements: dict[int, dict[str, Any]] = {}
159
+ for first_index, first in enumerate(blocks):
160
+ first_bbox = first.get("bbox")
161
+ first_rows = first.get("_local_line_bboxes")
162
+ angle = int(first.get("angle", 0) or 0) % 360
163
+ local_page_width = page_size[1] if angle in {90, 270} else page_size[0]
164
+ if (
165
+ first_index in consumed
166
+ or first.get("type") != "text"
167
+ or not isinstance(first_bbox, (list, tuple))
168
+ or not isinstance(first_rows, list)
169
+ or not 1 <= len(first_rows) <= 5
170
+ or first_bbox[2] - first_bbox[0] < 0.3 * local_page_width
171
+ ):
172
+ continue
173
+ first_heights = [
174
+ float(height) for height in first.get("_line_heights", []) if isinstance(height, (int, float)) and height > 0
175
+ ]
176
+ first_height = statistics.median(first_heights) if first_heights else first_bbox[3] - first_bbox[1]
177
+ for second_index in range(first_index + 1, len(blocks)):
178
+ second = blocks[second_index]
179
+ second_bbox = second.get("bbox")
180
+ second_rows = second.get("_local_line_bboxes")
181
+ if (
182
+ second_index in consumed
183
+ or second.get("type") != "text"
184
+ or int(second.get("angle", 0) or 0) % 360 != angle
185
+ or not isinstance(second_bbox, (list, tuple))
186
+ or not isinstance(second_rows, list)
187
+ or not 1 <= len(second_rows) <= 5
188
+ or second_bbox[2] - second_bbox[0] < 0.3 * local_page_width
189
+ ):
190
+ continue
191
+ later_block = max(
192
+ (first, second),
193
+ key=_text_component_sort_key,
194
+ )
195
+ if later_block.get("_hard_break_before") is True:
196
+ continue
197
+ second_heights = [
198
+ float(height) for height in second.get("_line_heights", []) if isinstance(height, (int, float)) and height > 0
199
+ ]
200
+ second_height = statistics.median(second_heights) if second_heights else second_bbox[3] - second_bbox[1]
201
+ pair_height = max(first_height, second_height, 0.1)
202
+ vertical_overlap = max(
203
+ 0.0,
204
+ min(first_bbox[3], second_bbox[3]) - max(first_bbox[1], second_bbox[1]),
205
+ )
206
+ minimum_box_height = max(
207
+ 0.1,
208
+ min(
209
+ first_bbox[3] - first_bbox[1],
210
+ second_bbox[3] - second_bbox[1],
211
+ ),
212
+ )
213
+ first_fonts = first.get("_font_signatures")
214
+ second_fonts = second.get("_font_signatures")
215
+ fonts_conflict = (
216
+ isinstance(first_fonts, set)
217
+ and isinstance(second_fonts, set)
218
+ and first_fonts
219
+ and second_fonts
220
+ and first_fonts.isdisjoint(second_fonts)
221
+ )
222
+ compact_overlap = (
223
+ len(first_rows) <= 2
224
+ and len(second_rows) <= 2
225
+ and min(len(first_rows), len(second_rows)) == 1
226
+ and abs(first_bbox[0] - second_bbox[0]) <= pair_height
227
+ and vertical_overlap / minimum_box_height >= 0.5
228
+ and _bbox_axis_overlap_ratio(first_bbox, second_bbox, axis="x") >= 0.75
229
+ )
230
+ boundary_overlap = _blocks_share_boundary_visual_row(
231
+ first,
232
+ second,
233
+ pair_height,
234
+ )
235
+ if fonts_conflict or not (compact_overlap or boundary_overlap):
236
+ continue
237
+ union_bbox = _bbox_union_many([first_bbox, second_bbox])
238
+ if not boundary_overlap and union_bbox[3] - union_bbox[1] > 4.0 * pair_height:
239
+ continue
240
+ replacement_index = min(first_index, second_index)
241
+ replacements[replacement_index] = _merge_internal_text_block_group(
242
+ blocks,
243
+ [first_index, second_index],
244
+ )
245
+ consumed.update({first_index, second_index})
246
+ break
247
+ return [
248
+ replacements.get(index, block) for index, block in enumerate(blocks) if index not in consumed or index in replacements
249
+ ]
250
+
251
+
252
+ def _merge_inline_math_fragment_text_blocks(
253
+ blocks: list[dict[str, Any]],
254
+ page_size: tuple[float, float],
255
+ ) -> list[dict[str, Any]]:
256
+ """把同一宽正文行上下叠放的多个小数学碎片收回一个文本块。"""
257
+
258
+ consumed: set[int] = set()
259
+ replacements: dict[int, dict[str, Any]] = {}
260
+ for host_index, host in enumerate(blocks):
261
+ host_bbox = host.get("bbox")
262
+ host_heights = host.get("_line_heights")
263
+ host_rows = host.get("_local_line_bboxes")
264
+ angle = int(host.get("angle", 0) or 0) % 360
265
+ local_page_width = page_size[1] if angle in {90, 270} else page_size[0]
266
+ if (
267
+ host_index in consumed
268
+ or host.get("type") != "text"
269
+ or (host.get("_single_run_row_id") is None and (not isinstance(host_rows, list) or len(host_rows) > 2))
270
+ or not isinstance(host_bbox, (list, tuple))
271
+ or host_bbox[2] - host_bbox[0] < 0.35 * local_page_width
272
+ or not isinstance(host_heights, list)
273
+ or not host_heights
274
+ ):
275
+ continue
276
+ host_height = statistics.median(
277
+ float(height) for height in host_heights if isinstance(height, (int, float)) and height > 0
278
+ )
279
+ fragment_indices: list[int] = []
280
+ for candidate_index, candidate in enumerate(blocks):
281
+ candidate_bbox = candidate.get("bbox")
282
+ candidate_rows = candidate.get("_local_line_bboxes")
283
+ if (
284
+ candidate_index == host_index
285
+ or candidate_index in consumed
286
+ or candidate.get("type") != "text"
287
+ or int(candidate.get("angle", 0) or 0) % 360 != angle
288
+ or not isinstance(candidate_bbox, (list, tuple))
289
+ or not isinstance(candidate_rows, list)
290
+ or len(candidate_rows) != 1
291
+ or candidate_bbox[2] - candidate_bbox[0] > 0.25 * (host_bbox[2] - host_bbox[0])
292
+ or _bbox_axis_overlap_ratio(
293
+ host_bbox,
294
+ candidate_bbox,
295
+ axis="x",
296
+ )
297
+ <= 0.0
298
+ ):
299
+ continue
300
+ union_bbox = _bbox_union_many([host_bbox, candidate_bbox])
301
+ vertical_gap = max(
302
+ 0.0,
303
+ max(host_bbox[1], candidate_bbox[1]) - min(host_bbox[3], candidate_bbox[3]),
304
+ )
305
+ if vertical_gap <= 0.75 * host_height and union_bbox[3] - union_bbox[1] <= 3.5 * host_height:
306
+ fragment_indices.append(candidate_index)
307
+ if len(fragment_indices) < 2:
308
+ continue
309
+ group_indices = [host_index, *fragment_indices]
310
+ replacement_index = min(group_indices)
311
+ replacements[replacement_index] = _merge_inline_math_recovery_group(
312
+ blocks,
313
+ group_indices,
314
+ )
315
+ consumed.update(group_indices)
316
+ output = [
317
+ replacements.get(index, block) for index, block in enumerate(blocks) if index not in consumed or index in replacements
318
+ ]
319
+ output = _merge_hostless_inline_math_fragment_blocks(output, page_size)
320
+ output = _merge_residual_narrow_math_text_blocks(output, page_size)
321
+ return _merge_inline_math_paragraph_continuations(output, page_size)
322
+
323
+
324
+ def _component_local_union_bbox(
325
+ block: dict[str, Any],
326
+ ) -> BBox | None:
327
+ """合并正文组件持有的正向行框,非法或缺失元数据时返回空。"""
328
+
329
+ rows = block.get("_local_line_bboxes")
330
+ if not isinstance(rows, list):
331
+ return None
332
+ bboxes: list[BBox] = []
333
+ for row in rows:
334
+ if not isinstance(row, (list, tuple)) or len(row) != 4:
335
+ continue
336
+ try:
337
+ bbox = tuple(float(value) for value in row)
338
+ except (TypeError, ValueError):
339
+ continue
340
+ if bbox[2] > bbox[0] and bbox[3] > bbox[1]:
341
+ bboxes.append(bbox) # type: ignore[arg-type]
342
+ return _bbox_union_many(bboxes) if bboxes else None
343
+
344
+
345
+ def _merge_paragraph_formula_context_blocks(
346
+ blocks: list[dict[str, Any]],
347
+ page_size: tuple[float, float],
348
+ ) -> list[dict[str, Any]]:
349
+ """把误似行间公式的复杂行内分式与同栏前后正文恢复成一个块。"""
350
+
351
+ terminal_re = re.compile(
352
+ r"[.!?。!?][\]\)})】》”’'\"]*$",
353
+ )
354
+ consumed: set[int] = set()
355
+ replacements: dict[int, dict[str, Any]] = {}
356
+ seed_indices = [
357
+ index
358
+ for index, block in enumerate(blocks)
359
+ if block.get("type") == "text" and block.get(_PARAGRAPH_FORMULA_CONTEXT_MARKER) is True
360
+ ]
361
+ for seed_index in seed_indices:
362
+ if seed_index in consumed:
363
+ continue
364
+ group = {seed_index}
365
+ changed = True
366
+ while changed:
367
+ changed = False
368
+ for candidate_index, candidate in enumerate(blocks):
369
+ if candidate_index in group or candidate_index in consumed or candidate.get("type") != "text":
370
+ continue
371
+ for member_index in group:
372
+ member = blocks[member_index]
373
+ if int(candidate.get("angle", 0) or 0) % 360 != int(member.get("angle", 0) or 0) % 360:
374
+ continue
375
+ pair_heights = [
376
+ float(height)
377
+ for block in (candidate, member)
378
+ for height in block.get("_line_heights", [])
379
+ if isinstance(height, (int, float)) and float(height) > 0
380
+ ]
381
+ if not pair_heights:
382
+ continue
383
+ pair_height = statistics.median(pair_heights)
384
+ if not _components_share_lane_role(
385
+ candidate,
386
+ member,
387
+ pair_height,
388
+ ):
389
+ continue
390
+ candidate_bbox = _component_local_union_bbox(candidate)
391
+ member_bbox = _component_local_union_bbox(member)
392
+ if candidate_bbox is None or member_bbox is None:
393
+ continue
394
+ vertical_gap = max(
395
+ candidate_bbox[1] - member_bbox[3],
396
+ member_bbox[1] - candidate_bbox[3],
397
+ 0.0,
398
+ )
399
+ if vertical_gap > 1.5 * pair_height:
400
+ continue
401
+ candidate_below = candidate_bbox[1] >= member_bbox[3]
402
+ member_below = member_bbox[1] >= candidate_bbox[3]
403
+ if candidate_below and (
404
+ candidate.get("_hard_break_before") is True
405
+ or terminal_re.search(str(member.get("content") or "").rstrip()) is not None
406
+ ):
407
+ continue
408
+ if member_below and (
409
+ member.get("_hard_break_before") is True
410
+ or terminal_re.search(str(candidate.get("content") or "").rstrip()) is not None
411
+ ):
412
+ continue
413
+ group.add(candidate_index)
414
+ changed = True
415
+ break
416
+ if changed:
417
+ break
418
+ if len(group) < 2:
419
+ continue
420
+ ordered_group = sorted(group)
421
+ replacement_index = min(ordered_group)
422
+ merged = _merge_internal_text_block_group(
423
+ blocks,
424
+ ordered_group,
425
+ )
426
+ local_rows = [
427
+ bbox for bbox in merged.get("_local_line_bboxes", []) if isinstance(bbox, (list, tuple)) and len(bbox) == 4
428
+ ]
429
+ maximum_width = max(
430
+ (bbox[2] - bbox[0] for bbox in local_rows),
431
+ default=0.0,
432
+ )
433
+ body_rows = [bbox for bbox in local_rows if bbox[2] - bbox[0] >= 0.75 * maximum_width]
434
+ if len(body_rows) >= 2:
435
+ # 复杂分式可能比正文左缘多探出少量 glyph;公开框按重复满行边界稳定收口。
436
+ local_merged_bbox = _bbox_union_many(local_rows)
437
+ local_output_bbox = (
438
+ min(bbox[0] for bbox in body_rows),
439
+ local_merged_bbox[1],
440
+ max(bbox[2] for bbox in body_rows),
441
+ local_merged_bbox[3],
442
+ )
443
+ merged["bbox"] = _rotate_bbox_from_upright(
444
+ local_output_bbox,
445
+ page_size,
446
+ int(merged.get("angle", 0) or 0) % 360,
447
+ )
448
+ replacements[replacement_index] = merged
449
+ consumed.update(ordered_group)
450
+ return [
451
+ replacements.get(index, block) for index, block in enumerate(blocks) if index not in consumed or index in replacements
452
+ ]
453
+
454
+
455
+ def _merge_residual_narrow_math_text_blocks(
456
+ blocks: list[dict[str, Any]],
457
+ page_size: tuple[float, float],
458
+ ) -> list[dict[str, Any]]:
459
+ """把仍嵌在宽正文行范围内的单个窄数学碎片吸收到唯一宿主块。"""
460
+
461
+ consumed: set[int] = set()
462
+ replacements: dict[int, dict[str, Any]] = {}
463
+ for candidate_index, candidate in enumerate(blocks):
464
+ candidate_bbox = candidate.get("bbox")
465
+ candidate_rows = candidate.get("_local_line_bboxes")
466
+ angle = int(candidate.get("angle", 0) or 0) % 360
467
+ local_page_width = page_size[1] if angle in {90, 270} else page_size[0]
468
+ if (
469
+ candidate.get("type") != "text"
470
+ or not isinstance(candidate_bbox, (list, tuple))
471
+ or not isinstance(candidate_rows, list)
472
+ or len(candidate_rows) != 1
473
+ or candidate_bbox[2] - candidate_bbox[0] > 0.05 * local_page_width
474
+ ):
475
+ continue
476
+ candidate_heights = [
477
+ float(height) for height in candidate.get("_line_heights", []) if isinstance(height, (int, float)) and height > 0
478
+ ]
479
+ candidate_height = statistics.median(candidate_heights) if candidate_heights else candidate_bbox[3] - candidate_bbox[1]
480
+ hosts: list[tuple[float, float, int]] = []
481
+ for host_index, host in enumerate(blocks):
482
+ host_bbox = host.get("bbox")
483
+ host_heights = [
484
+ float(height) for height in host.get("_line_heights", []) if isinstance(height, (int, float)) and height > 0
485
+ ]
486
+ if (
487
+ host_index == candidate_index
488
+ or host_index in consumed
489
+ or host.get("type") != "text"
490
+ or int(host.get("angle", 0) or 0) % 360 != angle
491
+ or not isinstance(host_bbox, (list, tuple))
492
+ or host_bbox[2] - host_bbox[0] < 0.3 * local_page_width
493
+ or candidate_bbox[0] < host_bbox[0]
494
+ or candidate_bbox[2] > host_bbox[2]
495
+ ):
496
+ continue
497
+ host_height = statistics.median(host_heights) if host_heights else host_bbox[3] - host_bbox[1]
498
+ vertical_overlap = max(
499
+ 0.0,
500
+ min(candidate_bbox[3], host_bbox[3]) - max(candidate_bbox[1], host_bbox[1]),
501
+ )
502
+ vertical_gap = max(
503
+ 0.0,
504
+ max(candidate_bbox[1], host_bbox[1]) - min(candidate_bbox[3], host_bbox[3]),
505
+ )
506
+ pair_height = max(candidate_height, host_height, 0.1)
507
+ union_bbox = _bbox_union_many([candidate_bbox, host_bbox])
508
+ if (
509
+ vertical_overlap < 0.35 * min(candidate_height, host_height) and vertical_gap > 0.5 * pair_height
510
+ ) or union_bbox[3] - union_bbox[1] > 3.0 * pair_height:
511
+ continue
512
+ hosts.append(
513
+ (
514
+ -vertical_overlap,
515
+ abs(_bbox_center_y(candidate_bbox) - _bbox_center_y(host_bbox)),
516
+ host_index,
517
+ )
518
+ )
519
+ if len(hosts) != 1:
520
+ continue
521
+ host_index = hosts[0][2]
522
+ replacement_index = min(candidate_index, host_index)
523
+ replacements[replacement_index] = _merge_inline_math_recovery_group(
524
+ blocks,
525
+ [candidate_index, host_index],
526
+ )
527
+ consumed.update({candidate_index, host_index})
528
+ return [
529
+ replacements.get(index, block) for index, block in enumerate(blocks) if index not in consumed or index in replacements
530
+ ]
531
+
532
+
533
+ def _merge_hostless_inline_math_fragment_blocks(
534
+ blocks: list[dict[str, Any]],
535
+ page_size: tuple[float, float],
536
+ ) -> list[dict[str, Any]]:
537
+ """把没有单一宽宿主但在一栏内二维密集排列的数学碎片合成文本块。"""
538
+
539
+ grouped_indices: list[list[int]] = []
540
+ for angle in sorted({int(block.get("angle", 0) or 0) % 360 for block in blocks if block.get("type") == "text"}):
541
+ local_page_width = page_size[1] if angle in {90, 270} else page_size[0]
542
+ candidates = {
543
+ index
544
+ for index, block in enumerate(blocks)
545
+ if block.get("type") == "text"
546
+ and int(block.get("angle", 0) or 0) % 360 == angle
547
+ and isinstance(block.get("bbox"), (list, tuple))
548
+ and isinstance(block.get("_local_line_bboxes"), list)
549
+ and len(block["_local_line_bboxes"]) == 1
550
+ and block["bbox"][2] - block["bbox"][0] <= 0.25 * local_page_width
551
+ }
552
+ while candidates:
553
+ component = {candidates.pop()}
554
+ changed = True
555
+ while changed:
556
+ changed = False
557
+ for candidate_index in list(candidates):
558
+ candidate_bbox = blocks[candidate_index]["bbox"]
559
+ candidate_heights = blocks[candidate_index].get(
560
+ "_line_heights",
561
+ [],
562
+ )
563
+ candidate_height = (
564
+ statistics.median(candidate_heights) if candidate_heights else candidate_bbox[3] - candidate_bbox[1]
565
+ )
566
+ if any(
567
+ (
568
+ max(
569
+ 0.0,
570
+ max(candidate_bbox[1], blocks[index]["bbox"][1])
571
+ - min(candidate_bbox[3], blocks[index]["bbox"][3]),
572
+ )
573
+ <= 0.75
574
+ * max(
575
+ candidate_height,
576
+ statistics.median(blocks[index].get("_line_heights", []))
577
+ if blocks[index].get("_line_heights")
578
+ else blocks[index]["bbox"][3] - blocks[index]["bbox"][1],
579
+ )
580
+ and max(
581
+ 0.0,
582
+ max(candidate_bbox[0], blocks[index]["bbox"][0])
583
+ - min(candidate_bbox[2], blocks[index]["bbox"][2]),
584
+ )
585
+ <= 1.5
586
+ * max(
587
+ candidate_height,
588
+ statistics.median(blocks[index].get("_line_heights", []))
589
+ if blocks[index].get("_line_heights")
590
+ else blocks[index]["bbox"][3] - blocks[index]["bbox"][1],
591
+ )
592
+ )
593
+ for index in component
594
+ ):
595
+ component.add(candidate_index)
596
+ candidates.remove(candidate_index)
597
+ changed = True
598
+ if len(component) < 4:
599
+ continue
600
+ component_heights = [
601
+ float(height)
602
+ for index in component
603
+ for height in blocks[index].get("_line_heights", [])
604
+ if isinstance(height, (int, float)) and height > 0
605
+ ]
606
+ if not component_heights:
607
+ continue
608
+ median_height = statistics.median(component_heights)
609
+ union_bbox = _bbox_union_many([blocks[index]["bbox"] for index in component])
610
+ font_signatures = set().union(
611
+ *[
612
+ signatures
613
+ for index in component
614
+ if isinstance(
615
+ (signatures := blocks[index].get("_font_signatures")),
616
+ set,
617
+ )
618
+ ]
619
+ )
620
+ if (
621
+ len(font_signatures) < 2
622
+ or not 0.25 * local_page_width <= union_bbox[2] - union_bbox[0] <= 0.5 * local_page_width
623
+ or union_bbox[3] - union_bbox[1] > 3.5 * median_height
624
+ or sum(blocks[index]["bbox"][2] - blocks[index]["bbox"][0] >= 3.5 * median_height for index in component) < 2
625
+ ):
626
+ continue
627
+ grouped_indices.append(sorted(component))
628
+
629
+ consumed: set[int] = set()
630
+ replacements: dict[int, dict[str, Any]] = {}
631
+ for group in grouped_indices:
632
+ if any(index in consumed for index in group):
633
+ continue
634
+ replacement_index = min(group)
635
+ replacements[replacement_index] = _merge_inline_math_recovery_group(
636
+ blocks,
637
+ group,
638
+ )
639
+ consumed.update(group)
640
+ return [
641
+ replacements.get(index, block) for index, block in enumerate(blocks) if index not in consumed or index in replacements
642
+ ]
643
+
644
+
645
+ def _merge_inline_math_recovery_group(
646
+ blocks: list[dict[str, Any]],
647
+ indices: list[int],
648
+ ) -> dict[str, Any]:
649
+ """合并数学碎片并保留仅供后续段落闭合使用的内部标记。"""
650
+ member_bboxes = [blocks[index]["bbox"] for index in indices]
651
+ widths = [bbox[2] - bbox[0] for bbox in member_bboxes]
652
+ maximum_width = max(widths, default=0.0)
653
+ detected_regions = [
654
+ bbox for bbox, width in zip(member_bboxes, widths) if maximum_width <= 0 or width <= 0.25 * maximum_width
655
+ ]
656
+ if not detected_regions:
657
+ detected_regions = list(member_bboxes)
658
+ merged = _merge_internal_text_block_group(blocks, indices)
659
+ merged[_INLINE_MATH_RECOVERY_MARKER] = True
660
+ merged["_inline_math_regions"] = [
661
+ *merged.get("_inline_math_regions", []),
662
+ *detected_regions,
663
+ ]
664
+ return merged
665
+
666
+
667
+ def _merge_inline_math_paragraph_continuations(
668
+ blocks: list[dict[str, Any]],
669
+ page_size: tuple[float, float],
670
+ ) -> list[dict[str, Any]]:
671
+ """在数学碎片恢复后,合并同栏连续且足够宽的正文段落块。"""
672
+
673
+ if sum(block.get(_INLINE_MATH_RECOVERY_MARKER) is True for block in blocks) < 2:
674
+ return blocks
675
+
676
+ lane_groups: list[list[int]] = []
677
+ for index, block in sorted(
678
+ enumerate(blocks),
679
+ key=lambda item: (
680
+ int(item[1].get("angle", 0) or 0) % 360,
681
+ _text_component_sort_key(item[1])
682
+ if isinstance(item[1].get("_local_line_bboxes"), list) and item[1]["_local_line_bboxes"]
683
+ else (float("inf"), float("inf")),
684
+ ),
685
+ ):
686
+ local_rows = block.get("_local_line_bboxes")
687
+ if (
688
+ block.get("type") != "text"
689
+ or not isinstance(block.get("_lane_is_span"), bool)
690
+ or not isinstance(local_rows, list)
691
+ or not local_rows
692
+ ):
693
+ continue
694
+ block_heights = [
695
+ float(height) for height in block.get("_line_heights", []) if isinstance(height, (int, float)) and height > 0
696
+ ]
697
+ if not block_heights:
698
+ continue
699
+ for lane_group in lane_groups:
700
+ representative = blocks[lane_group[0]]
701
+ if int(representative.get("angle", 0) or 0) % 360 != int(block.get("angle", 0) or 0) % 360:
702
+ continue
703
+ representative_heights = [
704
+ float(height)
705
+ for height in representative.get("_line_heights", [])
706
+ if isinstance(height, (int, float)) and height > 0
707
+ ]
708
+ pair_height = statistics.median([*representative_heights, *block_heights])
709
+ if _components_share_lane_role(representative, block, pair_height):
710
+ lane_group.append(index)
711
+ break
712
+ else:
713
+ lane_groups.append([index])
714
+
715
+ candidate_chains: list[list[int]] = []
716
+ for lane_group in lane_groups:
717
+ ordered_indices = sorted(
718
+ lane_group,
719
+ key=lambda index: _text_component_sort_key(blocks[index]),
720
+ )
721
+ chain = [ordered_indices[0]]
722
+ for current_index in ordered_indices[1:]:
723
+ previous_index = chain[-1]
724
+ previous = blocks[previous_index]
725
+ current = blocks[current_index]
726
+ previous_rows = previous["_local_line_bboxes"]
727
+ current_rows = current["_local_line_bboxes"]
728
+ previous_local_bbox = _bbox_union_many(previous_rows)
729
+ current_local_bbox = _bbox_union_many(current_rows)
730
+ pair_heights = [
731
+ float(height)
732
+ for block in (previous, current)
733
+ for height in block.get("_line_heights", [])
734
+ if isinstance(height, (int, float)) and height > 0
735
+ ]
736
+ pair_height = statistics.median(pair_heights)
737
+ angle = int(previous.get("angle", 0) or 0) % 360
738
+ local_page_width = page_size[1] if angle in {90, 270} else page_size[0]
739
+ lane_width = _compatible_component_lane_width(
740
+ previous,
741
+ current,
742
+ local_page_width,
743
+ pair_height,
744
+ )
745
+ previous_fonts = previous.get("_font_signatures")
746
+ current_fonts = current.get("_font_signatures")
747
+ fonts_conflict = (
748
+ isinstance(previous_fonts, set)
749
+ and isinstance(current_fonts, set)
750
+ and previous_fonts
751
+ and current_fonts
752
+ and previous_fonts.isdisjoint(current_fonts)
753
+ )
754
+ vertical_gap = current_local_bbox[1] - previous_local_bbox[3]
755
+ connects = (
756
+ _components_share_lane_role(previous, current, pair_height)
757
+ and previous_local_bbox[2] - previous_local_bbox[0] >= 0.8 * lane_width
758
+ and current_local_bbox[2] - current_local_bbox[0] >= 0.8 * lane_width
759
+ and abs(previous_local_bbox[0] - current_local_bbox[0]) <= 0.75 * pair_height
760
+ and -0.75 * pair_height <= vertical_gap <= 0.75 * pair_height
761
+ and not fonts_conflict
762
+ and not _component_connection_skips_block(
763
+ blocks,
764
+ previous_index,
765
+ current_index,
766
+ pair_height,
767
+ )
768
+ )
769
+ if connects:
770
+ chain.append(current_index)
771
+ else:
772
+ candidate_chains.append(chain)
773
+ chain = [current_index]
774
+ candidate_chains.append(chain)
775
+
776
+ consumed: set[int] = set()
777
+ replacements: dict[int, dict[str, Any]] = {}
778
+ for chain in candidate_chains:
779
+ if len(chain) < 3 or sum(blocks[index].get(_INLINE_MATH_RECOVERY_MARKER) is True for index in chain) < 2:
780
+ continue
781
+ chain_heights = [
782
+ float(height)
783
+ for index in chain
784
+ for height in blocks[index].get("_line_heights", [])
785
+ if isinstance(height, (int, float)) and height > 0
786
+ ]
787
+ median_height = statistics.median(chain_heights)
788
+ local_union = _bbox_union_many([bbox for index in chain for bbox in blocks[index]["_local_line_bboxes"]])
789
+ if local_union[3] - local_union[1] > 24.0 * median_height:
790
+ continue
791
+ replacement_index = min(chain)
792
+ merged = _merge_internal_text_block_group(blocks, chain)
793
+ merged[_INLINE_MATH_RECOVERY_MARKER] = True
794
+ replacements[replacement_index] = merged
795
+ consumed.update(chain)
796
+ return [
797
+ replacements.get(index, block) for index, block in enumerate(blocks) if index not in consumed or index in replacements
798
+ ]
799
+
800
+
801
+ def _merge_spatial_text_components(
802
+ blocks: list[dict[str, Any]],
803
+ page_size: tuple[float, float],
804
+ ) -> list[dict[str, Any]]:
805
+ """按短首行、紧邻续行和双栏递减尾行二次连接被栏带拆开的正文块。"""
806
+
807
+ parents = list(range(len(blocks)))
808
+
809
+ def find(index: int) -> int:
810
+ """查找正文组件所属合并组的根节点。"""
811
+
812
+ while parents[index] != index:
813
+ parents[index] = parents[parents[index]]
814
+ index = parents[index]
815
+ return index
816
+
817
+ def union(first_index: int, second_index: int) -> None:
818
+ """合并两个已经通过空间连续性校验的正文组件。"""
819
+
820
+ first_root = find(first_index)
821
+ second_root = find(second_index)
822
+ if first_root != second_root:
823
+ parents[second_root] = first_root
824
+
825
+ for angle in sorted(
826
+ {
827
+ int(block.get("angle", 0) or 0) % 360
828
+ for block in blocks
829
+ if block.get("type") == "text" and block.get("_local_line_bboxes")
830
+ }
831
+ ):
832
+ candidate_indices = [
833
+ index
834
+ for index, block in enumerate(blocks)
835
+ if block.get("type") == "text"
836
+ and int(block.get("angle", 0) or 0) % 360 == angle
837
+ and isinstance(block.get("_local_line_bboxes"), list)
838
+ and block["_local_line_bboxes"]
839
+ ]
840
+ if len(candidate_indices) < 2:
841
+ continue
842
+ local_page_width = page_size[1] if angle in {90, 270} else page_size[0]
843
+ all_heights = [
844
+ float(height)
845
+ for index in candidate_indices
846
+ for height in blocks[index].get("_line_heights", [])
847
+ if isinstance(height, (int, float)) and height > 0
848
+ ]
849
+ median_height = statistics.median(all_heights) if all_heights else 1.0
850
+ section_starts = {
851
+ index
852
+ for index in candidate_indices
853
+ if _block_starts_with_short_wide_rows(
854
+ blocks[index],
855
+ local_page_width,
856
+ )
857
+ }
858
+
859
+ opener_pairs = _find_short_opener_pairs(
860
+ blocks,
861
+ candidate_indices,
862
+ local_page_width,
863
+ median_height,
864
+ )
865
+ for opener_index, body_index in opener_pairs:
866
+ section_starts.add(opener_index)
867
+ union(opener_index, body_index)
868
+
869
+ for current_index in sorted(
870
+ candidate_indices,
871
+ key=lambda index: _text_component_sort_key(blocks[index]),
872
+ ):
873
+ section_roots = {find(index) for index in section_starts}
874
+ if find(current_index) not in section_roots:
875
+ continue
876
+ next_index = _nearest_following_text_component(
877
+ blocks,
878
+ current_index,
879
+ candidate_indices,
880
+ maximum_gap=0.75 * median_height,
881
+ section_starts=section_starts,
882
+ )
883
+ if next_index is not None:
884
+ union(current_index, next_index)
885
+
886
+ for current_index in candidate_indices:
887
+ if not _has_parallel_text_component(
888
+ blocks,
889
+ current_index,
890
+ candidate_indices,
891
+ ):
892
+ continue
893
+ next_index = _nearest_tapered_tail_component(
894
+ blocks,
895
+ current_index,
896
+ candidate_indices,
897
+ median_height,
898
+ section_starts,
899
+ )
900
+ if next_index is not None:
901
+ union(current_index, next_index)
902
+
903
+ grouped_indices: dict[int, list[int]] = {}
904
+ for index in range(len(blocks)):
905
+ grouped_indices.setdefault(find(index), []).append(index)
906
+
907
+ output: list[dict[str, Any]] = []
908
+ for indices in grouped_indices.values():
909
+ if len(indices) == 1:
910
+ output.append(blocks[indices[0]])
911
+ continue
912
+ ordered_indices = sorted(
913
+ indices,
914
+ key=lambda index: _text_component_sort_key(blocks[index]),
915
+ )
916
+ merged = dict(blocks[ordered_indices[0]])
917
+ merged["bbox"] = _bbox_union_many([blocks[index]["bbox"] for index in ordered_indices])
918
+ merged["content"] = _merge_text_line_content([str(blocks[index].get("content", "")) for index in ordered_indices])
919
+ merged["_visual_row_ids"] = set().union(
920
+ *[
921
+ block_ids
922
+ for index in ordered_indices
923
+ if isinstance(
924
+ (block_ids := blocks[index].get("_visual_row_ids")),
925
+ set,
926
+ )
927
+ ]
928
+ )
929
+ merged["_single_run_row_id"] = None
930
+ merged["_local_line_bboxes"] = [
931
+ bbox for index in ordered_indices for bbox in blocks[index].get("_local_line_bboxes", [])
932
+ ]
933
+ merged["_local_output_line_bboxes"] = [
934
+ bbox for index in ordered_indices for bbox in blocks[index].get("_local_output_line_bboxes", [])
935
+ ]
936
+ merged["_output_bbox_repaired"] = any(blocks[index].get("_output_bbox_repaired") is True for index in ordered_indices)
937
+ merged["_line_heights"] = [height for index in ordered_indices for height in blocks[index].get("_line_heights", [])]
938
+ merged["_font_signatures"] = set().union(
939
+ *[
940
+ signatures
941
+ for index in ordered_indices
942
+ if isinstance(
943
+ (signatures := blocks[index].get("_font_signatures")),
944
+ set,
945
+ )
946
+ ]
947
+ )
948
+ merged["_inline_math_regions"] = [
949
+ region for index in ordered_indices for region in blocks[index].get("_inline_math_regions", [])
950
+ ]
951
+ output.append(merged)
952
+ return output
953
+
954
+
955
+ def _merge_list_intro_text_components(
956
+ blocks: list[dict[str, Any]],
957
+ ) -> list[dict[str, Any]]:
958
+ """在编号列表硬边界前合并被误拆的连续引导段和冒号短尾。"""
959
+
960
+ consumed: set[int] = set()
961
+ replacements: dict[int, dict[str, Any]] = {}
962
+ text_indices = [
963
+ index
964
+ for index, block in enumerate(blocks)
965
+ if block.get("type") == "text" and isinstance(block.get("_local_line_bboxes"), list) and block["_local_line_bboxes"]
966
+ ]
967
+ for boundary_index in text_indices:
968
+ boundary = blocks[boundary_index]
969
+ if (
970
+ boundary.get("_hard_break_before") is not True
971
+ or _LIST_ITEM_RE.match(
972
+ str(boundary.get("content") or ""),
973
+ )
974
+ is None
975
+ ):
976
+ continue
977
+ preceding = [
978
+ index
979
+ for index in text_indices
980
+ if index not in consumed and _text_component_sort_key(blocks[index]) < _text_component_sort_key(boundary)
981
+ ]
982
+ if not preceding:
983
+ continue
984
+ immediate_index = max(
985
+ preceding,
986
+ key=lambda index: _text_component_sort_key(blocks[index]),
987
+ )
988
+ immediate = blocks[immediate_index]
989
+ immediate_rows = immediate["_local_line_bboxes"]
990
+ immediate_heights = [
991
+ float(height) for height in immediate.get("_line_heights", []) if isinstance(height, (int, float)) and height > 0
992
+ ]
993
+ pair_height = statistics.median(
994
+ immediate_heights
995
+ or [
996
+ immediate_rows[-1][3] - immediate_rows[-1][1],
997
+ ],
998
+ )
999
+ interval = _component_lane_interval(immediate)
1000
+ if (
1001
+ interval is None
1002
+ or immediate_rows[-1][2] - immediate_rows[-1][0] > 0.35 * (interval[1] - interval[0])
1003
+ or not str(immediate.get("content") or "").rstrip().endswith((":", ":"))
1004
+ or not _components_share_lane_role(
1005
+ immediate,
1006
+ boundary,
1007
+ pair_height,
1008
+ )
1009
+ ):
1010
+ continue
1011
+
1012
+ group = [immediate_index]
1013
+ cursor_index = immediate_index
1014
+ while len(group) < 3:
1015
+ earlier = [
1016
+ index
1017
+ for index in preceding
1018
+ if index not in group
1019
+ and _text_component_sort_key(blocks[index]) < _text_component_sort_key(blocks[cursor_index])
1020
+ and _components_share_lane_role(
1021
+ blocks[index],
1022
+ blocks[cursor_index],
1023
+ pair_height,
1024
+ )
1025
+ ]
1026
+ if not earlier:
1027
+ break
1028
+ previous_index = max(
1029
+ earlier,
1030
+ key=lambda index: _text_component_sort_key(blocks[index]),
1031
+ )
1032
+ previous = blocks[previous_index]
1033
+ current = blocks[cursor_index]
1034
+ previous_rows = previous["_local_line_bboxes"]
1035
+ current_rows = current["_local_line_bboxes"]
1036
+ heights = [
1037
+ float(height)
1038
+ for block in (previous, current)
1039
+ for height in block.get("_line_heights", [])
1040
+ if isinstance(height, (int, float)) and height > 0
1041
+ ]
1042
+ local_height = statistics.median(heights or [pair_height])
1043
+ vertical_gap = current_rows[0][1] - previous_rows[-1][3]
1044
+ if (
1045
+ previous.get("_hard_break_before") is True
1046
+ or not _components_share_lane_role(
1047
+ previous,
1048
+ current,
1049
+ local_height,
1050
+ )
1051
+ or not -local_height <= vertical_gap <= local_height
1052
+ ):
1053
+ break
1054
+ group.append(previous_index)
1055
+ cursor_index = previous_index
1056
+ if len(group) < 2:
1057
+ continue
1058
+ ordered_group = sorted(
1059
+ group,
1060
+ key=lambda index: _text_component_sort_key(blocks[index]),
1061
+ )
1062
+ replacement_index = min(ordered_group)
1063
+ replacements[replacement_index] = _merge_internal_text_block_group(
1064
+ blocks,
1065
+ ordered_group,
1066
+ )
1067
+ consumed.update(ordered_group)
1068
+
1069
+ return [
1070
+ replacements.get(index, block) for index, block in enumerate(blocks) if index not in consumed or index in replacements
1071
+ ]
1072
+
1073
+
1074
+ def _merge_unterminated_text_components(
1075
+ blocks: list[dict[str, Any]],
1076
+ ) -> list[dict[str, Any]]:
1077
+ """合并普通栏未终止正文,以及满足严格结构约束的满宽 span 正文。"""
1078
+
1079
+ output = list(blocks)
1080
+ terminal_re = re.compile(
1081
+ r"[.!?。!?::;;][\]\)})】》”’'\"]*$",
1082
+ )
1083
+ while True:
1084
+ text_indices = sorted(
1085
+ (
1086
+ index
1087
+ for index, block in enumerate(output)
1088
+ if block.get("type") == "text"
1089
+ and isinstance(
1090
+ block.get("_local_line_bboxes"),
1091
+ list,
1092
+ )
1093
+ and block["_local_line_bboxes"]
1094
+ ),
1095
+ key=lambda index: _text_component_sort_key(output[index]),
1096
+ )
1097
+ merged_pair: tuple[int, int] | None = None
1098
+ for first_index, second_index in zip(
1099
+ text_indices,
1100
+ text_indices[1:],
1101
+ ):
1102
+ first = output[first_index]
1103
+ second = output[second_index]
1104
+ second_rows = second["_local_line_bboxes"]
1105
+ first_rows = first["_local_line_bboxes"]
1106
+ heights = [
1107
+ float(height)
1108
+ for block in (first, second)
1109
+ for height in block.get("_line_heights", [])
1110
+ if isinstance(height, (int, float)) and height > 0
1111
+ ]
1112
+ pair_height = statistics.median(
1113
+ heights
1114
+ or [
1115
+ first_rows[-1][3] - first_rows[-1][1],
1116
+ second_rows[0][3] - second_rows[0][1],
1117
+ ],
1118
+ )
1119
+ second_interval = _component_lane_interval(second)
1120
+ first_declared_interval = _component_declared_lane_interval(
1121
+ first,
1122
+ )
1123
+ second_declared_interval = _component_declared_lane_interval(
1124
+ second,
1125
+ )
1126
+ row_pair_height = statistics.median(
1127
+ [
1128
+ max(
1129
+ 0.1,
1130
+ first_rows[-1][3] - first_rows[-1][1],
1131
+ ),
1132
+ max(
1133
+ 0.1,
1134
+ second_rows[0][3] - second_rows[0][1],
1135
+ ),
1136
+ ],
1137
+ )
1138
+ span_connection_height = max(
1139
+ pair_height,
1140
+ row_pair_height,
1141
+ )
1142
+ span_tolerance = 0.75 * span_connection_height
1143
+ span_pair = (
1144
+ first.get("_lane_is_span") is True
1145
+ and second.get("_lane_is_span") is True
1146
+ and int(first.get("angle", 0) or 0) % 360 == int(second.get("angle", 0) or 0) % 360
1147
+ and first_declared_interval is not None
1148
+ and second_declared_interval is not None
1149
+ and abs(
1150
+ first_declared_interval[0] - second_declared_interval[0],
1151
+ )
1152
+ <= span_tolerance
1153
+ and abs(
1154
+ first_declared_interval[1] - second_declared_interval[1],
1155
+ )
1156
+ <= span_tolerance
1157
+ )
1158
+ first_content = str(first.get("content") or "")
1159
+ second_content = str(second.get("content") or "")
1160
+ single_numbered_tail = (
1161
+ len(second_rows) == 1
1162
+ and second.get("_hard_break_before") is not True
1163
+ and _LIST_ITEM_RE.match(second_content) is not None
1164
+ and first_content.rstrip().endswith((":", ":"))
1165
+ )
1166
+ url_continuation = _URL_LINE_RE.match(second_content) is not None
1167
+ aligned_short_tail = (
1168
+ len(first_rows) >= 2
1169
+ and len(second_rows) == 1
1170
+ and abs(second_rows[0][0] - first_rows[-1][0]) <= 0.75 * pair_height
1171
+ )
1172
+ narrow_continuation = single_numbered_tail or url_continuation or aligned_short_tail
1173
+ starts_wide_label = (
1174
+ not url_continuation
1175
+ and _LABELLED_METADATA_RE.match(
1176
+ second_content,
1177
+ )
1178
+ is not None
1179
+ and (reference_interval := (second_declared_interval if span_pair else second_interval)) is not None
1180
+ and second_rows[0][2] - second_rows[0][0] >= 0.5 * (reference_interval[1] - reference_interval[0])
1181
+ )
1182
+ if (
1183
+ second.get("_protected_hard_break_before") is True
1184
+ or (second.get("_hard_break_before") is True and (span_pair or not narrow_continuation))
1185
+ or second.get("_leading_emphasis_start") is True
1186
+ or (starts_wide_label and (span_pair or not aligned_short_tail))
1187
+ or first.get("_hanging_indent_group") is not None
1188
+ or second.get("_hanging_indent_group") is not None
1189
+ or _FIGURE_CAPTION_MARKER_RE.match(
1190
+ first_content,
1191
+ )
1192
+ is not None
1193
+ or (
1194
+ not span_pair
1195
+ and not single_numbered_tail
1196
+ and not url_continuation
1197
+ and terminal_re.search(
1198
+ first_content.rstrip(),
1199
+ )
1200
+ is not None
1201
+ )
1202
+ ):
1203
+ continue
1204
+ interval = _component_lane_interval(first)
1205
+ if span_pair:
1206
+ interval = first_declared_interval
1207
+ elif interval is None or not _components_share_lane_role(
1208
+ first,
1209
+ second,
1210
+ pair_height,
1211
+ ):
1212
+ continue
1213
+ if interval is None:
1214
+ continue
1215
+ lane_width = interval[1] - interval[0]
1216
+ vertical_gap = second_rows[0][1] - first_rows[-1][3]
1217
+ minimum_first_fill = 0.8 if span_pair else 0.5 if single_numbered_tail else 0.65
1218
+ minimum_second_fill = 0.8 if span_pair else 0.65
1219
+ connection_height = span_connection_height if span_pair else pair_height
1220
+ first_reference_width = (
1221
+ max(row[2] - row[0] for row in first_rows) if narrow_continuation else first_rows[-1][2] - first_rows[-1][0]
1222
+ )
1223
+ first_fonts = first.get("_font_signatures")
1224
+ second_fonts = second.get("_font_signatures")
1225
+ fonts_conflict = (
1226
+ isinstance(first_fonts, set)
1227
+ and isinstance(second_fonts, set)
1228
+ and first_fonts
1229
+ and second_fonts
1230
+ and first_fonts.isdisjoint(second_fonts)
1231
+ )
1232
+ if (
1233
+ first_reference_width < minimum_first_fill * lane_width
1234
+ or (
1235
+ (span_pair or not narrow_continuation)
1236
+ and second_rows[0][2] - second_rows[0][0] < minimum_second_fill * lane_width
1237
+ )
1238
+ or not -connection_height <= vertical_gap <= 1.5 * connection_height
1239
+ or fonts_conflict
1240
+ or _component_connection_skips_block(
1241
+ output,
1242
+ first_index,
1243
+ second_index,
1244
+ connection_height,
1245
+ )
1246
+ ):
1247
+ continue
1248
+ merged_pair = (first_index, second_index)
1249
+ break
1250
+ if merged_pair is None:
1251
+ return output
1252
+ first_index, second_index = merged_pair
1253
+ output[first_index] = _merge_internal_text_block_group(
1254
+ output,
1255
+ [first_index, second_index],
1256
+ )
1257
+ output.pop(second_index)
1258
+
1259
+
1260
+ __all__ = [
1261
+ "_merge_short_same_baseline_prefix_blocks",
1262
+ "_blocks_share_boundary_visual_row",
1263
+ "_merge_overlapping_same_line_text_blocks",
1264
+ "_merge_inline_math_fragment_text_blocks",
1265
+ "_component_local_union_bbox",
1266
+ "_merge_paragraph_formula_context_blocks",
1267
+ "_merge_residual_narrow_math_text_blocks",
1268
+ "_merge_hostless_inline_math_fragment_blocks",
1269
+ "_merge_inline_math_recovery_group",
1270
+ "_merge_inline_math_paragraph_continuations",
1271
+ "_merge_spatial_text_components",
1272
+ "_merge_list_intro_text_components",
1273
+ "_merge_unterminated_text_components",
1274
+ ]