docvortex 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (364) hide show
  1. docvortex/__init__.py +21 -0
  2. docvortex/analyzers/__init__.py +3 -0
  3. docvortex/analyzers/native/__init__.py +37 -0
  4. docvortex/analyzers/native/_shared/__init__.py +3 -0
  5. docvortex/analyzers/native/_shared/hyperlink.py +16 -0
  6. docvortex/analyzers/native/_shared/image.py +8 -0
  7. docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
  8. docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
  9. docvortex/analyzers/native/_shared/markup/formula.py +37 -0
  10. docvortex/analyzers/native/_shared/markup/projector.py +57 -0
  11. docvortex/analyzers/native/_shared/markup/styles.py +19 -0
  12. docvortex/analyzers/native/_shared/mathml.py +17 -0
  13. docvortex/analyzers/native/_shared/names.py +7 -0
  14. docvortex/analyzers/native/_shared/xycut.py +414 -0
  15. docvortex/analyzers/native/contracts.py +44 -0
  16. docvortex/analyzers/native/csv.py +351 -0
  17. docvortex/analyzers/native/epub/__init__.py +18 -0
  18. docvortex/analyzers/native/epub/constants.py +45 -0
  19. docvortex/analyzers/native/epub/converter.py +80 -0
  20. docvortex/analyzers/native/epub/errors.py +20 -0
  21. docvortex/analyzers/native/epub/metadata.py +31 -0
  22. docvortex/analyzers/native/epub/package.py +649 -0
  23. docvortex/analyzers/native/epub/xhtml.py +306 -0
  24. docvortex/analyzers/native/html/__init__.py +6 -0
  25. docvortex/analyzers/native/html/anchors.py +271 -0
  26. docvortex/analyzers/native/html/constants.py +25 -0
  27. docvortex/analyzers/native/html/contracts.py +7 -0
  28. docvortex/analyzers/native/html/converter.py +117 -0
  29. docvortex/analyzers/native/html/document.py +389 -0
  30. docvortex/analyzers/native/html/errors.py +12 -0
  31. docvortex/analyzers/native/html/resources.py +365 -0
  32. docvortex/analyzers/native/html/selector.py +421 -0
  33. docvortex/analyzers/native/models.py +209 -0
  34. docvortex/analyzers/native/ofd/__init__.py +14 -0
  35. docvortex/analyzers/native/ofd/constants.py +59 -0
  36. docvortex/analyzers/native/ofd/converter.py +31 -0
  37. docvortex/analyzers/native/ofd/errors.py +16 -0
  38. docvortex/analyzers/native/ofd/geometry.py +215 -0
  39. docvortex/analyzers/native/ofd/images.py +124 -0
  40. docvortex/analyzers/native/ofd/metadata.py +53 -0
  41. docvortex/analyzers/native/ofd/models.py +170 -0
  42. docvortex/analyzers/native/ofd/package.py +346 -0
  43. docvortex/analyzers/native/ofd/path.py +201 -0
  44. docvortex/analyzers/native/ofd/reading_order.py +293 -0
  45. docvortex/analyzers/native/ofd/resources.py +122 -0
  46. docvortex/analyzers/native/ofd/scene.py +347 -0
  47. docvortex/analyzers/native/ofd/table.py +240 -0
  48. docvortex/analyzers/native/ofd/text.py +521 -0
  49. docvortex/analyzers/native/office/__init__.py +3 -0
  50. docvortex/analyzers/native/office/doc/__init__.py +3 -0
  51. docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
  52. docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
  53. docvortex/analyzers/native/office/doc/fib.py +261 -0
  54. docvortex/analyzers/native/office/doc/fields.py +101 -0
  55. docvortex/analyzers/native/office/doc/formatting.py +171 -0
  56. docvortex/analyzers/native/office/doc/images.py +162 -0
  57. docvortex/analyzers/native/office/doc/lists.py +356 -0
  58. docvortex/analyzers/native/office/doc/models.py +164 -0
  59. docvortex/analyzers/native/office/doc/parser.py +851 -0
  60. docvortex/analyzers/native/office/doc/pieces.py +256 -0
  61. docvortex/analyzers/native/office/doc/records.py +40 -0
  62. docvortex/analyzers/native/office/doc/sprm.py +269 -0
  63. docvortex/analyzers/native/office/doc/styles.py +215 -0
  64. docvortex/analyzers/native/office/docx/__init__.py +3 -0
  65. docvortex/analyzers/native/office/docx/context.py +71 -0
  66. docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
  67. docvortex/analyzers/native/office/docx/equationxml.py +119 -0
  68. docvortex/analyzers/native/office/docx/fields.py +816 -0
  69. docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
  70. docvortex/analyzers/native/office/docx/main.py +50 -0
  71. docvortex/analyzers/native/office/docx/numbering.py +491 -0
  72. docvortex/analyzers/native/office/docx/office_xml.py +57 -0
  73. docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
  74. docvortex/analyzers/native/office/docx/resources.py +526 -0
  75. docvortex/analyzers/native/office/docx/styles.py +693 -0
  76. docvortex/analyzers/native/office/docx/tables.py +515 -0
  77. docvortex/analyzers/native/office/equation/__init__.py +3 -0
  78. docvortex/analyzers/native/office/equation/image.py +470 -0
  79. docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
  80. docvortex/analyzers/native/office/equation/mtef.py +885 -0
  81. docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
  82. docvortex/analyzers/native/office/equation/omml.py +561 -0
  83. docvortex/analyzers/native/office/equation/ooxml.py +62 -0
  84. docvortex/analyzers/native/office/errors.py +33 -0
  85. docvortex/analyzers/native/office/image.py +307 -0
  86. docvortex/analyzers/native/office/legacy/__init__.py +3 -0
  87. docvortex/analyzers/native/office/legacy/binary.py +43 -0
  88. docvortex/analyzers/native/office/legacy/officeart.py +362 -0
  89. docvortex/analyzers/native/office/legacy/ole.py +110 -0
  90. docvortex/analyzers/native/office/limits.py +12 -0
  91. docvortex/analyzers/native/office/odf/__init__.py +3 -0
  92. docvortex/analyzers/native/office/odf/chart.py +76 -0
  93. docvortex/analyzers/native/office/odf/constants.py +70 -0
  94. docvortex/analyzers/native/office/odf/converters.py +403 -0
  95. docvortex/analyzers/native/office/odf/errors.py +18 -0
  96. docvortex/analyzers/native/office/odf/metadata.py +91 -0
  97. docvortex/analyzers/native/office/odf/models.py +176 -0
  98. docvortex/analyzers/native/office/odf/package.py +270 -0
  99. docvortex/analyzers/native/office/odf/styles.py +329 -0
  100. docvortex/analyzers/native/office/odf/table.py +469 -0
  101. docvortex/analyzers/native/office/odf/text.py +1002 -0
  102. docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
  103. docvortex/analyzers/native/office/opc.py +38 -0
  104. docvortex/analyzers/native/office/ppt/__init__.py +3 -0
  105. docvortex/analyzers/native/office/ppt/models.py +118 -0
  106. docvortex/analyzers/native/office/ppt/parser.py +1895 -0
  107. docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
  108. docvortex/analyzers/native/office/ppt/records.py +131 -0
  109. docvortex/analyzers/native/office/ppt/style_text.py +247 -0
  110. docvortex/analyzers/native/office/pptx/__init__.py +3 -0
  111. docvortex/analyzers/native/office/pptx/context.py +113 -0
  112. docvortex/analyzers/native/office/pptx/lists.py +558 -0
  113. docvortex/analyzers/native/office/pptx/main.py +20 -0
  114. docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
  115. docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
  116. docvortex/analyzers/native/office/pptx/resources.py +329 -0
  117. docvortex/analyzers/native/office/pptx/shapes.py +393 -0
  118. docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
  119. docvortex/analyzers/native/office/pptx/titles.py +178 -0
  120. docvortex/analyzers/native/office/rich_text.py +420 -0
  121. docvortex/analyzers/native/office/rtf/__init__.py +3 -0
  122. docvortex/analyzers/native/office/rtf/converter.py +708 -0
  123. docvortex/analyzers/native/office/rtf/lexer.py +213 -0
  124. docvortex/analyzers/native/office/rtf/math.py +339 -0
  125. docvortex/analyzers/native/office/rtf/models.py +185 -0
  126. docvortex/analyzers/native/office/rtf/parser.py +1553 -0
  127. docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
  128. docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
  129. docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
  130. docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
  131. docvortex/analyzers/native/office/streams.py +18 -0
  132. docvortex/analyzers/native/office/xls/__init__.py +3 -0
  133. docvortex/analyzers/native/office/xls/chart.py +132 -0
  134. docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
  135. docvortex/analyzers/native/office/xls/models.py +109 -0
  136. docvortex/analyzers/native/office/xls/number_format.py +521 -0
  137. docvortex/analyzers/native/office/xls/parser.py +1145 -0
  138. docvortex/analyzers/native/office/xls/records.py +201 -0
  139. docvortex/analyzers/native/office/xls/strings.py +205 -0
  140. docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
  141. docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
  142. docvortex/analyzers/native/office/xlsx/main.py +20 -0
  143. docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
  144. docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
  145. docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
  146. docvortex/analyzers/native/pdf/__init__.py +3 -0
  147. docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
  148. docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
  149. docvortex/analyzers/native/pdf/code_blocks.py +535 -0
  150. docvortex/analyzers/native/pdf/formulas.py +1985 -0
  151. docvortex/analyzers/native/pdf/geometry.py +281 -0
  152. docvortex/analyzers/native/pdf/graphics.py +1501 -0
  153. docvortex/analyzers/native/pdf/index_blocks.py +268 -0
  154. docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
  155. docvortex/analyzers/native/pdf/inline/common.py +112 -0
  156. docvortex/analyzers/native/pdf/inline/detection.py +590 -0
  157. docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
  158. docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
  159. docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
  160. docvortex/analyzers/native/pdf/inline/types.py +385 -0
  161. docvortex/analyzers/native/pdf/line_layout.py +1106 -0
  162. docvortex/analyzers/native/pdf/line_merging.py +1223 -0
  163. docvortex/analyzers/native/pdf/models.py +246 -0
  164. docvortex/analyzers/native/pdf/native_text.py +1004 -0
  165. docvortex/analyzers/native/pdf/pipeline.py +1390 -0
  166. docvortex/analyzers/native/pdf/script_geometry.py +636 -0
  167. docvortex/analyzers/native/pdf/shared.py +30 -0
  168. docvortex/analyzers/native/pdf/spatial_text.py +383 -0
  169. docvortex/analyzers/native/pdf/table_annotations.py +446 -0
  170. docvortex/analyzers/native/pdf/table_constants.py +48 -0
  171. docvortex/analyzers/native/pdf/table_detection.py +227 -0
  172. docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
  173. docvortex/analyzers/native/pdf/table_geometry.py +40 -0
  174. docvortex/analyzers/native/pdf/table_materialization.py +444 -0
  175. docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
  176. docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
  177. docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
  178. docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
  179. docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
  180. docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
  181. docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
  182. docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
  183. docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
  184. docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
  185. docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
  186. docvortex/analyzers/native/pdf/table_rows.py +34 -0
  187. docvortex/analyzers/native/pdf/table_rules.py +1129 -0
  188. docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
  189. docvortex/analyzers/native/pdf/tables.py +147 -0
  190. docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
  191. docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
  192. docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
  193. docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
  194. docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
  195. docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
  196. docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
  197. docvortex/analyzers/native/pdf/text_blocks.py +82 -0
  198. docvortex/analyzers/native/pdf/text_styles.py +55 -0
  199. docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
  200. docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
  201. docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
  202. docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
  203. docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
  204. docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
  205. docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
  206. docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
  207. docvortex/analyzers/native/pdf/titles.py +75 -0
  208. docvortex/analyzers/native/pdf/typography.py +19 -0
  209. docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
  210. docvortex/api.py +180 -0
  211. docvortex/assets/__init__.py +5 -0
  212. docvortex/assets/store.py +51 -0
  213. docvortex/cli.py +54 -0
  214. docvortex/codecs/__init__.py +3 -0
  215. docvortex/codecs/html/__init__.py +22 -0
  216. docvortex/codecs/html/contracts.py +236 -0
  217. docvortex/codecs/html/materializer.py +331 -0
  218. docvortex/codecs/html/parser.py +763 -0
  219. docvortex/codecs/html/resources.py +34 -0
  220. docvortex/codecs/json.py +17 -0
  221. docvortex/content/__init__.py +5 -0
  222. docvortex/content/inline.py +248 -0
  223. docvortex/content/markup/__init__.py +44 -0
  224. docvortex/content/markup/anchors.py +188 -0
  225. docvortex/content/markup/formula.py +280 -0
  226. docvortex/content/markup/projector.py +1237 -0
  227. docvortex/content/markup/styles.py +327 -0
  228. docvortex/content/mathml.py +167 -0
  229. docvortex/content/normalization.py +188 -0
  230. docvortex/content/spans.py +183 -0
  231. docvortex/content/table/__init__.py +18 -0
  232. docvortex/content/table/blocks.py +152 -0
  233. docvortex/content/table/content.py +425 -0
  234. docvortex/content/table/document.py +104 -0
  235. docvortex/content/table/html.py +399 -0
  236. docvortex/content/table/models.py +76 -0
  237. docvortex/content/table/rules.py +42 -0
  238. docvortex/content/table/structure.py +221 -0
  239. docvortex/content/tree.py +5 -0
  240. docvortex/document/__init__.py +3 -0
  241. docvortex/document/contracts.py +22 -0
  242. docvortex/document/detection.py +389 -0
  243. docvortex/document/filetypes.py +175 -0
  244. docvortex/document/page_range.py +167 -0
  245. docvortex/document/pdf/__init__.py +19 -0
  246. docvortex/document/pdf/classify.py +1138 -0
  247. docvortex/document/pdf/constants.py +132 -0
  248. docvortex/document/pdf/diagnostics.py +350 -0
  249. docvortex/document/pdf/document.py +582 -0
  250. docvortex/document/pdf/font_runtime.py +335 -0
  251. docvortex/document/pdf/geometry.py +31 -0
  252. docvortex/document/pdf/images.py +654 -0
  253. docvortex/document/pdf/native_annotations.py +367 -0
  254. docvortex/document/pdf/native_contracts.py +169 -0
  255. docvortex/document/pdf/native_coordinates.py +216 -0
  256. docvortex/document/pdf/native_lifecycle.py +16 -0
  257. docvortex/document/pdf/native_objects.py +902 -0
  258. docvortex/document/pdf/native_text_geometry.py +315 -0
  259. docvortex/document/pdf/pdfium.py +325 -0
  260. docvortex/document/pdf/raster.py +46 -0
  261. docvortex/document/pdf/text/__init__.py +62 -0
  262. docvortex/document/pdf/text/contracts.py +211 -0
  263. docvortex/document/pdf/text/extract.py +165 -0
  264. docvortex/document/pdf/text/geometry.py +16 -0
  265. docvortex/document/pdf/text/groups.py +162 -0
  266. docvortex/document/pdf/visual_geometry.py +201 -0
  267. docvortex/document/pdf/visuals.py +343 -0
  268. docvortex/document/source.py +92 -0
  269. docvortex/errors.py +21 -0
  270. docvortex/export/__init__.py +3 -0
  271. docvortex/export/bundle.py +98 -0
  272. docvortex/export/files.py +66 -0
  273. docvortex/export/middle.py +208 -0
  274. docvortex/foundation/__init__.py +3 -0
  275. docvortex/foundation/geometry.py +125 -0
  276. docvortex/foundation/hyperlink.py +65 -0
  277. docvortex/foundation/image.py +48 -0
  278. docvortex/foundation/image_encoding.py +30 -0
  279. docvortex/foundation/image_payload.py +280 -0
  280. docvortex/foundation/language.py +92 -0
  281. docvortex/foundation/platform.py +38 -0
  282. docvortex/foundation/text.py +153 -0
  283. docvortex/foundation/type_identity.py +20 -0
  284. docvortex/foundation/xml_names.py +20 -0
  285. docvortex/options.py +30 -0
  286. docvortex/postprocess/__init__.py +3 -0
  287. docvortex/postprocess/content.py +53 -0
  288. docvortex/postprocess/document.py +19 -0
  289. docvortex/postprocess/lists.py +236 -0
  290. docvortex/postprocess/page_blocks.py +214 -0
  291. docvortex/postprocess/pages.py +95 -0
  292. docvortex/postprocess/paragraphs.py +580 -0
  293. docvortex/postprocess/visual.py +715 -0
  294. docvortex/render/__init__.py +48 -0
  295. docvortex/render/_internal/__init__.py +3 -0
  296. docvortex/render/_internal/common/__init__.py +3 -0
  297. docvortex/render/_internal/common/context.py +43 -0
  298. docvortex/render/_internal/common/html_table.py +178 -0
  299. docvortex/render/_internal/common/index.py +33 -0
  300. docvortex/render/_internal/common/list_items.py +158 -0
  301. docvortex/render/_internal/common/planner.py +140 -0
  302. docvortex/render/_internal/docx/__init__.py +3 -0
  303. docvortex/render/_internal/docx/assets.py +202 -0
  304. docvortex/render/_internal/docx/inline.py +434 -0
  305. docvortex/render/_internal/docx/math.py +220 -0
  306. docvortex/render/_internal/docx/renderer.py +905 -0
  307. docvortex/render/_internal/docx/styles.py +195 -0
  308. docvortex/render/_internal/docx/table.py +442 -0
  309. docvortex/render/_internal/epub/__init__.py +5 -0
  310. docvortex/render/_internal/epub/assets.py +173 -0
  311. docvortex/render/_internal/epub/package.py +249 -0
  312. docvortex/render/_internal/epub/renderer.py +1156 -0
  313. docvortex/render/_internal/html/__init__.py +3 -0
  314. docvortex/render/_internal/html/inline.py +346 -0
  315. docvortex/render/_internal/html/renderer.py +1041 -0
  316. docvortex/render/_internal/html/sanitizer.py +478 -0
  317. docvortex/render/_internal/html/table.py +121 -0
  318. docvortex/render/_internal/latex/__init__.py +1 -0
  319. docvortex/render/_internal/latex/assets.py +85 -0
  320. docvortex/render/_internal/latex/inline.py +145 -0
  321. docvortex/render/_internal/latex/renderer.py +506 -0
  322. docvortex/render/_internal/latex/table.py +347 -0
  323. docvortex/render/_internal/markdown/__init__.py +3 -0
  324. docvortex/render/_internal/markdown/assets.py +78 -0
  325. docvortex/render/_internal/markdown/blocks.py +635 -0
  326. docvortex/render/_internal/markdown/escaping.py +51 -0
  327. docvortex/render/_internal/markdown/inline.py +260 -0
  328. docvortex/render/_internal/markdown/renderer.py +93 -0
  329. docvortex/render/_internal/markdown/table.py +281 -0
  330. docvortex/render/_internal/pdf/__init__.py +3 -0
  331. docvortex/render/_internal/pdf/assets.py +197 -0
  332. docvortex/render/_internal/pdf/formula.py +417 -0
  333. docvortex/render/_internal/pdf/inline.py +343 -0
  334. docvortex/render/_internal/pdf/renderer.py +734 -0
  335. docvortex/render/_internal/pdf/styles.py +206 -0
  336. docvortex/render/_internal/pdf/table.py +272 -0
  337. docvortex/render/_internal/structured_content/__init__.py +3 -0
  338. docvortex/render/_internal/structured_content/renderer.py +193 -0
  339. docvortex/render/api.py +199 -0
  340. docvortex/render/contracts.py +205 -0
  341. docvortex/render/docx.py +38 -0
  342. docvortex/render/epub.py +35 -0
  343. docvortex/render/fragments.py +70 -0
  344. docvortex/render/html.py +29 -0
  345. docvortex/render/latex.py +24 -0
  346. docvortex/render/markdown.py +50 -0
  347. docvortex/render/pdf.py +25 -0
  348. docvortex/render/structured_content.py +23 -0
  349. docvortex/resources/epub/docvortex.css +91 -0
  350. docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
  351. docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
  352. docvortex/resources/fonts/NOTICE +190 -0
  353. docvortex/resources/fonts/manifest.json +9 -0
  354. docvortex/resources/html/docvortex.css +601 -0
  355. docvortex/resources/html/docvortex.min.css +1 -0
  356. docvortex/result.py +91 -0
  357. docvortex/schema.py +1141 -0
  358. docvortex/version.py +3 -0
  359. docvortex-0.2.1.dist-info/METADATA +193 -0
  360. docvortex-0.2.1.dist-info/RECORD +364 -0
  361. docvortex-0.2.1.dist-info/WHEEL +5 -0
  362. docvortex-0.2.1.dist-info/entry_points.txt +2 -0
  363. docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
  364. docvortex-0.2.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,1931 @@
1
+ """基于 PDF 横竖线与矩形路径恢复原子网格和合并单元格。"""
2
+
3
+ from __future__ import annotations
4
+
5
+ import math
6
+ import statistics
7
+ from dataclasses import dataclass
8
+ from typing import Any
9
+
10
+ from .candidate import GridCellSpec, build_candidate
11
+ from .contracts import NativeTableCandidate, NativeTableInput, NativeTableText
12
+ from .geometry import (
13
+ bbox_area,
14
+ bbox_intersection,
15
+ clamp,
16
+ cluster_positions,
17
+ covered_interval_ratio,
18
+ normalize_angle,
19
+ normalize_bbox,
20
+ page_bbox_to_table_local,
21
+ rotate_local_bbox,
22
+ table_local_size,
23
+ )
24
+
25
+ MAX_PRIMITIVES_PER_TABLE = 5000
26
+ MAX_TRACKS_PER_AXIS = 200
27
+ MAX_ATOMIC_CELLS = 10000
28
+ SEPARATOR_COVERAGE_THRESHOLD = 0.80
29
+ MAX_TRACK_HYPOTHESES = 8
30
+
31
+
32
+ @dataclass(frozen=True, slots=True)
33
+ class _MergedRule:
34
+ """保存吸附并连接后的局部单轴线段。"""
35
+
36
+ orientation: str
37
+ coordinate: float
38
+ start: float
39
+ end: float
40
+
41
+
42
+ @dataclass(frozen=True, slots=True)
43
+ class _CanonicalTrack:
44
+ """保存折叠后的规范轨道及其全部原始坐标别名。"""
45
+
46
+ coordinate: float
47
+ aliases: tuple[float, ...]
48
+
49
+
50
+ @dataclass(frozen=True, slots=True)
51
+ class _PhysicalRowEvidence:
52
+ """保存一个原子行由 drawing 独立验证的边界可靠度。"""
53
+
54
+ row: int
55
+ top_coverage: float
56
+ bottom_coverage: float
57
+ left_coverage: float
58
+ right_coverage: float
59
+ height_ratio: float
60
+ glyph_crossing: bool
61
+
62
+ @property
63
+ def reliability(self) -> float:
64
+ """返回该行所有强物理条件中的最小可靠度。"""
65
+
66
+ if self.glyph_crossing:
67
+ return 0.0
68
+ return min(
69
+ self.top_coverage,
70
+ self.bottom_coverage,
71
+ self.left_coverage,
72
+ self.right_coverage,
73
+ self.height_ratio,
74
+ )
75
+
76
+ @property
77
+ def verified(self) -> bool:
78
+ """判断该行能否脱离文本占用独立证明结构存在。"""
79
+
80
+ return self.reliability >= SEPARATOR_COVERAGE_THRESHOLD
81
+
82
+
83
+ @dataclass(frozen=True, slots=True)
84
+ class _SingleRowEvidence:
85
+ """保存单物理行网格的全部外框和纵向隔断可靠度。"""
86
+
87
+ top_coverage: float
88
+ bottom_coverage: float
89
+ vertical_coverages: tuple[float, ...]
90
+ height_ratio: float
91
+ glyph_crossing: bool
92
+
93
+ @property
94
+ def reliability(self) -> float:
95
+ """返回单行网格所有不可替代物理证据中的最小值。"""
96
+
97
+ if self.glyph_crossing or not self.vertical_coverages:
98
+ return 0.0
99
+ return min(
100
+ self.top_coverage,
101
+ self.bottom_coverage,
102
+ min(self.vertical_coverages),
103
+ self.height_ratio,
104
+ )
105
+
106
+ @property
107
+ def verified(self) -> bool:
108
+ """判断单行网格是否可脱离文本对齐独立验证。"""
109
+
110
+ return self.reliability >= SEPARATOR_COVERAGE_THRESHOLD
111
+
112
+ @property
113
+ def confidence(self) -> float:
114
+ """把通过八成物理硬门的覆盖率校准到 verified 分数区间。"""
115
+
116
+ if not self.verified:
117
+ return 0.0
118
+ return min(
119
+ 1.0,
120
+ 0.95 + 0.05 * (self.reliability - SEPARATOR_COVERAGE_THRESHOLD) / (1.0 - SEPARATOR_COVERAGE_THRESHOLD),
121
+ )
122
+
123
+
124
+ @dataclass(frozen=True, slots=True)
125
+ class _SingleColumnEvidence:
126
+ """保存多行单列表单的横向边界和左右外框可靠度。"""
127
+
128
+ horizontal_coverages: tuple[float, ...]
129
+ left_coverage: float
130
+ right_coverage: float
131
+ minimum_height_ratio: float
132
+ glyph_crossing: bool
133
+
134
+ @property
135
+ def reliability(self) -> float:
136
+ """返回单列表单所有强物理条件中的最小可靠度。"""
137
+
138
+ if self.glyph_crossing or not self.horizontal_coverages:
139
+ return 0.0
140
+ return min(
141
+ min(self.horizontal_coverages),
142
+ self.left_coverage,
143
+ self.right_coverage,
144
+ self.minimum_height_ratio,
145
+ )
146
+
147
+ @property
148
+ def verified(self) -> bool:
149
+ """判断单列表单是否具备不依赖文本对齐的完整线框证据。"""
150
+
151
+ return self.reliability >= SEPARATOR_COVERAGE_THRESHOLD
152
+
153
+ @property
154
+ def confidence(self) -> float:
155
+ """把通过八成物理硬门的可靠度校准到 verified 分数区间。"""
156
+
157
+ if not self.verified:
158
+ return 0.0
159
+ return min(
160
+ 1.0,
161
+ 0.95 + 0.05 * (self.reliability - SEPARATOR_COVERAGE_THRESHOLD) / (1.0 - SEPARATOR_COVERAGE_THRESHOLD),
162
+ )
163
+
164
+
165
+ class _UnionFind:
166
+ """维护缺失内部隔断连接的原子网格并查集。"""
167
+
168
+ def __init__(self, size: int) -> None:
169
+ """为固定数量原子格初始化各自独立的集合。"""
170
+
171
+ self._parents = list(range(size))
172
+
173
+ def find(self, index: int) -> int:
174
+ """返回原子格根节点并执行路径压缩。"""
175
+
176
+ parent = self._parents[index]
177
+ if parent != index:
178
+ self._parents[index] = self.find(parent)
179
+ return self._parents[index]
180
+
181
+ def union(self, first: int, second: int) -> None:
182
+ """合并两个原子格所属集合。"""
183
+
184
+ first_root = self.find(first)
185
+ second_root = self.find(second)
186
+ if first_root != second_root:
187
+ self._parents[second_root] = first_root
188
+
189
+
190
+ def _drawing_bbox_to_table_local(
191
+ rule_bbox: tuple[float, float, float, float],
192
+ table_bbox: tuple[float, float, float, float],
193
+ angle: int,
194
+ evidence_halo: float,
195
+ ) -> tuple[float, float, float, float] | None:
196
+ """在不扩大字符区域的前提下,把邻近外框 drawing 转到表格局部坐标。"""
197
+
198
+ clipped = bbox_intersection(rule_bbox, table_bbox)
199
+ if clipped is None and evidence_halo > 0:
200
+ expanded_bbox = (
201
+ table_bbox[0] - evidence_halo,
202
+ table_bbox[1] - evidence_halo,
203
+ table_bbox[2] + evidence_halo,
204
+ table_bbox[3] + evidence_halo,
205
+ )
206
+ clipped = bbox_intersection(rule_bbox, expanded_bbox)
207
+ if clipped is None:
208
+ return None
209
+ width = table_bbox[2] - table_bbox[0]
210
+ height = table_bbox[3] - table_bbox[1]
211
+ relative = (
212
+ clipped[0] - table_bbox[0],
213
+ clipped[1] - table_bbox[1],
214
+ clipped[2] - table_bbox[0],
215
+ clipped[3] - table_bbox[1],
216
+ )
217
+ return rotate_local_bbox(relative, width, height, angle)
218
+
219
+
220
+ def _local_rule_fragments(
221
+ table_input: NativeTableInput,
222
+ snap_tolerance: float,
223
+ *,
224
+ include_drawing: bool = True,
225
+ include_rectangles: bool = True,
226
+ evidence_halo: float = 0.0,
227
+ ) -> list[_MergedRule]:
228
+ """按来源裁剪 drawing/矩形,并转换为局部轴线片段。"""
229
+
230
+ table_bbox = normalize_bbox(table_input.table_bbox)
231
+ if table_bbox is None:
232
+ return []
233
+ angle = normalize_angle(table_input.angle)
234
+ fragments: list[_MergedRule] = []
235
+ for rule in table_input.drawing_lines if include_drawing else ():
236
+ rule_bbox = normalize_bbox(rule.bbox)
237
+ if rule_bbox is None:
238
+ continue
239
+ local_bbox = _drawing_bbox_to_table_local(
240
+ rule_bbox,
241
+ table_bbox,
242
+ angle,
243
+ evidence_halo,
244
+ )
245
+ if local_bbox is None:
246
+ continue
247
+ width = local_bbox[2] - local_bbox[0]
248
+ height = local_bbox[3] - local_bbox[1]
249
+ orientation = "horizontal" if width >= height else "vertical"
250
+ if orientation == "horizontal" and width >= max(1.0, 2.0 * snap_tolerance):
251
+ fragments.append(
252
+ _MergedRule(
253
+ orientation="horizontal",
254
+ coordinate=(local_bbox[1] + local_bbox[3]) / 2.0,
255
+ start=local_bbox[0],
256
+ end=local_bbox[2],
257
+ )
258
+ )
259
+ elif orientation == "vertical" and height >= max(1.0, 2.0 * snap_tolerance):
260
+ fragments.append(
261
+ _MergedRule(
262
+ orientation="vertical",
263
+ coordinate=(local_bbox[0] + local_bbox[2]) / 2.0,
264
+ start=local_bbox[1],
265
+ end=local_bbox[3],
266
+ )
267
+ )
268
+
269
+ local_width, local_height = table_local_size(table_bbox, angle)
270
+ table_area = local_width * local_height
271
+ for rectangle in table_input.rectangles if include_rectangles else ():
272
+ if rectangle.segment_count != 5 or not (rectangle.fill_visible or rectangle.stroke_visible):
273
+ continue
274
+ rectangle_bbox = normalize_bbox(rectangle.bbox)
275
+ if rectangle_bbox is None:
276
+ continue
277
+ local_bbox = page_bbox_to_table_local(rectangle_bbox, table_bbox, angle)
278
+ if local_bbox is None:
279
+ continue
280
+ rect_width = local_bbox[2] - local_bbox[0]
281
+ rect_height = local_bbox[3] - local_bbox[1]
282
+ area_ratio = bbox_area(local_bbox) / table_area if table_area > 0 else 0.0
283
+ if rect_width <= snap_tolerance or rect_height <= snap_tolerance or area_ratio >= 0.85:
284
+ continue
285
+ fragments.extend(
286
+ [
287
+ _MergedRule("horizontal", local_bbox[1], local_bbox[0], local_bbox[2]),
288
+ _MergedRule("horizontal", local_bbox[3], local_bbox[0], local_bbox[2]),
289
+ _MergedRule("vertical", local_bbox[0], local_bbox[1], local_bbox[3]),
290
+ _MergedRule("vertical", local_bbox[2], local_bbox[1], local_bbox[3]),
291
+ ]
292
+ )
293
+ return fragments
294
+
295
+
296
+ def _merge_rule_fragments(
297
+ fragments: list[_MergedRule],
298
+ snap_tolerance: float,
299
+ join_gap: float,
300
+ ) -> list[_MergedRule]:
301
+ """按方向和轴坐标吸附线段,再连接小间隙共线片段。"""
302
+
303
+ output: list[_MergedRule] = []
304
+ for orientation in ("horizontal", "vertical"):
305
+ oriented = [fragment for fragment in fragments if fragment.orientation == orientation]
306
+ coordinates = cluster_positions(
307
+ (fragment.coordinate for fragment in oriented),
308
+ snap_tolerance,
309
+ )
310
+ for coordinate in coordinates:
311
+ intervals = sorted(
312
+ (
313
+ min(fragment.start, fragment.end),
314
+ max(fragment.start, fragment.end),
315
+ )
316
+ for fragment in oriented
317
+ if abs(fragment.coordinate - coordinate) <= snap_tolerance
318
+ )
319
+ if not intervals:
320
+ continue
321
+ current_start, current_end = intervals[0]
322
+ for start, end in intervals[1:]:
323
+ if start <= current_end + join_gap:
324
+ current_end = max(current_end, end)
325
+ continue
326
+ output.append(_MergedRule(orientation, coordinate, current_start, current_end))
327
+ current_start, current_end = start, end
328
+ output.append(_MergedRule(orientation, coordinate, current_start, current_end))
329
+ return output
330
+
331
+
332
+ def _repeated_long_rule_endpoints(
333
+ rules: list[_MergedRule],
334
+ axis_extent: float,
335
+ snap_tolerance: float,
336
+ ) -> list[float]:
337
+ """仅保留重复长线端点或接近表格外缘的外围轨道证据。"""
338
+
339
+ if not rules:
340
+ return []
341
+ maximum_length = max(rule.end - rule.start for rule in rules)
342
+ long_rules = [rule for rule in rules if rule.end - rule.start >= 0.80 * maximum_length]
343
+ required_support = max(2, math.ceil(0.50 * len(long_rules)))
344
+ positions = cluster_positions(
345
+ (endpoint for rule in long_rules for endpoint in (rule.start, rule.end)),
346
+ snap_tolerance,
347
+ )
348
+ output: list[float] = []
349
+ for position in positions:
350
+ support = sum(
351
+ min(
352
+ abs(rule.start - position),
353
+ abs(rule.end - position),
354
+ )
355
+ <= snap_tolerance
356
+ for rule in long_rules
357
+ )
358
+ if support >= required_support or position <= 2.0 * snap_tolerance or axis_extent - position <= 2.0 * snap_tolerance:
359
+ output.append(position)
360
+ return output
361
+
362
+
363
+ def _infer_grid_tracks(
364
+ rules: list[_MergedRule],
365
+ snap_tolerance: float,
366
+ width: float,
367
+ height: float,
368
+ *,
369
+ prune_unsupported_horizontal: bool = False,
370
+ ) -> tuple[list[float], list[float], list[float]]:
371
+ """融合物理轴线与受重复门约束的端点,恢复开放外框轨道。"""
372
+
373
+ horizontal = [rule for rule in rules if rule.orientation == "horizontal"]
374
+ vertical = [rule for rule in rules if rule.orientation == "vertical"]
375
+ x_values = [rule.coordinate for rule in vertical]
376
+ x_values.extend(
377
+ _repeated_long_rule_endpoints(
378
+ horizontal,
379
+ width,
380
+ snap_tolerance,
381
+ )
382
+ )
383
+ provisional_x_tracks = cluster_positions(x_values, snap_tolerance)
384
+ removed_horizontal_tracks: list[float] = []
385
+ supported_horizontal = horizontal
386
+ if prune_unsupported_horizontal:
387
+ supported_horizontal, removed_horizontal_tracks = _filter_horizontal_track_creators(
388
+ horizontal,
389
+ provisional_x_tracks,
390
+ width,
391
+ snap_tolerance,
392
+ )
393
+ y_values = [rule.coordinate for rule in supported_horizontal]
394
+ y_values.extend(
395
+ _repeated_long_rule_endpoints(
396
+ vertical,
397
+ height,
398
+ snap_tolerance,
399
+ )
400
+ )
401
+ return (
402
+ provisional_x_tracks,
403
+ cluster_positions(y_values, snap_tolerance),
404
+ removed_horizontal_tracks,
405
+ )
406
+
407
+
408
+ def _filter_horizontal_track_creators(
409
+ rules: list[_MergedRule],
410
+ x_tracks: list[float],
411
+ width: float,
412
+ snap_tolerance: float,
413
+ ) -> tuple[list[_MergedRule], list[float]]:
414
+ """剔除完全缩进在单元格内、不能形成真实横向轨道的装饰短线。"""
415
+
416
+ if len(x_tracks) < 2:
417
+ return rules, []
418
+ supported_coordinates: set[float] = set()
419
+ removed_coordinates: list[float] = []
420
+ for coordinate in cluster_positions(
421
+ (rule.coordinate for rule in rules),
422
+ snap_tolerance,
423
+ ):
424
+ coordinate_rules = [rule for rule in rules if abs(rule.coordinate - coordinate) <= snap_tolerance]
425
+ intervals = [(rule.start, rule.end) for rule in coordinate_rules]
426
+ full_width = covered_interval_ratio(intervals, 0.0, width)
427
+ band_supported = False
428
+ for left, right in zip(x_tracks, x_tracks[1:]):
429
+ if covered_interval_ratio(intervals, left, right) < SEPARATOR_COVERAGE_THRESHOLD:
430
+ continue
431
+ touches_left = any(
432
+ rule.start <= left + snap_tolerance and rule.end >= left - snap_tolerance for rule in coordinate_rules
433
+ )
434
+ touches_right = any(
435
+ rule.start <= right + snap_tolerance and rule.end >= right - snap_tolerance for rule in coordinate_rules
436
+ )
437
+ if touches_left and touches_right:
438
+ band_supported = True
439
+ break
440
+ if full_width >= SEPARATOR_COVERAGE_THRESHOLD or band_supported:
441
+ supported_coordinates.add(coordinate)
442
+ else:
443
+ removed_coordinates.append(coordinate)
444
+ return (
445
+ [
446
+ rule
447
+ for rule in rules
448
+ if any(abs(rule.coordinate - coordinate) <= snap_tolerance for coordinate in supported_coordinates)
449
+ ],
450
+ removed_coordinates,
451
+ )
452
+
453
+
454
+ def _rule_indices_for_track(
455
+ tracks_rules: list[_MergedRule],
456
+ orientation: str,
457
+ track: _CanonicalTrack,
458
+ snap_tolerance: float,
459
+ ) -> set[int]:
460
+ """返回能够归属指定轨道的全部物理线索引。"""
461
+
462
+ return {
463
+ index
464
+ for index, rule in enumerate(tracks_rules)
465
+ if rule.orientation == orientation and any(abs(rule.coordinate - alias) <= snap_tolerance for alias in track.aliases)
466
+ }
467
+
468
+
469
+ def _collapse_outer_duplicate_tracks(
470
+ tracks: list[_CanonicalTrack],
471
+ glyph_centers: list[float],
472
+ threshold: float,
473
+ extent: float,
474
+ rules: list[_MergedRule] | None,
475
+ orientation: str,
476
+ separator_extent: float,
477
+ snap_tolerance: float,
478
+ collapse_leading_edge: bool,
479
+ ) -> tuple[list[_CanonicalTrack], int, bool]:
480
+ """折叠外缘同一物理描边产生的重复轨,并保留独立双边界。"""
481
+
482
+ if rules is None or not orientation or len(tracks) < 2:
483
+ return tracks, 0, False
484
+ collapsed_count = 0
485
+ while len(tracks) >= 2:
486
+ candidate_indices = [len(tracks) - 2]
487
+ if collapse_leading_edge:
488
+ candidate_indices.insert(0, 0)
489
+ collapse_index: int | None = None
490
+ for index in candidate_indices:
491
+ left, right = tracks[index], tracks[index + 1]
492
+ if right.coordinate - left.coordinate > threshold or any(
493
+ left.coordinate < center < right.coordinate for center in glyph_centers
494
+ ):
495
+ continue
496
+ near_left_edge = right.coordinate <= max(
497
+ threshold,
498
+ 2.0 * snap_tolerance,
499
+ )
500
+ near_right_edge = extent - left.coordinate <= max(
501
+ threshold,
502
+ 2.0 * snap_tolerance,
503
+ )
504
+ if not (near_left_edge or near_right_edge):
505
+ continue
506
+ left_rules = _rule_indices_for_track(
507
+ rules,
508
+ orientation,
509
+ left,
510
+ snap_tolerance,
511
+ )
512
+ right_rules = _rule_indices_for_track(
513
+ rules,
514
+ orientation,
515
+ right,
516
+ snap_tolerance,
517
+ )
518
+ if left_rules and right_rules and left_rules.isdisjoint(right_rules):
519
+ continue
520
+ combined = _CanonicalTrack(
521
+ coordinate=float(
522
+ statistics.median(
523
+ (*left.aliases, *right.aliases),
524
+ )
525
+ ),
526
+ aliases=tuple(sorted({*left.aliases, *right.aliases})),
527
+ )
528
+ if (
529
+ combined.aliases[-1] - combined.aliases[0] > threshold
530
+ or _separator_coverage_for_track(
531
+ rules,
532
+ orientation,
533
+ combined,
534
+ 0.0,
535
+ separator_extent,
536
+ snap_tolerance,
537
+ )
538
+ < SEPARATOR_COVERAGE_THRESHOLD
539
+ ):
540
+ continue
541
+ collapse_index = index
542
+ break
543
+ if collapse_index is None:
544
+ break
545
+ left, right = tracks[collapse_index : collapse_index + 2]
546
+ aliases = tuple(sorted({*left.aliases, *right.aliases}))
547
+ if aliases[-1] - aliases[0] > threshold:
548
+ return tracks, collapsed_count, True
549
+ tracks[collapse_index : collapse_index + 2] = [
550
+ _CanonicalTrack(
551
+ coordinate=float(statistics.median(aliases)),
552
+ aliases=aliases,
553
+ )
554
+ ]
555
+ collapsed_count += 1
556
+ return tracks, collapsed_count, False
557
+
558
+
559
+ def _canonicalize_axis_tracks(
560
+ positions: list[float],
561
+ glyph_centers: list[float],
562
+ threshold: float,
563
+ extent: float,
564
+ evidence_halo: float,
565
+ minimum_track_count: int,
566
+ *,
567
+ rules: list[_MergedRule] | None = None,
568
+ orientation: str = "",
569
+ separator_extent: float = 0.0,
570
+ snap_tolerance: float = 0.0,
571
+ preserve_double_boundary: bool = False,
572
+ collapse_narrow_bands: bool = True,
573
+ collapse_leading_edge: bool = True,
574
+ ) -> tuple[list[_CanonicalTrack], bool, int]:
575
+ """吸附外缘并折叠无字形占用的窄带,同时保留全部原始别名。"""
576
+
577
+ tracks = [
578
+ _CanonicalTrack(
579
+ coordinate=coordinate,
580
+ aliases=(coordinate,),
581
+ )
582
+ for coordinate in positions
583
+ ]
584
+ for edge in (0.0, extent):
585
+ halo_indices = [index for index, track in enumerate(tracks) if abs(track.coordinate - edge) <= evidence_halo]
586
+ if not halo_indices:
587
+ continue
588
+ closest_coordinate = min(
589
+ (tracks[index].coordinate for index in halo_indices),
590
+ key=lambda coordinate: abs(coordinate - edge),
591
+ )
592
+ edge_indices = [index for index in halo_indices if abs(tracks[index].coordinate - closest_coordinate) <= snap_tolerance]
593
+ aliases = tuple(
594
+ sorted(
595
+ {
596
+ edge,
597
+ *(alias for index in edge_indices for alias in tracks[index].aliases),
598
+ }
599
+ )
600
+ )
601
+ first_index = edge_indices[0]
602
+ tracks[first_index : edge_indices[-1] + 1] = [
603
+ _CanonicalTrack(
604
+ coordinate=edge,
605
+ aliases=aliases,
606
+ )
607
+ ]
608
+
609
+ tracks, outer_collapse_count, outer_alias_conflict = _collapse_outer_duplicate_tracks(
610
+ tracks,
611
+ glyph_centers,
612
+ threshold,
613
+ extent,
614
+ rules,
615
+ orientation,
616
+ separator_extent,
617
+ snap_tolerance,
618
+ collapse_leading_edge,
619
+ )
620
+ if outer_alias_conflict:
621
+ return tracks, True, outer_collapse_count
622
+
623
+ while collapse_narrow_bands and len(tracks) > minimum_track_count:
624
+ collapse_index = next(
625
+ (
626
+ index
627
+ for index, (left, right) in enumerate(zip(tracks, tracks[1:]))
628
+ if right.coordinate - left.coordinate <= threshold
629
+ and not any(left.coordinate < center < right.coordinate for center in glyph_centers)
630
+ and not (
631
+ preserve_double_boundary
632
+ and rules is not None
633
+ and _separator_coverage_for_track(
634
+ rules,
635
+ orientation,
636
+ left,
637
+ 0.0,
638
+ separator_extent,
639
+ snap_tolerance,
640
+ )
641
+ > 0.0
642
+ and _separator_coverage_for_track(
643
+ rules,
644
+ orientation,
645
+ right,
646
+ 0.0,
647
+ separator_extent,
648
+ snap_tolerance,
649
+ )
650
+ > 0.0
651
+ )
652
+ ),
653
+ None,
654
+ )
655
+ if collapse_index is None:
656
+ break
657
+ aliases = tuple(
658
+ sorted(
659
+ {
660
+ *tracks[collapse_index].aliases,
661
+ *tracks[collapse_index + 1].aliases,
662
+ }
663
+ )
664
+ )
665
+ if aliases[-1] - aliases[0] > threshold:
666
+ return tracks, True, outer_collapse_count
667
+ tracks[collapse_index : collapse_index + 2] = [
668
+ _CanonicalTrack(
669
+ coordinate=float(statistics.median(aliases)),
670
+ aliases=aliases,
671
+ )
672
+ ]
673
+ return tracks, False, outer_collapse_count
674
+
675
+
676
+ def _canonical_track_coordinates(
677
+ tracks: list[_CanonicalTrack],
678
+ ) -> list[float]:
679
+ """提取规范轨道数值坐标,供网格 bbox 和稳定度计算使用。"""
680
+
681
+ return [track.coordinate for track in tracks]
682
+
683
+
684
+ def _canonical_tracks_are_unique(
685
+ tracks: list[_CanonicalTrack],
686
+ rules: list[_MergedRule],
687
+ orientation: str,
688
+ snap_tolerance: float,
689
+ ) -> bool:
690
+ """校验每条物理线最多只能归属一个规范轨道别名集合。"""
691
+
692
+ for rule in rules:
693
+ if rule.orientation != orientation:
694
+ continue
695
+ matching_tracks = sum(
696
+ any(abs(rule.coordinate - alias) <= snap_tolerance for alias in track.aliases) for track in tracks
697
+ )
698
+ if matching_tracks > 1:
699
+ return False
700
+ return True
701
+
702
+
703
+ def _separator_coverage_for_track(
704
+ rules: list[_MergedRule],
705
+ orientation: str,
706
+ track: _CanonicalTrack,
707
+ start: float,
708
+ end: float,
709
+ snap_tolerance: float,
710
+ ) -> float:
711
+ """按规范轨道全部 alias 合并计算 separator 覆盖率。"""
712
+
713
+ intervals = [
714
+ (rule.start, rule.end)
715
+ for rule in rules
716
+ if rule.orientation == orientation and any(abs(rule.coordinate - alias) <= snap_tolerance for alias in track.aliases)
717
+ ]
718
+ return covered_interval_ratio(intervals, start, end)
719
+
720
+
721
+ def _rect_lattice_is_repeated(
722
+ rules: list[_MergedRule],
723
+ x_tracks: list[float],
724
+ y_tracks: list[float],
725
+ snap_tolerance: float,
726
+ ) -> bool:
727
+ """要求矩形边缘在二维晶格中重复且覆盖至少八成理论边界。"""
728
+
729
+ rows = len(y_tracks) - 1
730
+ cols = len(x_tracks) - 1
731
+ if rows < 2 or cols < 2:
732
+ return False
733
+ segment_present: list[bool] = []
734
+ vertical_repetitions: list[int] = []
735
+ for boundary_index, coordinate in enumerate(x_tracks):
736
+ repetitions = 0
737
+ for top, bottom in zip(y_tracks, y_tracks[1:]):
738
+ present = (
739
+ _separator_coverage(
740
+ rules,
741
+ "vertical",
742
+ coordinate,
743
+ top,
744
+ bottom,
745
+ snap_tolerance,
746
+ )
747
+ >= SEPARATOR_COVERAGE_THRESHOLD
748
+ )
749
+ segment_present.append(present)
750
+ repetitions += int(present)
751
+ if 0 < boundary_index < len(x_tracks) - 1:
752
+ vertical_repetitions.append(repetitions)
753
+ horizontal_repetitions: list[int] = []
754
+ for boundary_index, coordinate in enumerate(y_tracks):
755
+ repetitions = 0
756
+ for left, right in zip(x_tracks, x_tracks[1:]):
757
+ present = (
758
+ _separator_coverage(
759
+ rules,
760
+ "horizontal",
761
+ coordinate,
762
+ left,
763
+ right,
764
+ snap_tolerance,
765
+ )
766
+ >= SEPARATOR_COVERAGE_THRESHOLD
767
+ )
768
+ segment_present.append(present)
769
+ repetitions += int(present)
770
+ if 0 < boundary_index < len(y_tracks) - 1:
771
+ horizontal_repetitions.append(repetitions)
772
+ coverage = sum(segment_present) / len(segment_present) if segment_present else 0.0
773
+ return (
774
+ coverage >= SEPARATOR_COVERAGE_THRESHOLD
775
+ and any(count >= 2 for count in vertical_repetitions)
776
+ and any(count >= 2 for count in horizontal_repetitions)
777
+ )
778
+
779
+
780
+ def _separator_coverage(
781
+ rules: list[_MergedRule],
782
+ orientation: str,
783
+ coordinate: float,
784
+ start: float,
785
+ end: float,
786
+ snap_tolerance: float,
787
+ ) -> float:
788
+ """计算指定潜在隔断被同轴物理线段覆盖的比例。"""
789
+
790
+ intervals = [
791
+ (rule.start, rule.end)
792
+ for rule in rules
793
+ if rule.orientation == orientation and abs(rule.coordinate - coordinate) <= snap_tolerance
794
+ ]
795
+ return covered_interval_ratio(intervals, start, end)
796
+
797
+
798
+ def _grid_index(row: int, col: int, cols: int) -> int:
799
+ """把二维原子格坐标转换为并查集线性索引。"""
800
+
801
+ return row * cols + col
802
+
803
+
804
+ def _build_component_specs(
805
+ union_find: _UnionFind,
806
+ rows: int,
807
+ cols: int,
808
+ x_tracks: list[float],
809
+ y_tracks: list[float],
810
+ ) -> tuple[GridCellSpec, ...] | None:
811
+ """把原子格连通分量转成矩形逻辑单元格,非矩形分量整体拒绝。"""
812
+
813
+ components: dict[int, list[tuple[int, int]]] = {}
814
+ for row in range(rows):
815
+ for col in range(cols):
816
+ root = union_find.find(_grid_index(row, col, cols))
817
+ components.setdefault(root, []).append((row, col))
818
+ specs: list[GridCellSpec] = []
819
+ for positions in components.values():
820
+ row_values = [item[0] for item in positions]
821
+ col_values = [item[1] for item in positions]
822
+ min_row, max_row = min(row_values), max(row_values)
823
+ min_col, max_col = min(col_values), max(col_values)
824
+ expected_size = (max_row - min_row + 1) * (max_col - min_col + 1)
825
+ if len(positions) != expected_size:
826
+ return None
827
+ specs.append(
828
+ GridCellSpec(
829
+ row=min_row,
830
+ col=min_col,
831
+ rowspan=max_row - min_row + 1,
832
+ colspan=max_col - min_col + 1,
833
+ bbox=(
834
+ x_tracks[min_col],
835
+ y_tracks[min_row],
836
+ x_tracks[max_col + 1],
837
+ y_tracks[max_row + 1],
838
+ ),
839
+ )
840
+ )
841
+ return tuple(sorted(specs, key=lambda item: (item.row, item.col)))
842
+
843
+
844
+ def _occupied_text_rows(
845
+ text: NativeTableText,
846
+ y_tracks: list[float],
847
+ ) -> set[int]:
848
+ """返回至少包含一个视觉文本行中心的物理行索引。"""
849
+
850
+ occupied_rows: set[int] = set()
851
+ for row in text.rows:
852
+ center_y = (row.bbox[1] + row.bbox[3]) / 2.0
853
+ for row_index, (top, bottom) in enumerate(zip(y_tracks, y_tracks[1:])):
854
+ if top <= center_y <= bottom:
855
+ occupied_rows.add(row_index)
856
+ break
857
+ return occupied_rows
858
+
859
+
860
+ def _line_grid_row_evidence(
861
+ rules: list[_MergedRule],
862
+ x_tracks: list[_CanonicalTrack],
863
+ y_tracks: list[_CanonicalTrack],
864
+ text: NativeTableText,
865
+ snap_tolerance: float,
866
+ minimum_row_height: float,
867
+ local_width: float,
868
+ ) -> tuple[_PhysicalRowEvidence, ...]:
869
+ """计算 line-grid 每个原子行的独立物理封闭证据。"""
870
+
871
+ left = x_tracks[0].coordinate
872
+ right = x_tracks[-1].coordinate
873
+ evidence: list[_PhysicalRowEvidence] = []
874
+ for row_index, (top_track, bottom_track) in enumerate(zip(y_tracks, y_tracks[1:])):
875
+ top = top_track.coordinate
876
+ bottom = bottom_track.coordinate
877
+ top_coverage = _separator_coverage_for_track(
878
+ rules,
879
+ "horizontal",
880
+ top_track,
881
+ left,
882
+ right,
883
+ snap_tolerance,
884
+ )
885
+ bottom_coverage = _separator_coverage_for_track(
886
+ rules,
887
+ "horizontal",
888
+ bottom_track,
889
+ left,
890
+ right,
891
+ snap_tolerance,
892
+ )
893
+ left_coverage = _separator_coverage_for_track(
894
+ rules,
895
+ "vertical",
896
+ x_tracks[0],
897
+ top,
898
+ bottom,
899
+ snap_tolerance,
900
+ )
901
+ right_coverage = _separator_coverage_for_track(
902
+ rules,
903
+ "vertical",
904
+ x_tracks[-1],
905
+ top,
906
+ bottom,
907
+ snap_tolerance,
908
+ )
909
+ endpoint_support = min(top_coverage, bottom_coverage)
910
+ if left <= 2.0 * snap_tolerance:
911
+ left_coverage = max(left_coverage, endpoint_support)
912
+ if local_width - right <= 2.0 * snap_tolerance:
913
+ right_coverage = max(right_coverage, endpoint_support)
914
+ glyph_crossing = any(
915
+ glyph.bbox[1] < bottom and glyph.bbox[3] > top and not (top <= (glyph.bbox[1] + glyph.bbox[3]) / 2.0 <= bottom)
916
+ for glyph in text.glyphs
917
+ )
918
+ evidence.append(
919
+ _PhysicalRowEvidence(
920
+ row=row_index,
921
+ top_coverage=top_coverage,
922
+ bottom_coverage=bottom_coverage,
923
+ left_coverage=left_coverage,
924
+ right_coverage=right_coverage,
925
+ height_ratio=min(
926
+ 1.0,
927
+ (bottom - top) / max(minimum_row_height, 0.1),
928
+ ),
929
+ glyph_crossing=glyph_crossing,
930
+ )
931
+ )
932
+ return tuple(evidence)
933
+
934
+
935
+ def _single_row_line_grid_evidence(
936
+ rules: list[_MergedRule],
937
+ x_tracks: list[_CanonicalTrack],
938
+ y_tracks: list[_CanonicalTrack],
939
+ text: NativeTableText,
940
+ snap_tolerance: float,
941
+ minimum_row_height: float,
942
+ ) -> _SingleRowEvidence:
943
+ """校验单物理行候选的上下外框及每一条纵向边界。"""
944
+
945
+ top_track, bottom_track = y_tracks
946
+ top = top_track.coordinate
947
+ bottom = bottom_track.coordinate
948
+ left = x_tracks[0].coordinate
949
+ right = x_tracks[-1].coordinate
950
+ top_coverage = _separator_coverage_for_track(
951
+ rules,
952
+ "horizontal",
953
+ top_track,
954
+ left,
955
+ right,
956
+ snap_tolerance,
957
+ )
958
+ bottom_coverage = _separator_coverage_for_track(
959
+ rules,
960
+ "horizontal",
961
+ bottom_track,
962
+ left,
963
+ right,
964
+ snap_tolerance,
965
+ )
966
+ vertical_coverages = tuple(
967
+ _separator_coverage_for_track(
968
+ rules,
969
+ "vertical",
970
+ track,
971
+ top,
972
+ bottom,
973
+ snap_tolerance,
974
+ )
975
+ for track in x_tracks
976
+ )
977
+ glyph_crossing = any(glyph.bbox[0] < track.coordinate < glyph.bbox[2] for glyph in text.glyphs for track in x_tracks[1:-1])
978
+ return _SingleRowEvidence(
979
+ top_coverage=top_coverage,
980
+ bottom_coverage=bottom_coverage,
981
+ vertical_coverages=vertical_coverages,
982
+ height_ratio=min(
983
+ 1.0,
984
+ (bottom - top) / max(minimum_row_height, 0.1),
985
+ ),
986
+ glyph_crossing=glyph_crossing,
987
+ )
988
+
989
+
990
+ def _single_column_line_grid_evidence(
991
+ rules: list[_MergedRule],
992
+ x_tracks: list[_CanonicalTrack],
993
+ y_tracks: list[_CanonicalTrack],
994
+ text: NativeTableText,
995
+ snap_tolerance: float,
996
+ minimum_row_height: float,
997
+ ) -> _SingleColumnEvidence:
998
+ """校验多行单列表单的全部横边和左右连续外框。"""
999
+
1000
+ left_track, right_track = x_tracks
1001
+ top = y_tracks[0].coordinate
1002
+ bottom = y_tracks[-1].coordinate
1003
+ horizontal_coverages = tuple(
1004
+ _separator_coverage_for_track(
1005
+ rules,
1006
+ "horizontal",
1007
+ track,
1008
+ left_track.coordinate,
1009
+ right_track.coordinate,
1010
+ snap_tolerance,
1011
+ )
1012
+ for track in y_tracks
1013
+ )
1014
+ left_coverage = _separator_coverage_for_track(
1015
+ rules,
1016
+ "vertical",
1017
+ left_track,
1018
+ top,
1019
+ bottom,
1020
+ snap_tolerance,
1021
+ )
1022
+ right_coverage = _separator_coverage_for_track(
1023
+ rules,
1024
+ "vertical",
1025
+ right_track,
1026
+ top,
1027
+ bottom,
1028
+ snap_tolerance,
1029
+ )
1030
+ minimum_height_ratio = min(
1031
+ (
1032
+ min(
1033
+ 1.0,
1034
+ (current.coordinate - previous.coordinate) / max(minimum_row_height, 0.1),
1035
+ )
1036
+ for previous, current in zip(y_tracks, y_tracks[1:])
1037
+ ),
1038
+ default=0.0,
1039
+ )
1040
+ glyph_crossing = any(glyph.bbox[1] < track.coordinate < glyph.bbox[3] for glyph in text.glyphs for track in y_tracks[1:-1])
1041
+ return _SingleColumnEvidence(
1042
+ horizontal_coverages=horizontal_coverages,
1043
+ left_coverage=left_coverage,
1044
+ right_coverage=right_coverage,
1045
+ minimum_height_ratio=minimum_height_ratio,
1046
+ glyph_crossing=glyph_crossing,
1047
+ )
1048
+
1049
+
1050
+ def _text_grid_stability(
1051
+ text: NativeTableText,
1052
+ x_tracks: list[float],
1053
+ y_tracks: list[float],
1054
+ physically_verified_rows: set[int] | None = None,
1055
+ ) -> tuple[float, float]:
1056
+ """衡量视觉文本行和文本项对推断行列轨道的占用稳定性。"""
1057
+
1058
+ occupied_rows = _occupied_text_rows(text, y_tracks)
1059
+ supported_rows = occupied_rows | (physically_verified_rows or set())
1060
+ occupied_cols: set[int] = set()
1061
+ for row in text.rows:
1062
+ for token in row.tokens:
1063
+ center_x = (token.bbox[0] + token.bbox[2]) / 2.0
1064
+ for col_index, (left, right) in enumerate(zip(x_tracks, x_tracks[1:])):
1065
+ if left <= center_x <= right:
1066
+ occupied_cols.add(col_index)
1067
+ break
1068
+ row_denominator = min(
1069
+ len(y_tracks) - 1,
1070
+ max(1, len(text.rows)),
1071
+ )
1072
+ col_denominator = min(
1073
+ len(x_tracks) - 1,
1074
+ max((len(row.tokens) for row in text.rows), default=1),
1075
+ )
1076
+ return (
1077
+ min(1.0, len(supported_rows) / row_denominator),
1078
+ min(1.0, len(occupied_cols) / col_denominator),
1079
+ )
1080
+
1081
+
1082
+ def _physical_row_dense_baseline_pairs(
1083
+ text: NativeTableText,
1084
+ x_tracks: list[float],
1085
+ y_tracks: list[float],
1086
+ ) -> tuple[dict[str, object], ...]:
1087
+ """识别同一物理行带内占用集合相同的多条稠密文本基线。"""
1088
+
1089
+ cols = len(x_tracks) - 1
1090
+ dense_column_count = max(2, math.ceil(0.60 * cols))
1091
+ rows_by_band: dict[int, list[tuple[int, tuple[int, ...]]]] = {row: [] for row in range(len(y_tracks) - 1)}
1092
+ for text_row in text.rows:
1093
+ center_y = (text_row.bbox[1] + text_row.bbox[3]) / 2.0
1094
+ band = next(
1095
+ (row for row, (top, bottom) in enumerate(zip(y_tracks, y_tracks[1:])) if top <= center_y <= bottom),
1096
+ None,
1097
+ )
1098
+ if band is None:
1099
+ continue
1100
+ occupied_cols: set[int] = set()
1101
+ for token in text_row.tokens:
1102
+ center_x = (token.bbox[0] + token.bbox[2]) / 2.0
1103
+ col = next(
1104
+ (index for index, (left, right) in enumerate(zip(x_tracks, x_tracks[1:])) if left <= center_x <= right),
1105
+ None,
1106
+ )
1107
+ if col is not None:
1108
+ occupied_cols.add(col)
1109
+ rows_by_band[band].append(
1110
+ (
1111
+ text_row.row_index,
1112
+ tuple(sorted(occupied_cols)),
1113
+ )
1114
+ )
1115
+
1116
+ ambiguous_pairs: list[dict[str, object]] = []
1117
+ for band, entries in rows_by_band.items():
1118
+ nonempty_entries = [entry for entry in entries if entry[1]]
1119
+ if (
1120
+ len(nonempty_entries) < 2
1121
+ or len(nonempty_entries[0][1]) < dense_column_count
1122
+ or any(entry[1] != nonempty_entries[0][1] for entry in nonempty_entries[1:])
1123
+ ):
1124
+ continue
1125
+ for previous, current in zip(nonempty_entries, nonempty_entries[1:]):
1126
+ ambiguous_pairs.append(
1127
+ {
1128
+ "physical_row": band,
1129
+ "visual_rows": [previous[0], current[0]],
1130
+ "occupied_cols": list(previous[1]),
1131
+ }
1132
+ )
1133
+ return tuple(ambiguous_pairs)
1134
+
1135
+
1136
+ def _looks_like_single_column_tracks(
1137
+ x_tracks: list[float],
1138
+ local_width: float,
1139
+ edge_band: float,
1140
+ ) -> bool:
1141
+ """判断初始 X 轨是否全部属于单列表单的左右外缘。"""
1142
+
1143
+ return (
1144
+ len(x_tracks) >= 2
1145
+ and local_width > 0
1146
+ and all(min(abs(track), abs(local_width - track)) <= edge_band for track in x_tracks)
1147
+ )
1148
+
1149
+
1150
+ @dataclass(frozen=True, slots=True)
1151
+ class _VectorTracks:
1152
+ """保存已通过别名、尺寸和物理行数校验的规范轨道。"""
1153
+
1154
+ snap_tolerance: float
1155
+ local_width: float
1156
+ rules: list[_MergedRule]
1157
+ canonical_x_tracks: list[_CanonicalTrack]
1158
+ canonical_y_tracks: list[_CanonicalTrack]
1159
+ x_tracks: list[float]
1160
+ y_tracks: list[float]
1161
+ narrow_empty_threshold: float
1162
+ is_line_grid: bool
1163
+ is_single_row_shape: bool
1164
+ is_single_column_shape: bool
1165
+ rows: int
1166
+ cols: int
1167
+
1168
+
1169
+ @dataclass(frozen=True, slots=True)
1170
+ class _VectorTopology:
1171
+ """保存隔断连接后的逻辑单元格及独立物理证据。"""
1172
+
1173
+ specs: tuple[GridCellSpec, ...]
1174
+ separator_decisions: list[float]
1175
+ ambiguous_ratio: float
1176
+ alias_separator_recoveries: int
1177
+ y_alias_separator_recoveries: int
1178
+ alias_affected_rows: set[int]
1179
+ single_row_evidence: _SingleRowEvidence | None
1180
+ single_column_evidence: _SingleColumnEvidence | None
1181
+
1182
+
1183
+ def _reject_vector_candidate(diagnostics: dict[str, Any] | None, gate: str) -> None:
1184
+ """记录当前假设的首个拒绝门,供各显式阶段保留统一诊断行为。"""
1185
+
1186
+ if diagnostics is not None:
1187
+ diagnostics["first_rejection_gate"] = gate
1188
+ return None
1189
+
1190
+
1191
+ def _build_vector_tracks(
1192
+ table_input: NativeTableInput,
1193
+ text: NativeTableText,
1194
+ *,
1195
+ include_drawing: bool,
1196
+ include_rectangles: bool,
1197
+ prune_unsupported_horizontal: bool,
1198
+ diagnostics: dict[str, Any] | None,
1199
+ ) -> _VectorTracks | None:
1200
+ """构造并规范化轨道,保持 halo、别名及物理行数的原有拒绝顺序。"""
1201
+
1202
+ snap_tolerance = clamp(
1203
+ 0.08 * text.median_glyph_height,
1204
+ 0.5,
1205
+ 2.5,
1206
+ )
1207
+ join_gap = clamp(
1208
+ 0.40 * text.median_glyph_height,
1209
+ 2.0,
1210
+ 8.0,
1211
+ )
1212
+ configured_evidence_halo = (
1213
+ clamp(
1214
+ 0.25 * text.median_glyph_height,
1215
+ 1.0,
1216
+ 3.0,
1217
+ )
1218
+ if include_drawing and not include_rectangles
1219
+ else 0.0
1220
+ )
1221
+ table_bbox = normalize_bbox(table_input.table_bbox)
1222
+ if table_bbox is None:
1223
+ return _reject_vector_candidate(diagnostics, "table_geometry")
1224
+ local_width, local_height = table_local_size(
1225
+ table_bbox,
1226
+ normalize_angle(table_input.angle),
1227
+ )
1228
+ exact_fragments = _local_rule_fragments(
1229
+ table_input,
1230
+ snap_tolerance,
1231
+ include_drawing=include_drawing,
1232
+ include_rectangles=include_rectangles,
1233
+ evidence_halo=0.0,
1234
+ )
1235
+ if not exact_fragments or len(exact_fragments) > MAX_PRIMITIVES_PER_TABLE:
1236
+ return _reject_vector_candidate(diagnostics, "raw_fragments")
1237
+ exact_rules = _merge_rule_fragments(
1238
+ exact_fragments,
1239
+ snap_tolerance,
1240
+ join_gap,
1241
+ )
1242
+ exact_x_tracks, exact_y_tracks, _removed = _infer_grid_tracks(
1243
+ exact_rules,
1244
+ snap_tolerance,
1245
+ local_width,
1246
+ local_height,
1247
+ )
1248
+ # 多行表格保持原始 bbox 裁剪;halo 只服务可能退化为单物理行的
1249
+ # 边界片段,避免吸入相邻行或页外端点改变既有拓扑。
1250
+ single_column_halo_hint = (
1251
+ include_drawing
1252
+ and not include_rectangles
1253
+ and len(exact_y_tracks) >= 3
1254
+ and _looks_like_single_column_tracks(
1255
+ exact_x_tracks,
1256
+ local_width,
1257
+ max(3.0 * configured_evidence_halo, 2.0),
1258
+ )
1259
+ )
1260
+ evidence_halo = (
1261
+ configured_evidence_halo
1262
+ if configured_evidence_halo > 0
1263
+ and (
1264
+ single_column_halo_hint
1265
+ or len(exact_y_tracks) == 2
1266
+ or (
1267
+ len(exact_y_tracks) == 3
1268
+ and min(
1269
+ current - previous
1270
+ for previous, current in zip(
1271
+ exact_y_tracks,
1272
+ exact_y_tracks[1:],
1273
+ )
1274
+ )
1275
+ <= 0.75 * text.median_glyph_height
1276
+ )
1277
+ )
1278
+ else 0.0
1279
+ )
1280
+ if evidence_halo > 0:
1281
+ raw_fragments = _local_rule_fragments(
1282
+ table_input,
1283
+ snap_tolerance,
1284
+ include_drawing=include_drawing,
1285
+ include_rectangles=include_rectangles,
1286
+ evidence_halo=evidence_halo,
1287
+ )
1288
+ rules = _merge_rule_fragments(
1289
+ raw_fragments,
1290
+ snap_tolerance,
1291
+ join_gap,
1292
+ )
1293
+ else:
1294
+ raw_fragments = exact_fragments
1295
+ rules = exact_rules
1296
+ if diagnostics is not None:
1297
+ diagnostics["raw_fragment_count"] = len(raw_fragments)
1298
+ inferred_x_tracks, inferred_y_tracks, removed_horizontal_tracks = _infer_grid_tracks(
1299
+ rules,
1300
+ snap_tolerance,
1301
+ local_width,
1302
+ local_height,
1303
+ prune_unsupported_horizontal=prune_unsupported_horizontal,
1304
+ )
1305
+ if diagnostics is not None:
1306
+ diagnostics["track_hypothesis"] = "supported" if prune_unsupported_horizontal else "raw"
1307
+ diagnostics["evidence_halo"] = evidence_halo
1308
+ diagnostics["removed_horizontal_tracks"] = removed_horizontal_tracks
1309
+ diagnostics["inferred_tracks"] = {
1310
+ "x": len(inferred_x_tracks),
1311
+ "y": len(inferred_y_tracks),
1312
+ }
1313
+ if (
1314
+ include_rectangles
1315
+ and not include_drawing
1316
+ and not _rect_lattice_is_repeated(
1317
+ rules,
1318
+ inferred_x_tracks,
1319
+ inferred_y_tracks,
1320
+ snap_tolerance,
1321
+ )
1322
+ ):
1323
+ return _reject_vector_candidate(diagnostics, "rect_lattice")
1324
+ line_widths = [rule.width for rule in table_input.drawing_lines if rule.width > 0]
1325
+ median_line_width = float(statistics.median(line_widths)) if line_widths else 0.0
1326
+ narrow_empty_threshold = max(
1327
+ 0.75 * text.median_glyph_height,
1328
+ 3.0 * median_line_width,
1329
+ )
1330
+ glyph_centers_x = [(glyph.bbox[0] + glyph.bbox[2]) / 2.0 for glyph in text.glyphs]
1331
+ glyph_centers_y = [(glyph.bbox[1] + glyph.bbox[3]) / 2.0 for glyph in text.glyphs]
1332
+ canonical_x_tracks, x_alias_conflict, outer_x_collapses = _canonicalize_axis_tracks(
1333
+ inferred_x_tracks,
1334
+ glyph_centers_x,
1335
+ narrow_empty_threshold,
1336
+ local_width,
1337
+ evidence_halo,
1338
+ 3,
1339
+ rules=rules,
1340
+ orientation="vertical",
1341
+ separator_extent=local_height,
1342
+ snap_tolerance=snap_tolerance,
1343
+ )
1344
+ canonical_y_tracks, y_alias_conflict, outer_y_collapses = _canonicalize_axis_tracks(
1345
+ inferred_y_tracks,
1346
+ glyph_centers_y,
1347
+ narrow_empty_threshold,
1348
+ local_height,
1349
+ evidence_halo,
1350
+ 2,
1351
+ rules=rules,
1352
+ orientation="horizontal",
1353
+ separator_extent=local_width,
1354
+ snap_tolerance=snap_tolerance,
1355
+ preserve_double_boundary=True,
1356
+ collapse_narrow_bands=len(exact_y_tracks) <= 3,
1357
+ collapse_leading_edge=False,
1358
+ )
1359
+ if (
1360
+ x_alias_conflict
1361
+ or y_alias_conflict
1362
+ or not _canonical_tracks_are_unique(
1363
+ canonical_x_tracks,
1364
+ rules,
1365
+ "vertical",
1366
+ snap_tolerance,
1367
+ )
1368
+ or not _canonical_tracks_are_unique(
1369
+ canonical_y_tracks,
1370
+ rules,
1371
+ "horizontal",
1372
+ snap_tolerance,
1373
+ )
1374
+ ):
1375
+ return _reject_vector_candidate(diagnostics, "canonical_alias")
1376
+ x_tracks = _canonical_track_coordinates(canonical_x_tracks)
1377
+ y_tracks = _canonical_track_coordinates(canonical_y_tracks)
1378
+ if diagnostics is not None:
1379
+ diagnostics["canonical_tracks"] = [
1380
+ {
1381
+ "coordinate": track.coordinate,
1382
+ "aliases": list(track.aliases),
1383
+ }
1384
+ for track in canonical_x_tracks
1385
+ ]
1386
+ diagnostics["canonical_y_tracks"] = [
1387
+ {
1388
+ "coordinate": track.coordinate,
1389
+ "aliases": list(track.aliases),
1390
+ }
1391
+ for track in canonical_y_tracks
1392
+ ]
1393
+ diagnostics["outer_track_collapses"] = {
1394
+ "x": outer_x_collapses,
1395
+ "y": outer_y_collapses,
1396
+ }
1397
+ if any(
1398
+ right - left <= narrow_empty_threshold and not any(left < center < right for center in glyph_centers_x)
1399
+ for left, right in zip(x_tracks, x_tracks[1:])
1400
+ ):
1401
+ return _reject_vector_candidate(diagnostics, "remaining_narrow_track")
1402
+ if any(
1403
+ bottom_track.coordinate - top_track.coordinate <= narrow_empty_threshold
1404
+ and not any(top_track.coordinate < center < bottom_track.coordinate for center in glyph_centers_y)
1405
+ and _separator_coverage_for_track(
1406
+ rules,
1407
+ "horizontal",
1408
+ top_track,
1409
+ x_tracks[0],
1410
+ x_tracks[-1],
1411
+ snap_tolerance,
1412
+ )
1413
+ >= SEPARATOR_COVERAGE_THRESHOLD
1414
+ and _separator_coverage_for_track(
1415
+ rules,
1416
+ "horizontal",
1417
+ bottom_track,
1418
+ x_tracks[0],
1419
+ x_tracks[-1],
1420
+ snap_tolerance,
1421
+ )
1422
+ >= SEPARATOR_COVERAGE_THRESHOLD
1423
+ for top_track, bottom_track in zip(
1424
+ canonical_y_tracks,
1425
+ canonical_y_tracks[1:],
1426
+ )
1427
+ ):
1428
+ return _reject_vector_candidate(diagnostics, "remaining_narrow_track")
1429
+ is_line_grid = include_drawing and not include_rectangles
1430
+ is_single_row_shape = is_line_grid and len(y_tracks) == 2 and len(x_tracks) >= 3
1431
+ is_single_column_shape = is_line_grid and len(x_tracks) == 2 and len(y_tracks) >= 3
1432
+ if (
1433
+ len(x_tracks) < (2 if is_single_column_shape else 3)
1434
+ or len(y_tracks) < (2 if is_single_row_shape else 3)
1435
+ or len(x_tracks) > MAX_TRACKS_PER_AXIS
1436
+ or len(y_tracks) > MAX_TRACKS_PER_AXIS
1437
+ ):
1438
+ return _reject_vector_candidate(diagnostics, "track_count")
1439
+ rows = len(y_tracks) - 1
1440
+ cols = len(x_tracks) - 1
1441
+ if diagnostics is not None:
1442
+ diagnostics["grid"] = {"rows": rows, "cols": cols}
1443
+ if rows * cols > MAX_ATOMIC_CELLS:
1444
+ return _reject_vector_candidate(diagnostics, "atomic_cell_limit")
1445
+ dense_baseline_pairs = (
1446
+ _physical_row_dense_baseline_pairs(
1447
+ text,
1448
+ x_tracks,
1449
+ y_tracks,
1450
+ )
1451
+ if is_line_grid
1452
+ else ()
1453
+ )
1454
+ if diagnostics is not None:
1455
+ diagnostics["physical_row_dense_baseline_pairs"] = list(dense_baseline_pairs)
1456
+ if dense_baseline_pairs:
1457
+ return _reject_vector_candidate(diagnostics, "physical_row_undercount")
1458
+
1459
+ return _VectorTracks(
1460
+ snap_tolerance=snap_tolerance,
1461
+ local_width=local_width,
1462
+ rules=rules,
1463
+ canonical_x_tracks=canonical_x_tracks,
1464
+ canonical_y_tracks=canonical_y_tracks,
1465
+ x_tracks=x_tracks,
1466
+ y_tracks=y_tracks,
1467
+ narrow_empty_threshold=narrow_empty_threshold,
1468
+ is_line_grid=is_line_grid,
1469
+ is_single_row_shape=is_single_row_shape,
1470
+ is_single_column_shape=is_single_column_shape,
1471
+ rows=rows,
1472
+ cols=cols,
1473
+ )
1474
+
1475
+
1476
+ def _build_vector_topology(
1477
+ tracks: _VectorTracks,
1478
+ text: NativeTableText,
1479
+ diagnostics: dict[str, Any] | None,
1480
+ ) -> _VectorTopology | None:
1481
+ """连接原子格并验证矩形拓扑,保留单行和单列的独立物理证据。"""
1482
+
1483
+ snap_tolerance = tracks.snap_tolerance
1484
+ rules = tracks.rules
1485
+ canonical_x_tracks = tracks.canonical_x_tracks
1486
+ canonical_y_tracks = tracks.canonical_y_tracks
1487
+ x_tracks = tracks.x_tracks
1488
+ y_tracks = tracks.y_tracks
1489
+ narrow_empty_threshold = tracks.narrow_empty_threshold
1490
+ is_single_row_shape = tracks.is_single_row_shape
1491
+ is_single_column_shape = tracks.is_single_column_shape
1492
+ rows = tracks.rows
1493
+ cols = tracks.cols
1494
+
1495
+ union_find = _UnionFind(rows * cols)
1496
+ separator_decisions: list[float] = []
1497
+ ambiguous_separator_count = 0
1498
+ alias_separator_recoveries = 0
1499
+ y_alias_separator_recoveries = 0
1500
+ alias_affected_rows: set[int] = set()
1501
+ for row in range(rows):
1502
+ for boundary_index in range(1, len(canonical_x_tracks) - 1):
1503
+ track = canonical_x_tracks[boundary_index]
1504
+ strict_coverage = _separator_coverage(
1505
+ rules,
1506
+ "vertical",
1507
+ track.coordinate,
1508
+ y_tracks[row],
1509
+ y_tracks[row + 1],
1510
+ snap_tolerance,
1511
+ )
1512
+ coverage = _separator_coverage_for_track(
1513
+ rules,
1514
+ "vertical",
1515
+ track,
1516
+ y_tracks[row],
1517
+ y_tracks[row + 1],
1518
+ snap_tolerance,
1519
+ )
1520
+ if (
1521
+ len(track.aliases) > 1
1522
+ and strict_coverage <= 1.0 - SEPARATOR_COVERAGE_THRESHOLD
1523
+ and coverage >= SEPARATOR_COVERAGE_THRESHOLD
1524
+ ):
1525
+ alias_separator_recoveries += 1
1526
+ alias_affected_rows.add(row)
1527
+ separator_decisions.append(max(coverage, 1.0 - coverage))
1528
+ if coverage <= 1.0 - SEPARATOR_COVERAGE_THRESHOLD:
1529
+ union_find.union(
1530
+ _grid_index(row, boundary_index - 1, cols),
1531
+ _grid_index(row, boundary_index, cols),
1532
+ )
1533
+ elif coverage < SEPARATOR_COVERAGE_THRESHOLD:
1534
+ ambiguous_separator_count += 1
1535
+ for boundary_index in range(1, len(canonical_y_tracks) - 1):
1536
+ track = canonical_y_tracks[boundary_index]
1537
+ for col in range(cols):
1538
+ strict_coverage = _separator_coverage(
1539
+ rules,
1540
+ "horizontal",
1541
+ track.coordinate,
1542
+ x_tracks[col],
1543
+ x_tracks[col + 1],
1544
+ snap_tolerance,
1545
+ )
1546
+ coverage = _separator_coverage_for_track(
1547
+ rules,
1548
+ "horizontal",
1549
+ track,
1550
+ x_tracks[col],
1551
+ x_tracks[col + 1],
1552
+ snap_tolerance,
1553
+ )
1554
+ if (
1555
+ len(track.aliases) > 1
1556
+ and strict_coverage <= 1.0 - SEPARATOR_COVERAGE_THRESHOLD
1557
+ and coverage >= SEPARATOR_COVERAGE_THRESHOLD
1558
+ ):
1559
+ y_alias_separator_recoveries += 1
1560
+ if boundary_index > 0:
1561
+ alias_affected_rows.add(boundary_index - 1)
1562
+ if boundary_index < rows:
1563
+ alias_affected_rows.add(boundary_index)
1564
+ separator_decisions.append(max(coverage, 1.0 - coverage))
1565
+ if coverage <= 1.0 - SEPARATOR_COVERAGE_THRESHOLD:
1566
+ union_find.union(
1567
+ _grid_index(boundary_index - 1, col, cols),
1568
+ _grid_index(boundary_index, col, cols),
1569
+ )
1570
+ elif coverage < SEPARATOR_COVERAGE_THRESHOLD:
1571
+ ambiguous_separator_count += 1
1572
+
1573
+ ambiguous_ratio = ambiguous_separator_count / len(separator_decisions) if separator_decisions else 0.0
1574
+ if diagnostics is not None:
1575
+ diagnostics["grid"] = {"rows": rows, "cols": cols}
1576
+ diagnostics["ambiguous_separator_ratio"] = ambiguous_ratio
1577
+ diagnostics["alias_separator_recoveries"] = alias_separator_recoveries
1578
+ diagnostics["y_alias_separator_recoveries"] = y_alias_separator_recoveries
1579
+ diagnostics["alias_affected_rows"] = sorted(alias_affected_rows)
1580
+ if ambiguous_ratio > 0.05:
1581
+ return _reject_vector_candidate(diagnostics, "ambiguous_separator")
1582
+
1583
+ single_row_evidence: _SingleRowEvidence | None = None
1584
+ if is_single_row_shape:
1585
+ single_row_evidence = _single_row_line_grid_evidence(
1586
+ rules,
1587
+ canonical_x_tracks,
1588
+ canonical_y_tracks,
1589
+ text,
1590
+ snap_tolerance,
1591
+ narrow_empty_threshold,
1592
+ )
1593
+ if diagnostics is not None:
1594
+ diagnostics["single_row_evidence"] = {
1595
+ "reliability": single_row_evidence.reliability,
1596
+ "confidence": single_row_evidence.confidence,
1597
+ "verified": single_row_evidence.verified,
1598
+ "top": single_row_evidence.top_coverage,
1599
+ "bottom": single_row_evidence.bottom_coverage,
1600
+ "vertical": list(single_row_evidence.vertical_coverages),
1601
+ "height_ratio": single_row_evidence.height_ratio,
1602
+ "glyph_crossing": single_row_evidence.glyph_crossing,
1603
+ }
1604
+ if not single_row_evidence.verified:
1605
+ return _reject_vector_candidate(diagnostics, "single_row_physical_evidence")
1606
+
1607
+ single_column_evidence: _SingleColumnEvidence | None = None
1608
+ if is_single_column_shape:
1609
+ single_column_evidence = _single_column_line_grid_evidence(
1610
+ rules,
1611
+ canonical_x_tracks,
1612
+ canonical_y_tracks,
1613
+ text,
1614
+ snap_tolerance,
1615
+ narrow_empty_threshold,
1616
+ )
1617
+ if diagnostics is not None:
1618
+ diagnostics["single_column_evidence"] = {
1619
+ "reliability": single_column_evidence.reliability,
1620
+ "confidence": single_column_evidence.confidence,
1621
+ "verified": single_column_evidence.verified,
1622
+ "horizontal": list(single_column_evidence.horizontal_coverages),
1623
+ "left": single_column_evidence.left_coverage,
1624
+ "right": single_column_evidence.right_coverage,
1625
+ "minimum_height_ratio": (single_column_evidence.minimum_height_ratio),
1626
+ "glyph_crossing": single_column_evidence.glyph_crossing,
1627
+ }
1628
+ if not single_column_evidence.verified:
1629
+ return _reject_vector_candidate(diagnostics, "single_column_physical_evidence")
1630
+
1631
+ specs = _build_component_specs(
1632
+ union_find,
1633
+ rows,
1634
+ cols,
1635
+ x_tracks,
1636
+ y_tracks,
1637
+ )
1638
+ if specs is None:
1639
+ return _reject_vector_candidate(diagnostics, "nonrectangular_topology")
1640
+ maximum_row_cells = max(sum(spec.row <= row_index < spec.row + spec.rowspan for spec in specs) for row_index in range(rows))
1641
+ maximum_col_cells = max(sum(spec.col <= col_index < spec.col + spec.colspan for spec in specs) for col_index in range(cols))
1642
+ if (maximum_row_cells < 2 and not is_single_column_shape) or (maximum_col_cells < 2 and not is_single_row_shape):
1643
+ return _reject_vector_candidate(diagnostics, "degenerate_grid")
1644
+ return _VectorTopology(
1645
+ specs=specs,
1646
+ separator_decisions=separator_decisions,
1647
+ ambiguous_ratio=ambiguous_ratio,
1648
+ alias_separator_recoveries=alias_separator_recoveries,
1649
+ y_alias_separator_recoveries=y_alias_separator_recoveries,
1650
+ alias_affected_rows=alias_affected_rows,
1651
+ single_row_evidence=single_row_evidence,
1652
+ single_column_evidence=single_column_evidence,
1653
+ )
1654
+
1655
+
1656
+ def _materialize_vector_candidate(
1657
+ tracks: _VectorTracks,
1658
+ topology: _VectorTopology,
1659
+ text: NativeTableText,
1660
+ evidence_label: str,
1661
+ diagnostics: dict[str, Any] | None,
1662
+ ) -> NativeTableCandidate | None:
1663
+ """将文本落格并评分,按原顺序执行完整性与空行发布门。"""
1664
+
1665
+ snap_tolerance = tracks.snap_tolerance
1666
+ local_width = tracks.local_width
1667
+ rules = tracks.rules
1668
+ canonical_x_tracks = tracks.canonical_x_tracks
1669
+ canonical_y_tracks = tracks.canonical_y_tracks
1670
+ x_tracks = tracks.x_tracks
1671
+ y_tracks = tracks.y_tracks
1672
+ narrow_empty_threshold = tracks.narrow_empty_threshold
1673
+ is_line_grid = tracks.is_line_grid
1674
+ is_single_row_shape = tracks.is_single_row_shape
1675
+ is_single_column_shape = tracks.is_single_column_shape
1676
+ rows = tracks.rows
1677
+ cols = tracks.cols
1678
+
1679
+ specs = topology.specs
1680
+ separator_decisions = topology.separator_decisions
1681
+ ambiguous_ratio = topology.ambiguous_ratio
1682
+ alias_separator_recoveries = topology.alias_separator_recoveries
1683
+ y_alias_separator_recoveries = topology.y_alias_separator_recoveries
1684
+ alias_affected_rows = topology.alias_affected_rows
1685
+ single_row_evidence = topology.single_row_evidence
1686
+ single_column_evidence = topology.single_column_evidence
1687
+
1688
+ decisiveness = float(statistics.mean(separator_decisions)) if separator_decisions else 1.0
1689
+ if single_row_evidence is not None:
1690
+ decisiveness = max(
1691
+ decisiveness,
1692
+ single_row_evidence.confidence,
1693
+ )
1694
+ if single_column_evidence is not None:
1695
+ decisiveness = max(
1696
+ decisiveness,
1697
+ single_column_evidence.confidence,
1698
+ )
1699
+ evidence_ratio = min(1.0, len(rules) / max(1, rows + cols))
1700
+ structure_support = min(
1701
+ decisiveness,
1702
+ evidence_ratio,
1703
+ 1.0 - ambiguous_ratio,
1704
+ (single_row_evidence.confidence if single_row_evidence is not None else 1.0),
1705
+ (single_column_evidence.confidence if single_column_evidence is not None else 1.0),
1706
+ )
1707
+ occupied_rows = _occupied_text_rows(text, y_tracks)
1708
+ line_row_evidence: tuple[_PhysicalRowEvidence, ...] = ()
1709
+ physically_verified_rows: set[int] = set()
1710
+ if is_line_grid:
1711
+ line_row_evidence = _line_grid_row_evidence(
1712
+ rules,
1713
+ canonical_x_tracks,
1714
+ canonical_y_tracks,
1715
+ text,
1716
+ snap_tolerance,
1717
+ narrow_empty_threshold,
1718
+ local_width,
1719
+ )
1720
+ physically_verified_rows = {evidence.row for evidence in line_row_evidence if evidence.verified}
1721
+ if diagnostics is not None:
1722
+ diagnostics["physical_rows"] = [
1723
+ {
1724
+ "row": evidence.row,
1725
+ "reliability": evidence.reliability,
1726
+ "verified": evidence.verified,
1727
+ "top": evidence.top_coverage,
1728
+ "bottom": evidence.bottom_coverage,
1729
+ "left": evidence.left_coverage,
1730
+ "right": evidence.right_coverage,
1731
+ "height_ratio": evidence.height_ratio,
1732
+ "glyph_crossing": evidence.glyph_crossing,
1733
+ }
1734
+ for evidence in line_row_evidence
1735
+ ]
1736
+ row_stability, column_stability = _text_grid_stability(
1737
+ text,
1738
+ x_tracks,
1739
+ y_tracks,
1740
+ physically_verified_rows=(physically_verified_rows if is_line_grid else None),
1741
+ )
1742
+ collapsed_track_count = sum(len(track.aliases) > 1 for track in canonical_x_tracks)
1743
+ collapsed_y_track_count = sum(len(track.aliases) > 1 for track in canonical_y_tracks)
1744
+ maximum_alias_span = max(
1745
+ (
1746
+ track.aliases[-1] - track.aliases[0]
1747
+ for track in (*canonical_x_tracks, *canonical_y_tracks)
1748
+ if len(track.aliases) > 1
1749
+ ),
1750
+ default=0.0,
1751
+ )
1752
+ potential_blank_rows = sorted(set(range(rows)) - occupied_rows)
1753
+ candidate = build_candidate(
1754
+ source="vector_grid",
1755
+ rows=rows,
1756
+ cols=cols,
1757
+ specs=specs,
1758
+ text=text,
1759
+ structure_support=structure_support,
1760
+ row_stability=row_stability,
1761
+ column_stability=column_stability,
1762
+ issues=(
1763
+ f"evidence={evidence_label}",
1764
+ f"ambiguous_separator_ratio={ambiguous_ratio:.4f}",
1765
+ f"collapsed_x_tracks={collapsed_track_count}",
1766
+ f"collapsed_y_tracks={collapsed_y_track_count}",
1767
+ f"alias_max_span={maximum_alias_span:.4f}",
1768
+ f"alias_separator_recoveries={alias_separator_recoveries}",
1769
+ f"y_alias_separator_recoveries={y_alias_separator_recoveries}",
1770
+ f"single_row_line_grid={str(is_single_row_shape).lower()}",
1771
+ f"single_column_line_grid={str(is_single_column_shape).lower()}",
1772
+ "single_row_reliability="
1773
+ + (f"{single_row_evidence.reliability:.4f}" if single_row_evidence is not None else "n/a"),
1774
+ "single_row_confidence=" + (f"{single_row_evidence.confidence:.4f}" if single_row_evidence is not None else "n/a"),
1775
+ "single_column_reliability="
1776
+ + (f"{single_column_evidence.reliability:.4f}" if single_column_evidence is not None else "n/a"),
1777
+ "single_column_confidence="
1778
+ + (f"{single_column_evidence.confidence:.4f}" if single_column_evidence is not None else "n/a"),
1779
+ "physical_blank_rows=" + ",".join(str(row) for row in potential_blank_rows if row in physically_verified_rows),
1780
+ ),
1781
+ allow_single_row=is_single_row_shape,
1782
+ allow_single_column=is_single_column_shape,
1783
+ use_grid_index=True,
1784
+ diagnostics=diagnostics,
1785
+ )
1786
+ if candidate is None:
1787
+ candidate_gate = (
1788
+ str(diagnostics.get("candidate_rejection_gate"))
1789
+ if diagnostics is not None and diagnostics.get("candidate_rejection_gate")
1790
+ else "candidate_hard_gate"
1791
+ )
1792
+ return _reject_vector_candidate(diagnostics, candidate_gate)
1793
+ if is_single_row_shape and (candidate.text_capture < 1.0 or candidate.order_consistency < 1.0):
1794
+ return _reject_vector_candidate(diagnostics, "single_row_text_integrity")
1795
+ if is_single_column_shape and (candidate.text_capture < 1.0 or candidate.order_consistency < 1.0):
1796
+ return _reject_vector_candidate(diagnostics, "single_column_text_integrity")
1797
+ row_content_support = [
1798
+ sum(bool(cell.content.strip()) for cell in candidate.cells if cell.row <= row_index < cell.row + cell.rowspan)
1799
+ for row_index in range(candidate.rows)
1800
+ ]
1801
+ empty_rows = {row_index for row_index, support in enumerate(row_content_support) if support == 0}
1802
+ if empty_rows and len(text.rows) >= 2:
1803
+ if (
1804
+ not is_line_grid
1805
+ or not empty_rows.isdisjoint(alias_affected_rows)
1806
+ or not empty_rows.issubset(physically_verified_rows)
1807
+ ):
1808
+ if diagnostics is not None:
1809
+ diagnostics["empty_rows"] = sorted(empty_rows)
1810
+ return _reject_vector_candidate(diagnostics, "empty_row")
1811
+ if diagnostics is not None:
1812
+ diagnostics["first_rejection_gate"] = None
1813
+ diagnostics["score"] = candidate.score
1814
+ diagnostics["empty_rows"] = sorted(empty_rows)
1815
+ return candidate
1816
+
1817
+
1818
+ def _build_vector_candidate(
1819
+ table_input: NativeTableInput,
1820
+ text: NativeTableText,
1821
+ *,
1822
+ include_drawing: bool,
1823
+ include_rectangles: bool,
1824
+ evidence_label: str,
1825
+ prune_unsupported_horizontal: bool = False,
1826
+ diagnostics: dict[str, Any] | None = None,
1827
+ ) -> NativeTableCandidate | None:
1828
+ """按轨道、拓扑、文本落格及评分的固定顺序构造矢量候选。"""
1829
+
1830
+ if diagnostics is not None:
1831
+ diagnostics["evidence"] = evidence_label
1832
+ tracks = _build_vector_tracks(
1833
+ table_input,
1834
+ text,
1835
+ include_drawing=include_drawing,
1836
+ include_rectangles=include_rectangles,
1837
+ prune_unsupported_horizontal=prune_unsupported_horizontal,
1838
+ diagnostics=diagnostics,
1839
+ )
1840
+ if tracks is None:
1841
+ return None
1842
+ topology = _build_vector_topology(tracks, text, diagnostics)
1843
+ if topology is None:
1844
+ return None
1845
+ return _materialize_vector_candidate(tracks, topology, text, evidence_label, diagnostics)
1846
+
1847
+
1848
+ def build_vector_candidates(
1849
+ table_input: NativeTableInput,
1850
+ text: NativeTableText,
1851
+ diagnostics: list[dict[str, Any]] | None = None,
1852
+ ) -> list[NativeTableCandidate]:
1853
+ """分别从 drawing 中心线和矩形晶格生成矢量网格候选。"""
1854
+
1855
+ candidates: list[NativeTableCandidate] = []
1856
+ raw_line_diagnostics: dict[str, Any] | None = {} if diagnostics is not None else None
1857
+ line_candidate = _build_vector_candidate(
1858
+ table_input,
1859
+ text,
1860
+ include_drawing=True,
1861
+ include_rectangles=False,
1862
+ evidence_label="line_grid",
1863
+ diagnostics=raw_line_diagnostics,
1864
+ )
1865
+ line_hypotheses = [raw_line_diagnostics] if raw_line_diagnostics is not None else []
1866
+ selected_line_diagnostics = raw_line_diagnostics
1867
+ if line_candidate is None and len(line_hypotheses) < MAX_TRACK_HYPOTHESES:
1868
+ supported_line_diagnostics: dict[str, Any] | None = {} if diagnostics is not None else None
1869
+ supported_line_candidate = _build_vector_candidate(
1870
+ table_input,
1871
+ text,
1872
+ include_drawing=True,
1873
+ include_rectangles=False,
1874
+ evidence_label="line_grid",
1875
+ prune_unsupported_horizontal=True,
1876
+ diagnostics=supported_line_diagnostics,
1877
+ )
1878
+ if supported_line_diagnostics is not None:
1879
+ removed_tracks = supported_line_diagnostics.get(
1880
+ "removed_horizontal_tracks",
1881
+ [],
1882
+ )
1883
+ if removed_tracks:
1884
+ line_hypotheses.append(supported_line_diagnostics)
1885
+ selected_line_diagnostics = supported_line_diagnostics
1886
+ if supported_line_candidate is not None:
1887
+ line_candidate = supported_line_candidate
1888
+ selected_line_diagnostics = supported_line_diagnostics
1889
+ if diagnostics is not None and selected_line_diagnostics is not None:
1890
+ line_record = dict(selected_line_diagnostics)
1891
+ line_record["track_hypotheses"] = [dict(hypothesis) for hypothesis in line_hypotheses if hypothesis is not None]
1892
+ diagnostics.append(line_record)
1893
+ if line_candidate is not None:
1894
+ candidates.append(line_candidate)
1895
+ rect_diagnostics: dict[str, Any] | None = {} if diagnostics is not None else None
1896
+ rect_candidate = _build_vector_candidate(
1897
+ table_input,
1898
+ text,
1899
+ include_drawing=False,
1900
+ include_rectangles=True,
1901
+ evidence_label="rect_grid",
1902
+ diagnostics=rect_diagnostics,
1903
+ )
1904
+ if diagnostics is not None and rect_diagnostics is not None:
1905
+ diagnostics.append(rect_diagnostics)
1906
+ if rect_candidate is not None:
1907
+ candidates.append(rect_candidate)
1908
+ return candidates
1909
+
1910
+
1911
+ def diagnose_vector_candidate_builds(
1912
+ table_input: NativeTableInput,
1913
+ text: NativeTableText,
1914
+ ) -> tuple[dict[str, Any], ...]:
1915
+ """返回 line/rect 假设真实首个拒绝门和物理证据。"""
1916
+
1917
+ diagnostics: list[dict[str, Any]] = []
1918
+ build_vector_candidates(
1919
+ table_input,
1920
+ text,
1921
+ diagnostics=diagnostics,
1922
+ )
1923
+ return tuple(diagnostics)
1924
+
1925
+
1926
+ __all__ = [
1927
+ "MAX_ATOMIC_CELLS",
1928
+ "MAX_PRIMITIVES_PER_TABLE",
1929
+ "MAX_TRACKS_PER_AXIS",
1930
+ "build_vector_candidates",
1931
+ ]