docvortex 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (364) hide show
  1. docvortex/__init__.py +21 -0
  2. docvortex/analyzers/__init__.py +3 -0
  3. docvortex/analyzers/native/__init__.py +37 -0
  4. docvortex/analyzers/native/_shared/__init__.py +3 -0
  5. docvortex/analyzers/native/_shared/hyperlink.py +16 -0
  6. docvortex/analyzers/native/_shared/image.py +8 -0
  7. docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
  8. docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
  9. docvortex/analyzers/native/_shared/markup/formula.py +37 -0
  10. docvortex/analyzers/native/_shared/markup/projector.py +57 -0
  11. docvortex/analyzers/native/_shared/markup/styles.py +19 -0
  12. docvortex/analyzers/native/_shared/mathml.py +17 -0
  13. docvortex/analyzers/native/_shared/names.py +7 -0
  14. docvortex/analyzers/native/_shared/xycut.py +414 -0
  15. docvortex/analyzers/native/contracts.py +44 -0
  16. docvortex/analyzers/native/csv.py +351 -0
  17. docvortex/analyzers/native/epub/__init__.py +18 -0
  18. docvortex/analyzers/native/epub/constants.py +45 -0
  19. docvortex/analyzers/native/epub/converter.py +80 -0
  20. docvortex/analyzers/native/epub/errors.py +20 -0
  21. docvortex/analyzers/native/epub/metadata.py +31 -0
  22. docvortex/analyzers/native/epub/package.py +649 -0
  23. docvortex/analyzers/native/epub/xhtml.py +306 -0
  24. docvortex/analyzers/native/html/__init__.py +6 -0
  25. docvortex/analyzers/native/html/anchors.py +271 -0
  26. docvortex/analyzers/native/html/constants.py +25 -0
  27. docvortex/analyzers/native/html/contracts.py +7 -0
  28. docvortex/analyzers/native/html/converter.py +117 -0
  29. docvortex/analyzers/native/html/document.py +389 -0
  30. docvortex/analyzers/native/html/errors.py +12 -0
  31. docvortex/analyzers/native/html/resources.py +365 -0
  32. docvortex/analyzers/native/html/selector.py +421 -0
  33. docvortex/analyzers/native/models.py +209 -0
  34. docvortex/analyzers/native/ofd/__init__.py +14 -0
  35. docvortex/analyzers/native/ofd/constants.py +59 -0
  36. docvortex/analyzers/native/ofd/converter.py +31 -0
  37. docvortex/analyzers/native/ofd/errors.py +16 -0
  38. docvortex/analyzers/native/ofd/geometry.py +215 -0
  39. docvortex/analyzers/native/ofd/images.py +124 -0
  40. docvortex/analyzers/native/ofd/metadata.py +53 -0
  41. docvortex/analyzers/native/ofd/models.py +170 -0
  42. docvortex/analyzers/native/ofd/package.py +346 -0
  43. docvortex/analyzers/native/ofd/path.py +201 -0
  44. docvortex/analyzers/native/ofd/reading_order.py +293 -0
  45. docvortex/analyzers/native/ofd/resources.py +122 -0
  46. docvortex/analyzers/native/ofd/scene.py +347 -0
  47. docvortex/analyzers/native/ofd/table.py +240 -0
  48. docvortex/analyzers/native/ofd/text.py +521 -0
  49. docvortex/analyzers/native/office/__init__.py +3 -0
  50. docvortex/analyzers/native/office/doc/__init__.py +3 -0
  51. docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
  52. docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
  53. docvortex/analyzers/native/office/doc/fib.py +261 -0
  54. docvortex/analyzers/native/office/doc/fields.py +101 -0
  55. docvortex/analyzers/native/office/doc/formatting.py +171 -0
  56. docvortex/analyzers/native/office/doc/images.py +162 -0
  57. docvortex/analyzers/native/office/doc/lists.py +356 -0
  58. docvortex/analyzers/native/office/doc/models.py +164 -0
  59. docvortex/analyzers/native/office/doc/parser.py +851 -0
  60. docvortex/analyzers/native/office/doc/pieces.py +256 -0
  61. docvortex/analyzers/native/office/doc/records.py +40 -0
  62. docvortex/analyzers/native/office/doc/sprm.py +269 -0
  63. docvortex/analyzers/native/office/doc/styles.py +215 -0
  64. docvortex/analyzers/native/office/docx/__init__.py +3 -0
  65. docvortex/analyzers/native/office/docx/context.py +71 -0
  66. docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
  67. docvortex/analyzers/native/office/docx/equationxml.py +119 -0
  68. docvortex/analyzers/native/office/docx/fields.py +816 -0
  69. docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
  70. docvortex/analyzers/native/office/docx/main.py +50 -0
  71. docvortex/analyzers/native/office/docx/numbering.py +491 -0
  72. docvortex/analyzers/native/office/docx/office_xml.py +57 -0
  73. docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
  74. docvortex/analyzers/native/office/docx/resources.py +526 -0
  75. docvortex/analyzers/native/office/docx/styles.py +693 -0
  76. docvortex/analyzers/native/office/docx/tables.py +515 -0
  77. docvortex/analyzers/native/office/equation/__init__.py +3 -0
  78. docvortex/analyzers/native/office/equation/image.py +470 -0
  79. docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
  80. docvortex/analyzers/native/office/equation/mtef.py +885 -0
  81. docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
  82. docvortex/analyzers/native/office/equation/omml.py +561 -0
  83. docvortex/analyzers/native/office/equation/ooxml.py +62 -0
  84. docvortex/analyzers/native/office/errors.py +33 -0
  85. docvortex/analyzers/native/office/image.py +307 -0
  86. docvortex/analyzers/native/office/legacy/__init__.py +3 -0
  87. docvortex/analyzers/native/office/legacy/binary.py +43 -0
  88. docvortex/analyzers/native/office/legacy/officeart.py +362 -0
  89. docvortex/analyzers/native/office/legacy/ole.py +110 -0
  90. docvortex/analyzers/native/office/limits.py +12 -0
  91. docvortex/analyzers/native/office/odf/__init__.py +3 -0
  92. docvortex/analyzers/native/office/odf/chart.py +76 -0
  93. docvortex/analyzers/native/office/odf/constants.py +70 -0
  94. docvortex/analyzers/native/office/odf/converters.py +403 -0
  95. docvortex/analyzers/native/office/odf/errors.py +18 -0
  96. docvortex/analyzers/native/office/odf/metadata.py +91 -0
  97. docvortex/analyzers/native/office/odf/models.py +176 -0
  98. docvortex/analyzers/native/office/odf/package.py +270 -0
  99. docvortex/analyzers/native/office/odf/styles.py +329 -0
  100. docvortex/analyzers/native/office/odf/table.py +469 -0
  101. docvortex/analyzers/native/office/odf/text.py +1002 -0
  102. docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
  103. docvortex/analyzers/native/office/opc.py +38 -0
  104. docvortex/analyzers/native/office/ppt/__init__.py +3 -0
  105. docvortex/analyzers/native/office/ppt/models.py +118 -0
  106. docvortex/analyzers/native/office/ppt/parser.py +1895 -0
  107. docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
  108. docvortex/analyzers/native/office/ppt/records.py +131 -0
  109. docvortex/analyzers/native/office/ppt/style_text.py +247 -0
  110. docvortex/analyzers/native/office/pptx/__init__.py +3 -0
  111. docvortex/analyzers/native/office/pptx/context.py +113 -0
  112. docvortex/analyzers/native/office/pptx/lists.py +558 -0
  113. docvortex/analyzers/native/office/pptx/main.py +20 -0
  114. docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
  115. docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
  116. docvortex/analyzers/native/office/pptx/resources.py +329 -0
  117. docvortex/analyzers/native/office/pptx/shapes.py +393 -0
  118. docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
  119. docvortex/analyzers/native/office/pptx/titles.py +178 -0
  120. docvortex/analyzers/native/office/rich_text.py +420 -0
  121. docvortex/analyzers/native/office/rtf/__init__.py +3 -0
  122. docvortex/analyzers/native/office/rtf/converter.py +708 -0
  123. docvortex/analyzers/native/office/rtf/lexer.py +213 -0
  124. docvortex/analyzers/native/office/rtf/math.py +339 -0
  125. docvortex/analyzers/native/office/rtf/models.py +185 -0
  126. docvortex/analyzers/native/office/rtf/parser.py +1553 -0
  127. docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
  128. docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
  129. docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
  130. docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
  131. docvortex/analyzers/native/office/streams.py +18 -0
  132. docvortex/analyzers/native/office/xls/__init__.py +3 -0
  133. docvortex/analyzers/native/office/xls/chart.py +132 -0
  134. docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
  135. docvortex/analyzers/native/office/xls/models.py +109 -0
  136. docvortex/analyzers/native/office/xls/number_format.py +521 -0
  137. docvortex/analyzers/native/office/xls/parser.py +1145 -0
  138. docvortex/analyzers/native/office/xls/records.py +201 -0
  139. docvortex/analyzers/native/office/xls/strings.py +205 -0
  140. docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
  141. docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
  142. docvortex/analyzers/native/office/xlsx/main.py +20 -0
  143. docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
  144. docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
  145. docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
  146. docvortex/analyzers/native/pdf/__init__.py +3 -0
  147. docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
  148. docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
  149. docvortex/analyzers/native/pdf/code_blocks.py +535 -0
  150. docvortex/analyzers/native/pdf/formulas.py +1985 -0
  151. docvortex/analyzers/native/pdf/geometry.py +281 -0
  152. docvortex/analyzers/native/pdf/graphics.py +1501 -0
  153. docvortex/analyzers/native/pdf/index_blocks.py +268 -0
  154. docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
  155. docvortex/analyzers/native/pdf/inline/common.py +112 -0
  156. docvortex/analyzers/native/pdf/inline/detection.py +590 -0
  157. docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
  158. docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
  159. docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
  160. docvortex/analyzers/native/pdf/inline/types.py +385 -0
  161. docvortex/analyzers/native/pdf/line_layout.py +1106 -0
  162. docvortex/analyzers/native/pdf/line_merging.py +1223 -0
  163. docvortex/analyzers/native/pdf/models.py +246 -0
  164. docvortex/analyzers/native/pdf/native_text.py +1004 -0
  165. docvortex/analyzers/native/pdf/pipeline.py +1390 -0
  166. docvortex/analyzers/native/pdf/script_geometry.py +636 -0
  167. docvortex/analyzers/native/pdf/shared.py +30 -0
  168. docvortex/analyzers/native/pdf/spatial_text.py +383 -0
  169. docvortex/analyzers/native/pdf/table_annotations.py +446 -0
  170. docvortex/analyzers/native/pdf/table_constants.py +48 -0
  171. docvortex/analyzers/native/pdf/table_detection.py +227 -0
  172. docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
  173. docvortex/analyzers/native/pdf/table_geometry.py +40 -0
  174. docvortex/analyzers/native/pdf/table_materialization.py +444 -0
  175. docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
  176. docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
  177. docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
  178. docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
  179. docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
  180. docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
  181. docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
  182. docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
  183. docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
  184. docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
  185. docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
  186. docvortex/analyzers/native/pdf/table_rows.py +34 -0
  187. docvortex/analyzers/native/pdf/table_rules.py +1129 -0
  188. docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
  189. docvortex/analyzers/native/pdf/tables.py +147 -0
  190. docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
  191. docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
  192. docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
  193. docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
  194. docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
  195. docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
  196. docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
  197. docvortex/analyzers/native/pdf/text_blocks.py +82 -0
  198. docvortex/analyzers/native/pdf/text_styles.py +55 -0
  199. docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
  200. docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
  201. docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
  202. docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
  203. docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
  204. docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
  205. docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
  206. docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
  207. docvortex/analyzers/native/pdf/titles.py +75 -0
  208. docvortex/analyzers/native/pdf/typography.py +19 -0
  209. docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
  210. docvortex/api.py +180 -0
  211. docvortex/assets/__init__.py +5 -0
  212. docvortex/assets/store.py +51 -0
  213. docvortex/cli.py +54 -0
  214. docvortex/codecs/__init__.py +3 -0
  215. docvortex/codecs/html/__init__.py +22 -0
  216. docvortex/codecs/html/contracts.py +236 -0
  217. docvortex/codecs/html/materializer.py +331 -0
  218. docvortex/codecs/html/parser.py +763 -0
  219. docvortex/codecs/html/resources.py +34 -0
  220. docvortex/codecs/json.py +17 -0
  221. docvortex/content/__init__.py +5 -0
  222. docvortex/content/inline.py +248 -0
  223. docvortex/content/markup/__init__.py +44 -0
  224. docvortex/content/markup/anchors.py +188 -0
  225. docvortex/content/markup/formula.py +280 -0
  226. docvortex/content/markup/projector.py +1237 -0
  227. docvortex/content/markup/styles.py +327 -0
  228. docvortex/content/mathml.py +167 -0
  229. docvortex/content/normalization.py +188 -0
  230. docvortex/content/spans.py +183 -0
  231. docvortex/content/table/__init__.py +18 -0
  232. docvortex/content/table/blocks.py +152 -0
  233. docvortex/content/table/content.py +425 -0
  234. docvortex/content/table/document.py +104 -0
  235. docvortex/content/table/html.py +399 -0
  236. docvortex/content/table/models.py +76 -0
  237. docvortex/content/table/rules.py +42 -0
  238. docvortex/content/table/structure.py +221 -0
  239. docvortex/content/tree.py +5 -0
  240. docvortex/document/__init__.py +3 -0
  241. docvortex/document/contracts.py +22 -0
  242. docvortex/document/detection.py +389 -0
  243. docvortex/document/filetypes.py +175 -0
  244. docvortex/document/page_range.py +167 -0
  245. docvortex/document/pdf/__init__.py +19 -0
  246. docvortex/document/pdf/classify.py +1138 -0
  247. docvortex/document/pdf/constants.py +132 -0
  248. docvortex/document/pdf/diagnostics.py +350 -0
  249. docvortex/document/pdf/document.py +582 -0
  250. docvortex/document/pdf/font_runtime.py +335 -0
  251. docvortex/document/pdf/geometry.py +31 -0
  252. docvortex/document/pdf/images.py +654 -0
  253. docvortex/document/pdf/native_annotations.py +367 -0
  254. docvortex/document/pdf/native_contracts.py +169 -0
  255. docvortex/document/pdf/native_coordinates.py +216 -0
  256. docvortex/document/pdf/native_lifecycle.py +16 -0
  257. docvortex/document/pdf/native_objects.py +902 -0
  258. docvortex/document/pdf/native_text_geometry.py +315 -0
  259. docvortex/document/pdf/pdfium.py +325 -0
  260. docvortex/document/pdf/raster.py +46 -0
  261. docvortex/document/pdf/text/__init__.py +62 -0
  262. docvortex/document/pdf/text/contracts.py +211 -0
  263. docvortex/document/pdf/text/extract.py +165 -0
  264. docvortex/document/pdf/text/geometry.py +16 -0
  265. docvortex/document/pdf/text/groups.py +162 -0
  266. docvortex/document/pdf/visual_geometry.py +201 -0
  267. docvortex/document/pdf/visuals.py +343 -0
  268. docvortex/document/source.py +92 -0
  269. docvortex/errors.py +21 -0
  270. docvortex/export/__init__.py +3 -0
  271. docvortex/export/bundle.py +98 -0
  272. docvortex/export/files.py +66 -0
  273. docvortex/export/middle.py +208 -0
  274. docvortex/foundation/__init__.py +3 -0
  275. docvortex/foundation/geometry.py +125 -0
  276. docvortex/foundation/hyperlink.py +65 -0
  277. docvortex/foundation/image.py +48 -0
  278. docvortex/foundation/image_encoding.py +30 -0
  279. docvortex/foundation/image_payload.py +280 -0
  280. docvortex/foundation/language.py +92 -0
  281. docvortex/foundation/platform.py +38 -0
  282. docvortex/foundation/text.py +153 -0
  283. docvortex/foundation/type_identity.py +20 -0
  284. docvortex/foundation/xml_names.py +20 -0
  285. docvortex/options.py +30 -0
  286. docvortex/postprocess/__init__.py +3 -0
  287. docvortex/postprocess/content.py +53 -0
  288. docvortex/postprocess/document.py +19 -0
  289. docvortex/postprocess/lists.py +236 -0
  290. docvortex/postprocess/page_blocks.py +214 -0
  291. docvortex/postprocess/pages.py +95 -0
  292. docvortex/postprocess/paragraphs.py +580 -0
  293. docvortex/postprocess/visual.py +715 -0
  294. docvortex/render/__init__.py +48 -0
  295. docvortex/render/_internal/__init__.py +3 -0
  296. docvortex/render/_internal/common/__init__.py +3 -0
  297. docvortex/render/_internal/common/context.py +43 -0
  298. docvortex/render/_internal/common/html_table.py +178 -0
  299. docvortex/render/_internal/common/index.py +33 -0
  300. docvortex/render/_internal/common/list_items.py +158 -0
  301. docvortex/render/_internal/common/planner.py +140 -0
  302. docvortex/render/_internal/docx/__init__.py +3 -0
  303. docvortex/render/_internal/docx/assets.py +202 -0
  304. docvortex/render/_internal/docx/inline.py +434 -0
  305. docvortex/render/_internal/docx/math.py +220 -0
  306. docvortex/render/_internal/docx/renderer.py +905 -0
  307. docvortex/render/_internal/docx/styles.py +195 -0
  308. docvortex/render/_internal/docx/table.py +442 -0
  309. docvortex/render/_internal/epub/__init__.py +5 -0
  310. docvortex/render/_internal/epub/assets.py +173 -0
  311. docvortex/render/_internal/epub/package.py +249 -0
  312. docvortex/render/_internal/epub/renderer.py +1156 -0
  313. docvortex/render/_internal/html/__init__.py +3 -0
  314. docvortex/render/_internal/html/inline.py +346 -0
  315. docvortex/render/_internal/html/renderer.py +1041 -0
  316. docvortex/render/_internal/html/sanitizer.py +478 -0
  317. docvortex/render/_internal/html/table.py +121 -0
  318. docvortex/render/_internal/latex/__init__.py +1 -0
  319. docvortex/render/_internal/latex/assets.py +85 -0
  320. docvortex/render/_internal/latex/inline.py +145 -0
  321. docvortex/render/_internal/latex/renderer.py +506 -0
  322. docvortex/render/_internal/latex/table.py +347 -0
  323. docvortex/render/_internal/markdown/__init__.py +3 -0
  324. docvortex/render/_internal/markdown/assets.py +78 -0
  325. docvortex/render/_internal/markdown/blocks.py +635 -0
  326. docvortex/render/_internal/markdown/escaping.py +51 -0
  327. docvortex/render/_internal/markdown/inline.py +260 -0
  328. docvortex/render/_internal/markdown/renderer.py +93 -0
  329. docvortex/render/_internal/markdown/table.py +281 -0
  330. docvortex/render/_internal/pdf/__init__.py +3 -0
  331. docvortex/render/_internal/pdf/assets.py +197 -0
  332. docvortex/render/_internal/pdf/formula.py +417 -0
  333. docvortex/render/_internal/pdf/inline.py +343 -0
  334. docvortex/render/_internal/pdf/renderer.py +734 -0
  335. docvortex/render/_internal/pdf/styles.py +206 -0
  336. docvortex/render/_internal/pdf/table.py +272 -0
  337. docvortex/render/_internal/structured_content/__init__.py +3 -0
  338. docvortex/render/_internal/structured_content/renderer.py +193 -0
  339. docvortex/render/api.py +199 -0
  340. docvortex/render/contracts.py +205 -0
  341. docvortex/render/docx.py +38 -0
  342. docvortex/render/epub.py +35 -0
  343. docvortex/render/fragments.py +70 -0
  344. docvortex/render/html.py +29 -0
  345. docvortex/render/latex.py +24 -0
  346. docvortex/render/markdown.py +50 -0
  347. docvortex/render/pdf.py +25 -0
  348. docvortex/render/structured_content.py +23 -0
  349. docvortex/resources/epub/docvortex.css +91 -0
  350. docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
  351. docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
  352. docvortex/resources/fonts/NOTICE +190 -0
  353. docvortex/resources/fonts/manifest.json +9 -0
  354. docvortex/resources/html/docvortex.css +601 -0
  355. docvortex/resources/html/docvortex.min.css +1 -0
  356. docvortex/result.py +91 -0
  357. docvortex/schema.py +1141 -0
  358. docvortex/version.py +3 -0
  359. docvortex-0.2.1.dist-info/METADATA +193 -0
  360. docvortex-0.2.1.dist-info/RECORD +364 -0
  361. docvortex-0.2.1.dist-info/WHEEL +5 -0
  362. docvortex-0.2.1.dist-info/entry_points.txt +2 -0
  363. docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
  364. docvortex-0.2.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,399 @@
1
+ """HTML 表格解析、行列扫描和结构状态缓存。"""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Any
6
+
7
+ from bs4 import BeautifulSoup, Tag
8
+
9
+ from ...foundation.text import full_to_half
10
+
11
+ from .models import MAX_HEADER_ROWS, RenderedCellSegment, RowMetrics, RowScanResult, RowSignature, TableMergeState
12
+
13
+
14
+ def _colspan(cell: Any) -> int:
15
+ """读取 HTML 单元格 colspan,非法值交由上层安全降级。"""
16
+ val = cell.get("colspan", "1")
17
+ assert isinstance(val, str)
18
+ return int(val)
19
+
20
+
21
+ def _rowspan(cell: Any) -> int:
22
+ """读取 HTML 单元格 rowspan,非法值交由上层安全降级。"""
23
+ val = cell.get("rowspan", "1")
24
+ assert isinstance(val, str)
25
+ return int(val)
26
+
27
+
28
+ def _normalize_cell_text(cell: Tag) -> str:
29
+ """生成表头匹配使用的半角无空白文本。"""
30
+ return "".join(full_to_half(cell.get_text()).split())
31
+
32
+
33
+ def _display_cell_text(cell: Tag) -> str:
34
+ """生成保留内部空白的半角展示文本。"""
35
+ return full_to_half(cell.get_text().strip())
36
+
37
+
38
+ def _scan_rows(rows: list[Tag], initial_occupied: dict[int, set[int]] | None = None, start_row_idx: int = 0) -> RowScanResult:
39
+ """单次扫描 HTML 行并缓存有效列、显式列和跨行占位指标。
40
+
41
+ ``initial_occupied`` 使用相对首行的偏移记录未来行占位,从而保留跨越
42
+ 前后表边界的 rowspan 结构。
43
+ """
44
+ occupied: dict[int, dict[int, bool]] = {}
45
+ max_cols = 0
46
+
47
+ for row_offset, cols in (initial_occupied or {}).items():
48
+ if not cols:
49
+ continue
50
+ occupied[row_offset] = dict.fromkeys(cols, True)
51
+ max_cols = max(max_cols, max(cols) + 1)
52
+
53
+ row_effective_cols: list[int] = []
54
+ row_metrics: list[RowMetrics] = []
55
+ last_nonempty_row_metrics: RowMetrics | None = None
56
+
57
+ for local_idx, row in enumerate(rows):
58
+ occupied_row = occupied.setdefault(local_idx, {})
59
+ col_idx = 0
60
+ cells = row.find_all(["td", "th"])
61
+ actual_cols = 0
62
+
63
+ for cell in cells:
64
+ while col_idx in occupied_row:
65
+ col_idx += 1
66
+
67
+ colspan = _colspan(cell)
68
+ rowspan = _rowspan(cell)
69
+ actual_cols += colspan
70
+
71
+ for row_offset in range(rowspan):
72
+ target_idx = local_idx + row_offset
73
+ occupied_target = occupied.setdefault(target_idx, {})
74
+ for col in range(col_idx, col_idx + colspan):
75
+ occupied_target[col] = True
76
+
77
+ col_idx += colspan
78
+ max_cols = max(max_cols, col_idx)
79
+
80
+ effective_cols = max(occupied_row.keys()) + 1 if occupied_row else 0
81
+ row_effective_cols.append(effective_cols)
82
+ max_cols = max(max_cols, effective_cols)
83
+
84
+ metrics = RowMetrics(
85
+ row_idx=start_row_idx + local_idx,
86
+ effective_cols=effective_cols,
87
+ actual_cols=actual_cols,
88
+ visual_cols=len(cells),
89
+ )
90
+ row_metrics.append(metrics)
91
+ if cells:
92
+ last_nonempty_row_metrics = metrics
93
+
94
+ tail_occupied = {
95
+ row_idx - len(rows): set(cols.keys()) for row_idx, cols in occupied.items() if row_idx >= len(rows) and cols
96
+ }
97
+
98
+ return RowScanResult(
99
+ row_effective_cols=row_effective_cols,
100
+ row_metrics=row_metrics,
101
+ total_cols=max_cols,
102
+ last_nonempty_row_metrics=last_nonempty_row_metrics,
103
+ tail_occupied=tail_occupied,
104
+ )
105
+
106
+
107
+ def _build_row_signature(row: Tag, effective_cols: int) -> RowSignature:
108
+ """构建表头检测使用的行结构与文本签名。"""
109
+ cells = row.find_all(["td", "th"])
110
+ return RowSignature(
111
+ effective_cols=effective_cols,
112
+ colspans=tuple(_colspan(cell) for cell in cells),
113
+ rowspans=tuple(_rowspan(cell) for cell in cells),
114
+ normalized_texts=tuple(_normalize_cell_text(cell) for cell in cells),
115
+ display_texts=tuple(_display_cell_text(cell) for cell in cells),
116
+ )
117
+
118
+
119
+ def _build_front_cache(
120
+ rows: list[Tag], max_header_rows: int = MAX_HEADER_ROWS
121
+ ) -> tuple[list[RowSignature], dict[int, RowMetrics]]:
122
+ """缓存表格前部表头签名和首批数据行指标。"""
123
+ front_limit = min(len(rows), max_header_rows + 1)
124
+ front_rows = rows[:front_limit]
125
+ front_scan = _scan_rows(front_rows)
126
+
127
+ front_header_info = [
128
+ _build_row_signature(front_rows[idx], front_scan.row_effective_cols[idx])
129
+ for idx in range(min(len(front_rows), max_header_rows))
130
+ ]
131
+ front_first_data_row_metrics = dict(enumerate(front_scan.row_metrics))
132
+ return front_header_info, front_first_data_row_metrics
133
+
134
+
135
+ def _refresh_table_state_metrics(state: TableMergeState) -> None:
136
+ """HTML 结构调整后重新计算表格状态指标。"""
137
+ scan = _scan_rows(state.rows)
138
+ state.row_effective_cols = scan.row_effective_cols
139
+ state.total_cols = scan.total_cols
140
+ state.last_data_row_metrics = scan.last_nonempty_row_metrics
141
+ state.tail_occupied = scan.tail_occupied
142
+ state.front_header_info, state.front_first_data_row_metrics = _build_front_cache(state.rows)
143
+
144
+
145
+ def build_table_state_from_html(
146
+ html: str,
147
+ max_header_rows: int = MAX_HEADER_ROWS,
148
+ ) -> TableMergeState | None:
149
+ """从原始 HTML 构建 TableMergeState,不依赖 DocVortex block 结构。
150
+
151
+ 供外部工具(如 mineru-vl-utils)调用,用于跨页表格结构检测。
152
+ 返回的 state 供 HTML-only 结构 helper 使用,不包含 DocVortex block 所有者。
153
+ """
154
+ if not html:
155
+ return None
156
+
157
+ soup = BeautifulSoup(html, "html.parser")
158
+ tbody = soup.find("tbody") or soup.find("table")
159
+ rows = soup.find_all("tr")
160
+ if tbody is None or not rows:
161
+ return None
162
+
163
+ try:
164
+ scan = _scan_rows(rows)
165
+ front_header_info, front_first_data_row_metrics = _build_front_cache(
166
+ rows,
167
+ max_header_rows=max_header_rows,
168
+ )
169
+ except (AssertionError, TypeError, ValueError):
170
+ return None
171
+ if scan.total_cols <= 0 or scan.last_nonempty_row_metrics is None:
172
+ return None
173
+
174
+ return TableMergeState(
175
+ owner_block=None,
176
+ body_block=None,
177
+ soup=soup,
178
+ tbody=tbody,
179
+ rows=rows,
180
+ total_cols=scan.total_cols,
181
+ front_header_info=front_header_info,
182
+ front_first_data_row_metrics=front_first_data_row_metrics,
183
+ last_data_row_metrics=scan.last_nonempty_row_metrics,
184
+ row_effective_cols=scan.row_effective_cols,
185
+ tail_occupied=scan.tail_occupied,
186
+ )
187
+
188
+
189
+ def _serialize_table_state_html(state: TableMergeState) -> bool:
190
+ """将合并后的 BeautifulSoup 写回克隆表体,缺失表体时返回失败。"""
191
+ if state.body_block is None:
192
+ return False
193
+ state.body_block["content"] = str(state.soup)
194
+ state.dirty = False
195
+ return True
196
+
197
+
198
+ def calculate_table_total_columns(soup: BeautifulSoup) -> int:
199
+ """计算表格的总列数,通过分析整个表格结构来处理rowspan和colspan."""
200
+ rows = soup.find_all("tr")
201
+ return _scan_rows(rows).total_cols if rows else 0
202
+
203
+
204
+ def build_table_occupied_matrix(soup: BeautifulSoup) -> dict[int, int]:
205
+ """构建表格的占用矩阵,返回每行的有效列数."""
206
+ rows = soup.find_all("tr")
207
+ if not rows:
208
+ return {}
209
+
210
+ scan = _scan_rows(rows)
211
+ return dict(enumerate(scan.row_effective_cols))
212
+
213
+
214
+ def calculate_row_effective_columns(soup: BeautifulSoup, row_idx: int) -> int:
215
+ """计算指定行的有效列数(考虑rowspan占用)."""
216
+ row_effective_cols = build_table_occupied_matrix(soup)
217
+ return row_effective_cols.get(row_idx, 0)
218
+
219
+
220
+ def calculate_row_columns(row: Tag) -> int:
221
+ """计算表格行的实际列数,考虑colspan属性."""
222
+ cells = row.find_all(["td", "th"])
223
+ column_count = 0
224
+
225
+ for cell in cells:
226
+ colspan = _colspan(cell)
227
+ column_count += colspan
228
+
229
+ return column_count
230
+
231
+
232
+ def calculate_visual_columns(row: Tag) -> int:
233
+ """计算表格行的视觉列数(实际td/th单元格数量,不考虑colspan)."""
234
+ cells = row.find_all(["td", "th"])
235
+ return len(cells)
236
+
237
+
238
+ def _scan_row_visual_sources(
239
+ rows: list[Tag],
240
+ target_row_index: int,
241
+ initial_occupied: dict[int, set[int]] | None = None,
242
+ ) -> tuple[dict[int, tuple[int, int]], int]:
243
+ """扫描到目标行,记录每个视觉列当前由哪个源单元格占据。
244
+
245
+ initial_occupied 表示从上一页延续过来的 rowspan 占位,行号相对
246
+ rows[0] 计算。它只作为虚拟源单元格参与列定位,不对应当前页真实
247
+ <td>/<th> 元素。
248
+ """
249
+ if target_row_index < 0:
250
+ target_row_index += len(rows)
251
+ if target_row_index < 0 or target_row_index >= len(rows):
252
+ return {}, 0
253
+
254
+ # occupied[row_idx][col_idx] = (source_row_idx, source_cell_idx)
255
+ occupied: dict[int, dict[int, tuple[int, int]]] = {}
256
+ total_cols = 0
257
+ for row_offset, cols in (initial_occupied or {}).items():
258
+ if not cols:
259
+ continue
260
+ occupied[row_offset] = {col: (-1, col) for col in cols}
261
+ total_cols = max(total_cols, max(cols) + 1)
262
+
263
+ for r_idx in range(target_row_index + 1):
264
+ occupied_row = occupied.setdefault(r_idx, {})
265
+ col_idx = 0
266
+ cells = rows[r_idx].find_all(["td", "th"])
267
+ for cell_idx, cell in enumerate(cells):
268
+ while col_idx in occupied_row:
269
+ col_idx += 1
270
+ colspan = _colspan(cell)
271
+ rowspan = _rowspan(cell)
272
+ source_marker = (r_idx, cell_idx)
273
+ for ro in range(rowspan):
274
+ target_idx = r_idx + ro
275
+ occ = occupied.setdefault(target_idx, {})
276
+ for c in range(col_idx, col_idx + colspan):
277
+ occ[c] = source_marker
278
+ col_idx += colspan
279
+ total_cols = max(total_cols, col_idx)
280
+
281
+ return occupied.get(target_row_index, {}), total_cols
282
+
283
+
284
+ def build_visual_col_mapping(
285
+ rows: list[Tag],
286
+ target_row_index: int,
287
+ initial_occupied: dict[int, set[int]] | None = None,
288
+ ) -> list[int]:
289
+ """构建目标行中每个显式 <td>/<th> 元素到视觉列位置的映射。
290
+
291
+ 该映射会正确考虑从前序行继承而来的 rowspan 占位。
292
+ initial_occupied 可额外传入上一页延续到当前切片的 rowspan 占位。
293
+ """
294
+ if target_row_index < 0:
295
+ target_row_index += len(rows)
296
+ if target_row_index < 0 or target_row_index >= len(rows):
297
+ return []
298
+
299
+ target_occupied, _ = _scan_row_visual_sources(
300
+ rows,
301
+ target_row_index,
302
+ initial_occupied=initial_occupied,
303
+ )
304
+
305
+ col_idx = 0
306
+ mapping = []
307
+ target_cells = rows[target_row_index].find_all(["td", "th"])
308
+ for cell in target_cells:
309
+ while col_idx in target_occupied and target_occupied[col_idx][0] < target_row_index:
310
+ col_idx += 1
311
+ mapping.append(col_idx)
312
+ colspan = _colspan(cell)
313
+ col_idx += colspan
314
+ return mapping
315
+
316
+
317
+ def build_row_rendered_cell_segments(
318
+ rows: list[Tag],
319
+ target_row_index: int,
320
+ initial_occupied: dict[int, set[int]] | None = None,
321
+ ) -> list[RenderedCellSegment]:
322
+ """构建目标行的渲染单元格段,保留每段覆盖的视觉列范围。
323
+
324
+ 该函数复用表格行视觉来源扫描结果,语义与 calculate_row_rendered_segments()
325
+ 保持一致:colspan 只算一个渲染段,rowspan 延续下来的单元格也会作为
326
+ 目标行的渲染段返回。
327
+ """
328
+ if target_row_index < 0:
329
+ target_row_index += len(rows)
330
+ if target_row_index < 0 or target_row_index >= len(rows):
331
+ return []
332
+
333
+ target_occupied, total_cols = _scan_row_visual_sources(
334
+ rows,
335
+ target_row_index,
336
+ initial_occupied=initial_occupied,
337
+ )
338
+ if total_cols == 0:
339
+ return []
340
+
341
+ segments: list[RenderedCellSegment] = []
342
+ current_marker: tuple[int, int] | None = None
343
+ current_start_col: int | None = None
344
+ current_text = ""
345
+
346
+ # 连续视觉列来自同一个源单元格时,合并为一个渲染段。
347
+ for col_idx in range(total_cols):
348
+ marker = target_occupied.get(col_idx)
349
+ if marker is None:
350
+ if current_marker is not None and current_start_col is not None:
351
+ segments.append(RenderedCellSegment(text=current_text, start_col=current_start_col, end_col=col_idx))
352
+ current_marker = None
353
+ current_start_col = None
354
+ current_text = ""
355
+ continue
356
+
357
+ if marker != current_marker:
358
+ if current_marker is not None and current_start_col is not None:
359
+ segments.append(RenderedCellSegment(text=current_text, start_col=current_start_col, end_col=col_idx))
360
+ current_marker = marker
361
+ current_start_col = col_idx
362
+ source_row_idx, source_cell_idx = marker
363
+ current_text = ""
364
+ if source_row_idx >= 0:
365
+ source_cells = rows[source_row_idx].find_all(["td", "th"])
366
+ if source_cell_idx < len(source_cells):
367
+ current_text = _display_cell_text(source_cells[source_cell_idx])
368
+
369
+ if current_marker is not None and current_start_col is not None:
370
+ segments.append(RenderedCellSegment(text=current_text, start_col=current_start_col, end_col=total_cols))
371
+
372
+ return segments
373
+
374
+
375
+ def calculate_row_rendered_segments(rows: list[Tag], target_row_index: int) -> int:
376
+ """计算目标行渲染后的视觉段数。
377
+
378
+ 段数按“渲染出来的单元格块”统计:
379
+ - 当前行显式单元格各算一段,不展开 colspan
380
+ - 从前序行继承而来的 rowspan 占位也算段
381
+ - 只有连续列且来自同一个源单元格时才算同一段
382
+ """
383
+ target_occupied, total_cols = _scan_row_visual_sources(rows, target_row_index)
384
+ if total_cols == 0:
385
+ return 0
386
+
387
+ segment_count = 0
388
+ previous_marker: tuple[int, int] | None = None
389
+
390
+ for col_idx in range(total_cols):
391
+ marker = target_occupied.get(col_idx)
392
+ if marker is None:
393
+ previous_marker = None
394
+ continue
395
+ if marker != previous_marker:
396
+ segment_count += 1
397
+ previous_marker = marker
398
+
399
+ return segment_count
@@ -0,0 +1,76 @@
1
+ """跨页表格合并使用的内部状态模型。"""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import dataclass
6
+ from typing import Any, TypeAlias
7
+
8
+ MAX_HEADER_ROWS = 5
9
+
10
+ BlockDict: TypeAlias = dict[str, Any]
11
+ PageInfoDict: TypeAlias = dict[str, Any]
12
+ CalculationBBox: TypeAlias = tuple[int, int, int, int]
13
+
14
+
15
+ @dataclass
16
+ class RowMetrics:
17
+ """记录单行的有效列、实际列和视觉列指标。"""
18
+
19
+ row_idx: int
20
+ effective_cols: int
21
+ actual_cols: int
22
+ visual_cols: int
23
+
24
+
25
+ @dataclass
26
+ class RowSignature:
27
+ """记录表头行的列结构与规范化文本签名。"""
28
+
29
+ effective_cols: int
30
+ colspans: tuple[int, ...]
31
+ rowspans: tuple[int, ...]
32
+ normalized_texts: tuple[str, ...]
33
+ display_texts: tuple[str, ...]
34
+
35
+ @property
36
+ def cell_count(self) -> int:
37
+ """返回签名中的显式单元格数量。"""
38
+ return len(self.colspans)
39
+
40
+
41
+ @dataclass
42
+ class RenderedCellSegment:
43
+ """记录一个渲染单元格覆盖的视觉列区间。"""
44
+
45
+ text: str
46
+ start_col: int
47
+ end_col: int
48
+
49
+
50
+ @dataclass
51
+ class RowScanResult:
52
+ """封装一次 HTML 行扫描得到的列指标与跨行占位。"""
53
+
54
+ row_effective_cols: list[int]
55
+ row_metrics: list[RowMetrics]
56
+ total_cols: int
57
+ last_nonempty_row_metrics: RowMetrics | None
58
+ tail_occupied: dict[int, set[int]]
59
+
60
+
61
+ @dataclass
62
+ class TableMergeState:
63
+ """缓存单张表格的 block 所有者、HTML 树和结构指标。"""
64
+
65
+ owner_block: BlockDict | None
66
+ body_block: BlockDict | None
67
+ soup: Any
68
+ tbody: Any
69
+ rows: list[Any]
70
+ total_cols: int
71
+ front_header_info: list[RowSignature]
72
+ front_first_data_row_metrics: dict[int, RowMetrics]
73
+ last_data_row_metrics: RowMetrics | None
74
+ row_effective_cols: list[int]
75
+ tail_occupied: dict[int, set[int]]
76
+ dirty: bool = False
@@ -0,0 +1,42 @@
1
+ """跨页表格延续文本与 caption 的轻量判定规则。"""
2
+
3
+ from ...foundation.text import full_to_half
4
+
5
+ CONTINUATION_END_MARKERS = [
6
+ "(续)",
7
+ "(续表)",
8
+ "(续上表)",
9
+ "(continued)",
10
+ "(cont.)",
11
+ "(cont’d)",
12
+ "(…continued)",
13
+ "continued",
14
+ "续表",
15
+ ]
16
+
17
+ CONTINUATION_INLINE_MARKERS = [
18
+ "(continued)",
19
+ ]
20
+
21
+
22
+ def is_table_continuation_text(text: str) -> bool:
23
+ """判断文本是否表达续表语义,供表格归组和跨页合并共同复用。"""
24
+ continuation_text = full_to_half((text or "").strip()).lower()
25
+ if not continuation_text:
26
+ return False
27
+
28
+ return any(
29
+ _matches_continuation_end_marker(continuation_text, marker.lower()) for marker in CONTINUATION_END_MARKERS
30
+ ) or any(marker.lower() in continuation_text for marker in CONTINUATION_INLINE_MARKERS)
31
+
32
+
33
+ def _matches_continuation_end_marker(text: str, marker: str) -> bool:
34
+ """判断续表后缀是否按词边界命中,避免 discontinued 误命中 continued。"""
35
+ if not text.endswith(marker):
36
+ return False
37
+
38
+ if marker == "continued":
39
+ marker_start = len(text) - len(marker)
40
+ return marker_start == 0 or not text[marker_start - 1].isalpha()
41
+
42
+ return True