docvortex 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (364) hide show
  1. docvortex/__init__.py +21 -0
  2. docvortex/analyzers/__init__.py +3 -0
  3. docvortex/analyzers/native/__init__.py +37 -0
  4. docvortex/analyzers/native/_shared/__init__.py +3 -0
  5. docvortex/analyzers/native/_shared/hyperlink.py +16 -0
  6. docvortex/analyzers/native/_shared/image.py +8 -0
  7. docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
  8. docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
  9. docvortex/analyzers/native/_shared/markup/formula.py +37 -0
  10. docvortex/analyzers/native/_shared/markup/projector.py +57 -0
  11. docvortex/analyzers/native/_shared/markup/styles.py +19 -0
  12. docvortex/analyzers/native/_shared/mathml.py +17 -0
  13. docvortex/analyzers/native/_shared/names.py +7 -0
  14. docvortex/analyzers/native/_shared/xycut.py +414 -0
  15. docvortex/analyzers/native/contracts.py +44 -0
  16. docvortex/analyzers/native/csv.py +351 -0
  17. docvortex/analyzers/native/epub/__init__.py +18 -0
  18. docvortex/analyzers/native/epub/constants.py +45 -0
  19. docvortex/analyzers/native/epub/converter.py +80 -0
  20. docvortex/analyzers/native/epub/errors.py +20 -0
  21. docvortex/analyzers/native/epub/metadata.py +31 -0
  22. docvortex/analyzers/native/epub/package.py +649 -0
  23. docvortex/analyzers/native/epub/xhtml.py +306 -0
  24. docvortex/analyzers/native/html/__init__.py +6 -0
  25. docvortex/analyzers/native/html/anchors.py +271 -0
  26. docvortex/analyzers/native/html/constants.py +25 -0
  27. docvortex/analyzers/native/html/contracts.py +7 -0
  28. docvortex/analyzers/native/html/converter.py +117 -0
  29. docvortex/analyzers/native/html/document.py +389 -0
  30. docvortex/analyzers/native/html/errors.py +12 -0
  31. docvortex/analyzers/native/html/resources.py +365 -0
  32. docvortex/analyzers/native/html/selector.py +421 -0
  33. docvortex/analyzers/native/models.py +209 -0
  34. docvortex/analyzers/native/ofd/__init__.py +14 -0
  35. docvortex/analyzers/native/ofd/constants.py +59 -0
  36. docvortex/analyzers/native/ofd/converter.py +31 -0
  37. docvortex/analyzers/native/ofd/errors.py +16 -0
  38. docvortex/analyzers/native/ofd/geometry.py +215 -0
  39. docvortex/analyzers/native/ofd/images.py +124 -0
  40. docvortex/analyzers/native/ofd/metadata.py +53 -0
  41. docvortex/analyzers/native/ofd/models.py +170 -0
  42. docvortex/analyzers/native/ofd/package.py +346 -0
  43. docvortex/analyzers/native/ofd/path.py +201 -0
  44. docvortex/analyzers/native/ofd/reading_order.py +293 -0
  45. docvortex/analyzers/native/ofd/resources.py +122 -0
  46. docvortex/analyzers/native/ofd/scene.py +347 -0
  47. docvortex/analyzers/native/ofd/table.py +240 -0
  48. docvortex/analyzers/native/ofd/text.py +521 -0
  49. docvortex/analyzers/native/office/__init__.py +3 -0
  50. docvortex/analyzers/native/office/doc/__init__.py +3 -0
  51. docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
  52. docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
  53. docvortex/analyzers/native/office/doc/fib.py +261 -0
  54. docvortex/analyzers/native/office/doc/fields.py +101 -0
  55. docvortex/analyzers/native/office/doc/formatting.py +171 -0
  56. docvortex/analyzers/native/office/doc/images.py +162 -0
  57. docvortex/analyzers/native/office/doc/lists.py +356 -0
  58. docvortex/analyzers/native/office/doc/models.py +164 -0
  59. docvortex/analyzers/native/office/doc/parser.py +851 -0
  60. docvortex/analyzers/native/office/doc/pieces.py +256 -0
  61. docvortex/analyzers/native/office/doc/records.py +40 -0
  62. docvortex/analyzers/native/office/doc/sprm.py +269 -0
  63. docvortex/analyzers/native/office/doc/styles.py +215 -0
  64. docvortex/analyzers/native/office/docx/__init__.py +3 -0
  65. docvortex/analyzers/native/office/docx/context.py +71 -0
  66. docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
  67. docvortex/analyzers/native/office/docx/equationxml.py +119 -0
  68. docvortex/analyzers/native/office/docx/fields.py +816 -0
  69. docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
  70. docvortex/analyzers/native/office/docx/main.py +50 -0
  71. docvortex/analyzers/native/office/docx/numbering.py +491 -0
  72. docvortex/analyzers/native/office/docx/office_xml.py +57 -0
  73. docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
  74. docvortex/analyzers/native/office/docx/resources.py +526 -0
  75. docvortex/analyzers/native/office/docx/styles.py +693 -0
  76. docvortex/analyzers/native/office/docx/tables.py +515 -0
  77. docvortex/analyzers/native/office/equation/__init__.py +3 -0
  78. docvortex/analyzers/native/office/equation/image.py +470 -0
  79. docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
  80. docvortex/analyzers/native/office/equation/mtef.py +885 -0
  81. docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
  82. docvortex/analyzers/native/office/equation/omml.py +561 -0
  83. docvortex/analyzers/native/office/equation/ooxml.py +62 -0
  84. docvortex/analyzers/native/office/errors.py +33 -0
  85. docvortex/analyzers/native/office/image.py +307 -0
  86. docvortex/analyzers/native/office/legacy/__init__.py +3 -0
  87. docvortex/analyzers/native/office/legacy/binary.py +43 -0
  88. docvortex/analyzers/native/office/legacy/officeart.py +362 -0
  89. docvortex/analyzers/native/office/legacy/ole.py +110 -0
  90. docvortex/analyzers/native/office/limits.py +12 -0
  91. docvortex/analyzers/native/office/odf/__init__.py +3 -0
  92. docvortex/analyzers/native/office/odf/chart.py +76 -0
  93. docvortex/analyzers/native/office/odf/constants.py +70 -0
  94. docvortex/analyzers/native/office/odf/converters.py +403 -0
  95. docvortex/analyzers/native/office/odf/errors.py +18 -0
  96. docvortex/analyzers/native/office/odf/metadata.py +91 -0
  97. docvortex/analyzers/native/office/odf/models.py +176 -0
  98. docvortex/analyzers/native/office/odf/package.py +270 -0
  99. docvortex/analyzers/native/office/odf/styles.py +329 -0
  100. docvortex/analyzers/native/office/odf/table.py +469 -0
  101. docvortex/analyzers/native/office/odf/text.py +1002 -0
  102. docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
  103. docvortex/analyzers/native/office/opc.py +38 -0
  104. docvortex/analyzers/native/office/ppt/__init__.py +3 -0
  105. docvortex/analyzers/native/office/ppt/models.py +118 -0
  106. docvortex/analyzers/native/office/ppt/parser.py +1895 -0
  107. docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
  108. docvortex/analyzers/native/office/ppt/records.py +131 -0
  109. docvortex/analyzers/native/office/ppt/style_text.py +247 -0
  110. docvortex/analyzers/native/office/pptx/__init__.py +3 -0
  111. docvortex/analyzers/native/office/pptx/context.py +113 -0
  112. docvortex/analyzers/native/office/pptx/lists.py +558 -0
  113. docvortex/analyzers/native/office/pptx/main.py +20 -0
  114. docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
  115. docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
  116. docvortex/analyzers/native/office/pptx/resources.py +329 -0
  117. docvortex/analyzers/native/office/pptx/shapes.py +393 -0
  118. docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
  119. docvortex/analyzers/native/office/pptx/titles.py +178 -0
  120. docvortex/analyzers/native/office/rich_text.py +420 -0
  121. docvortex/analyzers/native/office/rtf/__init__.py +3 -0
  122. docvortex/analyzers/native/office/rtf/converter.py +708 -0
  123. docvortex/analyzers/native/office/rtf/lexer.py +213 -0
  124. docvortex/analyzers/native/office/rtf/math.py +339 -0
  125. docvortex/analyzers/native/office/rtf/models.py +185 -0
  126. docvortex/analyzers/native/office/rtf/parser.py +1553 -0
  127. docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
  128. docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
  129. docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
  130. docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
  131. docvortex/analyzers/native/office/streams.py +18 -0
  132. docvortex/analyzers/native/office/xls/__init__.py +3 -0
  133. docvortex/analyzers/native/office/xls/chart.py +132 -0
  134. docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
  135. docvortex/analyzers/native/office/xls/models.py +109 -0
  136. docvortex/analyzers/native/office/xls/number_format.py +521 -0
  137. docvortex/analyzers/native/office/xls/parser.py +1145 -0
  138. docvortex/analyzers/native/office/xls/records.py +201 -0
  139. docvortex/analyzers/native/office/xls/strings.py +205 -0
  140. docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
  141. docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
  142. docvortex/analyzers/native/office/xlsx/main.py +20 -0
  143. docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
  144. docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
  145. docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
  146. docvortex/analyzers/native/pdf/__init__.py +3 -0
  147. docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
  148. docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
  149. docvortex/analyzers/native/pdf/code_blocks.py +535 -0
  150. docvortex/analyzers/native/pdf/formulas.py +1985 -0
  151. docvortex/analyzers/native/pdf/geometry.py +281 -0
  152. docvortex/analyzers/native/pdf/graphics.py +1501 -0
  153. docvortex/analyzers/native/pdf/index_blocks.py +268 -0
  154. docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
  155. docvortex/analyzers/native/pdf/inline/common.py +112 -0
  156. docvortex/analyzers/native/pdf/inline/detection.py +590 -0
  157. docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
  158. docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
  159. docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
  160. docvortex/analyzers/native/pdf/inline/types.py +385 -0
  161. docvortex/analyzers/native/pdf/line_layout.py +1106 -0
  162. docvortex/analyzers/native/pdf/line_merging.py +1223 -0
  163. docvortex/analyzers/native/pdf/models.py +246 -0
  164. docvortex/analyzers/native/pdf/native_text.py +1004 -0
  165. docvortex/analyzers/native/pdf/pipeline.py +1390 -0
  166. docvortex/analyzers/native/pdf/script_geometry.py +636 -0
  167. docvortex/analyzers/native/pdf/shared.py +30 -0
  168. docvortex/analyzers/native/pdf/spatial_text.py +383 -0
  169. docvortex/analyzers/native/pdf/table_annotations.py +446 -0
  170. docvortex/analyzers/native/pdf/table_constants.py +48 -0
  171. docvortex/analyzers/native/pdf/table_detection.py +227 -0
  172. docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
  173. docvortex/analyzers/native/pdf/table_geometry.py +40 -0
  174. docvortex/analyzers/native/pdf/table_materialization.py +444 -0
  175. docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
  176. docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
  177. docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
  178. docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
  179. docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
  180. docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
  181. docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
  182. docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
  183. docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
  184. docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
  185. docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
  186. docvortex/analyzers/native/pdf/table_rows.py +34 -0
  187. docvortex/analyzers/native/pdf/table_rules.py +1129 -0
  188. docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
  189. docvortex/analyzers/native/pdf/tables.py +147 -0
  190. docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
  191. docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
  192. docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
  193. docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
  194. docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
  195. docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
  196. docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
  197. docvortex/analyzers/native/pdf/text_blocks.py +82 -0
  198. docvortex/analyzers/native/pdf/text_styles.py +55 -0
  199. docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
  200. docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
  201. docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
  202. docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
  203. docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
  204. docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
  205. docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
  206. docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
  207. docvortex/analyzers/native/pdf/titles.py +75 -0
  208. docvortex/analyzers/native/pdf/typography.py +19 -0
  209. docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
  210. docvortex/api.py +180 -0
  211. docvortex/assets/__init__.py +5 -0
  212. docvortex/assets/store.py +51 -0
  213. docvortex/cli.py +54 -0
  214. docvortex/codecs/__init__.py +3 -0
  215. docvortex/codecs/html/__init__.py +22 -0
  216. docvortex/codecs/html/contracts.py +236 -0
  217. docvortex/codecs/html/materializer.py +331 -0
  218. docvortex/codecs/html/parser.py +763 -0
  219. docvortex/codecs/html/resources.py +34 -0
  220. docvortex/codecs/json.py +17 -0
  221. docvortex/content/__init__.py +5 -0
  222. docvortex/content/inline.py +248 -0
  223. docvortex/content/markup/__init__.py +44 -0
  224. docvortex/content/markup/anchors.py +188 -0
  225. docvortex/content/markup/formula.py +280 -0
  226. docvortex/content/markup/projector.py +1237 -0
  227. docvortex/content/markup/styles.py +327 -0
  228. docvortex/content/mathml.py +167 -0
  229. docvortex/content/normalization.py +188 -0
  230. docvortex/content/spans.py +183 -0
  231. docvortex/content/table/__init__.py +18 -0
  232. docvortex/content/table/blocks.py +152 -0
  233. docvortex/content/table/content.py +425 -0
  234. docvortex/content/table/document.py +104 -0
  235. docvortex/content/table/html.py +399 -0
  236. docvortex/content/table/models.py +76 -0
  237. docvortex/content/table/rules.py +42 -0
  238. docvortex/content/table/structure.py +221 -0
  239. docvortex/content/tree.py +5 -0
  240. docvortex/document/__init__.py +3 -0
  241. docvortex/document/contracts.py +22 -0
  242. docvortex/document/detection.py +389 -0
  243. docvortex/document/filetypes.py +175 -0
  244. docvortex/document/page_range.py +167 -0
  245. docvortex/document/pdf/__init__.py +19 -0
  246. docvortex/document/pdf/classify.py +1138 -0
  247. docvortex/document/pdf/constants.py +132 -0
  248. docvortex/document/pdf/diagnostics.py +350 -0
  249. docvortex/document/pdf/document.py +582 -0
  250. docvortex/document/pdf/font_runtime.py +335 -0
  251. docvortex/document/pdf/geometry.py +31 -0
  252. docvortex/document/pdf/images.py +654 -0
  253. docvortex/document/pdf/native_annotations.py +367 -0
  254. docvortex/document/pdf/native_contracts.py +169 -0
  255. docvortex/document/pdf/native_coordinates.py +216 -0
  256. docvortex/document/pdf/native_lifecycle.py +16 -0
  257. docvortex/document/pdf/native_objects.py +902 -0
  258. docvortex/document/pdf/native_text_geometry.py +315 -0
  259. docvortex/document/pdf/pdfium.py +325 -0
  260. docvortex/document/pdf/raster.py +46 -0
  261. docvortex/document/pdf/text/__init__.py +62 -0
  262. docvortex/document/pdf/text/contracts.py +211 -0
  263. docvortex/document/pdf/text/extract.py +165 -0
  264. docvortex/document/pdf/text/geometry.py +16 -0
  265. docvortex/document/pdf/text/groups.py +162 -0
  266. docvortex/document/pdf/visual_geometry.py +201 -0
  267. docvortex/document/pdf/visuals.py +343 -0
  268. docvortex/document/source.py +92 -0
  269. docvortex/errors.py +21 -0
  270. docvortex/export/__init__.py +3 -0
  271. docvortex/export/bundle.py +98 -0
  272. docvortex/export/files.py +66 -0
  273. docvortex/export/middle.py +208 -0
  274. docvortex/foundation/__init__.py +3 -0
  275. docvortex/foundation/geometry.py +125 -0
  276. docvortex/foundation/hyperlink.py +65 -0
  277. docvortex/foundation/image.py +48 -0
  278. docvortex/foundation/image_encoding.py +30 -0
  279. docvortex/foundation/image_payload.py +280 -0
  280. docvortex/foundation/language.py +92 -0
  281. docvortex/foundation/platform.py +38 -0
  282. docvortex/foundation/text.py +153 -0
  283. docvortex/foundation/type_identity.py +20 -0
  284. docvortex/foundation/xml_names.py +20 -0
  285. docvortex/options.py +30 -0
  286. docvortex/postprocess/__init__.py +3 -0
  287. docvortex/postprocess/content.py +53 -0
  288. docvortex/postprocess/document.py +19 -0
  289. docvortex/postprocess/lists.py +236 -0
  290. docvortex/postprocess/page_blocks.py +214 -0
  291. docvortex/postprocess/pages.py +95 -0
  292. docvortex/postprocess/paragraphs.py +580 -0
  293. docvortex/postprocess/visual.py +715 -0
  294. docvortex/render/__init__.py +48 -0
  295. docvortex/render/_internal/__init__.py +3 -0
  296. docvortex/render/_internal/common/__init__.py +3 -0
  297. docvortex/render/_internal/common/context.py +43 -0
  298. docvortex/render/_internal/common/html_table.py +178 -0
  299. docvortex/render/_internal/common/index.py +33 -0
  300. docvortex/render/_internal/common/list_items.py +158 -0
  301. docvortex/render/_internal/common/planner.py +140 -0
  302. docvortex/render/_internal/docx/__init__.py +3 -0
  303. docvortex/render/_internal/docx/assets.py +202 -0
  304. docvortex/render/_internal/docx/inline.py +434 -0
  305. docvortex/render/_internal/docx/math.py +220 -0
  306. docvortex/render/_internal/docx/renderer.py +905 -0
  307. docvortex/render/_internal/docx/styles.py +195 -0
  308. docvortex/render/_internal/docx/table.py +442 -0
  309. docvortex/render/_internal/epub/__init__.py +5 -0
  310. docvortex/render/_internal/epub/assets.py +173 -0
  311. docvortex/render/_internal/epub/package.py +249 -0
  312. docvortex/render/_internal/epub/renderer.py +1156 -0
  313. docvortex/render/_internal/html/__init__.py +3 -0
  314. docvortex/render/_internal/html/inline.py +346 -0
  315. docvortex/render/_internal/html/renderer.py +1041 -0
  316. docvortex/render/_internal/html/sanitizer.py +478 -0
  317. docvortex/render/_internal/html/table.py +121 -0
  318. docvortex/render/_internal/latex/__init__.py +1 -0
  319. docvortex/render/_internal/latex/assets.py +85 -0
  320. docvortex/render/_internal/latex/inline.py +145 -0
  321. docvortex/render/_internal/latex/renderer.py +506 -0
  322. docvortex/render/_internal/latex/table.py +347 -0
  323. docvortex/render/_internal/markdown/__init__.py +3 -0
  324. docvortex/render/_internal/markdown/assets.py +78 -0
  325. docvortex/render/_internal/markdown/blocks.py +635 -0
  326. docvortex/render/_internal/markdown/escaping.py +51 -0
  327. docvortex/render/_internal/markdown/inline.py +260 -0
  328. docvortex/render/_internal/markdown/renderer.py +93 -0
  329. docvortex/render/_internal/markdown/table.py +281 -0
  330. docvortex/render/_internal/pdf/__init__.py +3 -0
  331. docvortex/render/_internal/pdf/assets.py +197 -0
  332. docvortex/render/_internal/pdf/formula.py +417 -0
  333. docvortex/render/_internal/pdf/inline.py +343 -0
  334. docvortex/render/_internal/pdf/renderer.py +734 -0
  335. docvortex/render/_internal/pdf/styles.py +206 -0
  336. docvortex/render/_internal/pdf/table.py +272 -0
  337. docvortex/render/_internal/structured_content/__init__.py +3 -0
  338. docvortex/render/_internal/structured_content/renderer.py +193 -0
  339. docvortex/render/api.py +199 -0
  340. docvortex/render/contracts.py +205 -0
  341. docvortex/render/docx.py +38 -0
  342. docvortex/render/epub.py +35 -0
  343. docvortex/render/fragments.py +70 -0
  344. docvortex/render/html.py +29 -0
  345. docvortex/render/latex.py +24 -0
  346. docvortex/render/markdown.py +50 -0
  347. docvortex/render/pdf.py +25 -0
  348. docvortex/render/structured_content.py +23 -0
  349. docvortex/resources/epub/docvortex.css +91 -0
  350. docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
  351. docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
  352. docvortex/resources/fonts/NOTICE +190 -0
  353. docvortex/resources/fonts/manifest.json +9 -0
  354. docvortex/resources/html/docvortex.css +601 -0
  355. docvortex/resources/html/docvortex.min.css +1 -0
  356. docvortex/result.py +91 -0
  357. docvortex/schema.py +1141 -0
  358. docvortex/version.py +3 -0
  359. docvortex-0.2.1.dist-info/METADATA +193 -0
  360. docvortex-0.2.1.dist-info/RECORD +364 -0
  361. docvortex-0.2.1.dist-info/WHEEL +5 -0
  362. docvortex-0.2.1.dist-info/entry_points.txt +2 -0
  363. docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
  364. docvortex-0.2.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,425 @@
1
+ """跨页表格 HTML 内容、行列结构和单元格语义合并。"""
2
+
3
+ from __future__ import annotations
4
+
5
+ from copy import deepcopy
6
+
7
+ from bs4 import Tag
8
+
9
+ from ...schema import BlockType
10
+
11
+ from .blocks import _build_post_body_child_index, _build_table_state, _table_children
12
+ from .html import (
13
+ _colspan,
14
+ _refresh_table_state_metrics,
15
+ _rowspan,
16
+ _scan_rows,
17
+ _serialize_table_state_html,
18
+ build_visual_col_mapping,
19
+ calculate_row_columns,
20
+ calculate_visual_columns,
21
+ )
22
+ from .models import BlockDict, TableMergeState
23
+ from .structure import _expand_header_count_by_rowspan, can_merge_tables, check_row_columns_match, detect_table_headers
24
+
25
+
26
+ def adjust_table_rows_colspan(
27
+ rows: list[Tag],
28
+ start_idx: int,
29
+ end_idx: int,
30
+ row_effective_cols: list[int],
31
+ reference_structure: list[int],
32
+ reference_visual_cols: int,
33
+ target_cols: int,
34
+ match_reference_row: Tag,
35
+ ) -> None:
36
+ """调整表格行的colspan属性以匹配目标列数."""
37
+ reference_row_copy = deepcopy(match_reference_row)
38
+
39
+ for row_idx in range(start_idx, end_idx):
40
+ row = rows[row_idx]
41
+ cells = row.find_all(["td", "th"])
42
+ if not cells:
43
+ continue
44
+
45
+ current_row_effective_cols = row_effective_cols[row_idx]
46
+ current_row_cols = calculate_row_columns(row)
47
+
48
+ if current_row_effective_cols >= target_cols or current_row_cols >= target_cols:
49
+ continue
50
+
51
+ if calculate_visual_columns(row) == reference_visual_cols and check_row_columns_match(row, reference_row_copy):
52
+ if len(cells) <= len(reference_structure):
53
+ for cell_idx, cell in enumerate(cells):
54
+ if cell_idx < len(reference_structure) and reference_structure[cell_idx] > 1:
55
+ cell["colspan"] = str(reference_structure[cell_idx])
56
+ else:
57
+ cols_diff = target_cols - current_row_effective_cols
58
+ if cols_diff > 0:
59
+ last_cell = cells[-1]
60
+ current_last_span = _colspan(last_cell)
61
+ last_cell["colspan"] = str(current_last_span + cols_diff)
62
+
63
+
64
+ def _cell_has_semantic_content(cell: Tag) -> bool:
65
+ """判断单元格是否仍包含用户可见的语义内容。"""
66
+ if cell.get_text(strip=True):
67
+ return True
68
+
69
+ return cell.find(["img", "svg", "math", "eq", "table", "figure", "object", "embed", "canvas"]) is not None
70
+
71
+
72
+ def _row_has_semantic_content(row: Tag) -> bool:
73
+ """判断整行是否仍保留未并回的语义内容。"""
74
+ return any(_cell_has_semantic_content(cell) for cell in row.find_all(["td", "th"]))
75
+
76
+
77
+ def _insert_cell_before_visual_column(rows: list[Tag], target_row_index: int, start_vcol: int, cell: Tag) -> None:
78
+ """将单元格插入到目标行中对应视觉列之前。"""
79
+ target_row = rows[target_row_index]
80
+ target_cells = target_row.find_all(["td", "th"])
81
+ target_vcol_map = build_visual_col_mapping(rows, target_row_index)
82
+
83
+ for idx, target_start_vcol in enumerate(target_vcol_map):
84
+ if target_start_vcol >= start_vcol:
85
+ target_cells[idx].insert_before(cell)
86
+ return
87
+
88
+ target_row.append(cell)
89
+
90
+
91
+ def _carry_rowspan_structure_to_next_row(rows: list[Tag], row_idx: int) -> None:
92
+ """下沉空白结构占位单元格,避免删除当前行后破坏后续列对齐。"""
93
+ next_row_idx = row_idx + 1
94
+ if next_row_idx >= len(rows):
95
+ return
96
+
97
+ current_row = rows[row_idx]
98
+ current_cells = current_row.find_all(["td", "th"])
99
+ current_vcol_map = build_visual_col_mapping(rows, row_idx)
100
+ carried_cells = []
101
+
102
+ for cell, start_vcol in zip(current_cells, current_vcol_map):
103
+ rowspan = _rowspan(cell)
104
+ if rowspan <= 1 or _cell_has_semantic_content(cell):
105
+ continue
106
+
107
+ carried_cell = deepcopy(cell)
108
+ new_rowspan = rowspan - 1
109
+ if new_rowspan > 1:
110
+ carried_cell["rowspan"] = str(new_rowspan)
111
+ else:
112
+ carried_cell.attrs.pop("rowspan", None)
113
+ carried_cells.append((start_vcol, carried_cell))
114
+
115
+ for start_vcol, carried_cell in sorted(carried_cells, key=lambda item: item[0], reverse=True):
116
+ _insert_cell_before_visual_column(rows, next_row_idx, start_vcol, carried_cell)
117
+
118
+
119
+ def _clip_overlapped_blank_rowspan_cells(
120
+ rows: list[Tag],
121
+ initial_occupied: dict[int, set[int]],
122
+ ) -> bool:
123
+ """裁剪被上页 rowspan 覆盖的当前页空白结构占位。
124
+
125
+ 跨页表格中,上一页未结束的 rowspan 会通过 initial_occupied 占住
126
+ 当前页开头的视觉列。如果当前页表格识别又生成了同位置的空白
127
+ rowspan 单元格,这个单元格只是结构占位;直接拼接会把同一视觉列
128
+ 当成两列。这里仅裁剪无语义内容的空白占位,真实内容单元格不处理。
129
+ """
130
+ if not rows or not initial_occupied:
131
+ return False
132
+
133
+ cells_to_remove = []
134
+ cells_to_move = []
135
+
136
+ for row_idx, row in enumerate(rows):
137
+ cells = row.find_all(["td", "th"])
138
+ visual_col_map = build_visual_col_mapping(rows, row_idx)
139
+ for cell, start_vcol in zip(cells, visual_col_map):
140
+ rowspan = _rowspan(cell)
141
+ if rowspan <= 1 or _cell_has_semantic_content(cell):
142
+ continue
143
+
144
+ colspan = _colspan(cell)
145
+ occupied_cols = set(range(start_vcol, start_vcol + colspan))
146
+ if not occupied_cols:
147
+ continue
148
+
149
+ overlap_rows = 0
150
+ while overlap_rows < rowspan:
151
+ covered_cols = initial_occupied.get(row_idx + overlap_rows, set())
152
+ if not occupied_cols.issubset(covered_cols):
153
+ break
154
+ overlap_rows += 1
155
+
156
+ if overlap_rows == 0:
157
+ continue
158
+
159
+ remaining_rowspan = rowspan - overlap_rows
160
+ target_row_idx = row_idx + overlap_rows
161
+ if remaining_rowspan > 0 and target_row_idx >= len(rows):
162
+ continue
163
+
164
+ cells_to_remove.append(cell)
165
+ if remaining_rowspan > 0:
166
+ moved_cell = deepcopy(cell)
167
+ if remaining_rowspan > 1:
168
+ moved_cell["rowspan"] = str(remaining_rowspan)
169
+ else:
170
+ moved_cell.attrs.pop("rowspan", None)
171
+ cells_to_move.append((target_row_idx, start_vcol, moved_cell))
172
+
173
+ if not cells_to_remove:
174
+ return False
175
+
176
+ for cell in cells_to_remove:
177
+ cell.extract()
178
+
179
+ for target_row_idx, start_vcol, moved_cell in sorted(
180
+ cells_to_move,
181
+ key=lambda item: (item[0], item[1]),
182
+ reverse=True,
183
+ ):
184
+ _insert_cell_before_visual_column(rows, target_row_idx, start_vcol, moved_cell)
185
+
186
+ return True
187
+
188
+
189
+ def _apply_cell_merge(
190
+ previous_state: TableMergeState,
191
+ current_state: TableMergeState,
192
+ header_count: int,
193
+ ) -> bool:
194
+ """应用 cell_merge 语义合并。
195
+
196
+ 当 cell_merge 中的值为 1 时,将下表第一数据行对应单元格的内容
197
+ 追加到上表最后一行对应单元格中。全部为 1 时删除该数据行,
198
+ 混合时清空已合并单元格的内容但保留行。
199
+
200
+ cell_merge 按视觉列索引对齐,通过构建视觉列映射来正确匹配
201
+ 两个表格中可能因 rowspan 而具有不同 <td> 元素数量的行。
202
+ 元数据从当前页 table 根块读取,HTML 与列结构仍由唯一 table body 提供。
203
+ """
204
+ current_table_block = current_state.owner_block
205
+ if not isinstance(current_table_block, dict):
206
+ return False
207
+
208
+ cell_merge = current_table_block.get("cell_merge")
209
+ if not isinstance(cell_merge, list) or not cell_merge:
210
+ return False
211
+
212
+ rows2 = current_state.rows
213
+ if header_count >= len(rows2):
214
+ return False
215
+ if not previous_state.rows:
216
+ return False
217
+
218
+ first_data_row = rows2[header_count]
219
+ last_row = previous_state.rows[-1]
220
+
221
+ cells1 = last_row.find_all(["td", "th"])
222
+ cells2 = first_data_row.find_all(["td", "th"])
223
+
224
+ # 构建视觉列到单元格索引的映射
225
+ last_row_idx = len(previous_state.rows) - 1
226
+ vcol_map1 = build_visual_col_mapping(previous_state.rows, last_row_idx)
227
+ current_merge_rows = rows2[header_count:]
228
+ vcol_map2 = build_visual_col_mapping(
229
+ current_merge_rows,
230
+ 0,
231
+ initial_occupied=previous_state.tail_occupied,
232
+ )
233
+
234
+ # 构建视觉列 -> 单元格索引的反向映射(展开 colspan)
235
+ vcol_to_cell1: dict[int, int] = {}
236
+ for ci, start_vcol in enumerate(vcol_map1):
237
+ colspan = int(cells1[ci].get("colspan", 1))
238
+ for c in range(start_vcol, start_vcol + colspan):
239
+ vcol_to_cell1[c] = ci
240
+ vcol_to_cell2: dict[int, int] = {}
241
+ for ci, start_vcol in enumerate(vcol_map2):
242
+ colspan = int(cells2[ci].get("colspan", 1))
243
+ for c in range(start_vcol, start_vcol + colspan):
244
+ vcol_to_cell2[c] = ci
245
+
246
+ # 按唯一 (src_cell_idx, dst_cell_idx) 对执行一次转移,避免 colspan 重复处理
247
+ transferred_pairs: set[tuple[int, int]] = set()
248
+ for vi, merge_flag in enumerate(cell_merge):
249
+ if merge_flag == 1:
250
+ ci1 = vcol_to_cell1.get(vi)
251
+ ci2 = vcol_to_cell2.get(vi)
252
+ if ci1 is not None and ci2 is not None:
253
+ pair = (ci1, ci2)
254
+ if pair not in transferred_pairs:
255
+ for child in list(cells2[ci2].children):
256
+ cells1[ci1].append(child.extract())
257
+ transferred_pairs.add(pair)
258
+
259
+ # 只清空确实成功转移过的源单元格
260
+ cleared_ci2: set[int] = set()
261
+ for vi, merge_flag in enumerate(cell_merge):
262
+ if merge_flag == 1:
263
+ ci1 = vcol_to_cell1.get(vi)
264
+ ci2 = vcol_to_cell2.get(vi)
265
+ if ci1 is not None and ci2 is not None and ci2 not in cleared_ci2:
266
+ cells2[ci2].clear()
267
+ cleared_ci2.add(ci2)
268
+
269
+ if not _row_has_semantic_content(first_data_row):
270
+ _carry_rowspan_structure_to_next_row(rows2, header_count)
271
+ first_data_row.extract()
272
+ if first_data_row in rows2:
273
+ rows2.remove(first_data_row)
274
+
275
+ return bool(transferred_pairs)
276
+
277
+
278
+ def _perform_table_content_merge(
279
+ previous_state: TableMergeState,
280
+ current_state: TableMergeState,
281
+ previous_table_block: BlockDict,
282
+ current_table_block: BlockDict,
283
+ ) -> bool:
284
+ """在两个克隆表格上执行 HTML、单元格和 footnote 的内容合并。"""
285
+ header_count, _, _ = detect_table_headers(previous_state, current_state)
286
+ header_count = _expand_header_count_by_rowspan(current_state.rows, header_count)
287
+
288
+ rows1 = previous_state.rows
289
+ rows2 = current_state.rows
290
+ if not rows1 or header_count >= len(rows2):
291
+ return False
292
+
293
+ previous_adjusted = False
294
+
295
+ if header_count < len(rows2):
296
+ current_merge_rows = rows2[header_count:]
297
+ if _clip_overlapped_blank_rowspan_cells(current_merge_rows, previous_state.tail_occupied):
298
+ _refresh_table_state_metrics(current_state)
299
+
300
+ if rows1 and rows2 and header_count < len(rows2):
301
+ last_row1 = rows1[-1]
302
+ first_data_row2 = rows2[header_count]
303
+ table_cols1 = previous_state.total_cols
304
+ table_cols2 = current_state.total_cols
305
+
306
+ if table_cols1 > table_cols2:
307
+ reference_structure = [int(cell.get("colspan", 1)) for cell in last_row1.find_all(["td", "th"])]
308
+ reference_visual_cols = calculate_visual_columns(last_row1)
309
+ adjust_table_rows_colspan(
310
+ rows2,
311
+ header_count,
312
+ len(rows2),
313
+ current_state.row_effective_cols,
314
+ reference_structure,
315
+ reference_visual_cols,
316
+ table_cols1,
317
+ first_data_row2,
318
+ )
319
+ elif table_cols2 > table_cols1:
320
+ reference_structure = [int(cell.get("colspan", 1)) for cell in first_data_row2.find_all(["td", "th"])]
321
+ reference_visual_cols = calculate_visual_columns(first_data_row2)
322
+ adjust_table_rows_colspan(
323
+ rows1,
324
+ 0,
325
+ len(rows1),
326
+ previous_state.row_effective_cols,
327
+ reference_structure,
328
+ reference_visual_cols,
329
+ table_cols2,
330
+ last_row1,
331
+ )
332
+ previous_adjusted = True
333
+
334
+ if previous_adjusted:
335
+ _refresh_table_state_metrics(previous_state)
336
+
337
+ cell_merge_applied = _apply_cell_merge(previous_state, current_state, header_count)
338
+
339
+ appended_rows = rows2[header_count:]
340
+ append_start_idx = len(previous_state.rows)
341
+ merged_rows = []
342
+
343
+ if previous_state.tbody is None or current_state.tbody is None:
344
+ return False
345
+
346
+ for row in appended_rows:
347
+ row.extract()
348
+ previous_state.tbody.append(row)
349
+ merged_rows.append(row)
350
+
351
+ if not merged_rows and not cell_merge_applied:
352
+ return False
353
+
354
+ previous_state.rows.extend(merged_rows)
355
+
356
+ if merged_rows:
357
+ appended_scan = _scan_rows(
358
+ merged_rows,
359
+ initial_occupied=previous_state.tail_occupied,
360
+ start_row_idx=append_start_idx,
361
+ )
362
+ previous_state.row_effective_cols.extend(appended_scan.row_effective_cols)
363
+ previous_state.total_cols = max(previous_state.total_cols, appended_scan.total_cols)
364
+ if appended_scan.last_nonempty_row_metrics is not None:
365
+ previous_state.last_data_row_metrics = appended_scan.last_nonempty_row_metrics
366
+ previous_state.tail_occupied = appended_scan.tail_occupied
367
+
368
+ previous_content = previous_table_block.get("content")
369
+ if not isinstance(previous_content, list):
370
+ return False
371
+
372
+ previous_table_block["content"] = [
373
+ block for block in previous_content if not isinstance(block, dict) or block.get("type") != BlockType.TABLE_FOOTNOTE
374
+ ]
375
+ current_footnotes = [
376
+ block for block in _table_children(current_table_block) if block.get("type") == BlockType.TABLE_FOOTNOTE
377
+ ]
378
+ footnote_base_index = _build_post_body_child_index(previous_table_block, 0)
379
+ for footnote_offset, table_footnote in enumerate(current_footnotes, start=1):
380
+ temp_table_footnote = deepcopy(table_footnote)
381
+ temp_table_footnote.pop("_cross_page", None)
382
+ if footnote_base_index is None:
383
+ temp_table_footnote["index"] = 0
384
+ else:
385
+ temp_table_footnote["index"] = footnote_base_index + footnote_offset
386
+ previous_table_block["content"].append(temp_table_footnote)
387
+
388
+ previous_state.dirty = True
389
+ return _serialize_table_state_html(previous_state)
390
+
391
+
392
+ def merge_table_content(previous_table: BlockDict, current_table: BlockDict) -> BlockDict | None:
393
+ """纯函数式合并两张跨页表格的内容,失败时返回 ``None``。
394
+
395
+ 两个输入都会先深拷贝;返回块保留前表外层信息,只改克隆表体 HTML
396
+ 并用当前表 footnote 替换前表 footnote,不会修改任何输入对象。
397
+ """
398
+ if (
399
+ not isinstance(previous_table, dict)
400
+ or not isinstance(current_table, dict)
401
+ or previous_table.get("type") != BlockType.TABLE
402
+ or current_table.get("type") != BlockType.TABLE
403
+ ):
404
+ return None
405
+
406
+ previous_clone = deepcopy(previous_table)
407
+ current_clone = deepcopy(current_table)
408
+ try:
409
+ previous_state = _build_table_state(previous_clone)
410
+ current_state = _build_table_state(current_clone)
411
+ if previous_state is None or current_state is None:
412
+ return None
413
+ if not can_merge_tables(current_state, previous_state):
414
+ return None
415
+ if not _perform_table_content_merge(
416
+ previous_state,
417
+ current_state,
418
+ previous_clone,
419
+ current_clone,
420
+ ):
421
+ return None
422
+ except (AssertionError, TypeError, ValueError):
423
+ return None
424
+
425
+ return previous_clone
@@ -0,0 +1,104 @@
1
+ """文档页边界上的跨页表格识别与延续标记编排。"""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Any
6
+
7
+ from ...schema import MERGE_TRANSPARENT_BLOCK_TYPES, BlockType
8
+
9
+ from .blocks import _get_or_create_table_state
10
+ from .models import BlockDict, PageInfoDict, TableMergeState
11
+ from .structure import can_merge_tables
12
+
13
+ TABLE_BOUNDARY_IGNORED_TYPES = set(MERGE_TRANSPARENT_BLOCK_TYPES)
14
+
15
+
16
+ def _clear_table_continuation_marker(table_block: BlockDict) -> None:
17
+ """递归清除 table 根块及其子块中过期的 ``continues_prev``。"""
18
+ table_block.pop("continues_prev", None)
19
+ content = table_block.get("content")
20
+ if not isinstance(content, list):
21
+ return
22
+ for child in content:
23
+ if isinstance(child, dict):
24
+ child.pop("continues_prev", None)
25
+ _clear_nested_continuation_markers(child)
26
+
27
+
28
+ def _clear_nested_continuation_markers(block: BlockDict) -> None:
29
+ """清除表格子树中的旧延续标记,避免标记落到嵌套子块。"""
30
+ content = block.get("content")
31
+ if not isinstance(content, list):
32
+ return
33
+ for child in content:
34
+ if isinstance(child, dict):
35
+ child.pop("continues_prev", None)
36
+ _clear_nested_continuation_markers(child)
37
+
38
+
39
+ def _find_boundary_table(blocks: list[Any], *, from_end: bool) -> BlockDict | None:
40
+ """从页边界扫描 table;噪声块可跳过,其他语义块立即阻断。"""
41
+ ordered_blocks = reversed(blocks) if from_end else iter(blocks)
42
+ for block in ordered_blocks:
43
+ if not isinstance(block, dict):
44
+ return None
45
+ block_type = block.get("type")
46
+ if block_type in TABLE_BOUNDARY_IGNORED_TYPES:
47
+ continue
48
+ if block_type == BlockType.TABLE:
49
+ return block
50
+ return None
51
+ return None
52
+
53
+
54
+ def _is_consecutive_page_pair(previous_page: PageInfoDict, current_page: PageInfoDict) -> bool:
55
+ """按显式零基 page_idx 判断页面在文档中是否严格连续。"""
56
+ previous_page_idx = previous_page.get("page_idx")
57
+ current_page_idx = current_page.get("page_idx")
58
+ return type(previous_page_idx) is int and type(current_page_idx) is int and current_page_idx == previous_page_idx + 1
59
+
60
+
61
+ def merge_table(page_info_list: list[PageInfoDict]) -> None:
62
+ """倒序识别连续页边界表格,并只在后表写入延续标记。"""
63
+ if not isinstance(page_info_list, list):
64
+ return
65
+
66
+ for page_info in page_info_list:
67
+ if not isinstance(page_info, dict):
68
+ continue
69
+ blocks = page_info.get("blocks")
70
+ if not isinstance(blocks, list):
71
+ continue
72
+ for block in blocks:
73
+ if isinstance(block, dict) and block.get("type") == BlockType.TABLE:
74
+ _clear_table_continuation_marker(block)
75
+
76
+ state_cache: dict[int, TableMergeState] = {}
77
+
78
+ for page_position in range(len(page_info_list) - 1, 0, -1):
79
+ current_page = page_info_list[page_position]
80
+ previous_page = page_info_list[page_position - 1]
81
+ if not isinstance(current_page, dict) or not isinstance(previous_page, dict):
82
+ continue
83
+ if not _is_consecutive_page_pair(previous_page, current_page):
84
+ continue
85
+
86
+ current_blocks = current_page.get("blocks")
87
+ previous_blocks = previous_page.get("blocks")
88
+ if not isinstance(current_blocks, list) or not isinstance(previous_blocks, list):
89
+ continue
90
+
91
+ current_table_block = _find_boundary_table(current_blocks, from_end=False)
92
+ previous_table_block = _find_boundary_table(previous_blocks, from_end=True)
93
+ if current_table_block is None or previous_table_block is None:
94
+ continue
95
+
96
+ current_state = _get_or_create_table_state(current_table_block, state_cache)
97
+ previous_state = _get_or_create_table_state(previous_table_block, state_cache)
98
+ if current_state is None or previous_state is None:
99
+ continue
100
+
101
+ if not can_merge_tables(current_state, previous_state):
102
+ continue
103
+
104
+ current_table_block["continues_prev"] = True