docvortex 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (364) hide show
  1. docvortex/__init__.py +21 -0
  2. docvortex/analyzers/__init__.py +3 -0
  3. docvortex/analyzers/native/__init__.py +37 -0
  4. docvortex/analyzers/native/_shared/__init__.py +3 -0
  5. docvortex/analyzers/native/_shared/hyperlink.py +16 -0
  6. docvortex/analyzers/native/_shared/image.py +8 -0
  7. docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
  8. docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
  9. docvortex/analyzers/native/_shared/markup/formula.py +37 -0
  10. docvortex/analyzers/native/_shared/markup/projector.py +57 -0
  11. docvortex/analyzers/native/_shared/markup/styles.py +19 -0
  12. docvortex/analyzers/native/_shared/mathml.py +17 -0
  13. docvortex/analyzers/native/_shared/names.py +7 -0
  14. docvortex/analyzers/native/_shared/xycut.py +414 -0
  15. docvortex/analyzers/native/contracts.py +44 -0
  16. docvortex/analyzers/native/csv.py +351 -0
  17. docvortex/analyzers/native/epub/__init__.py +18 -0
  18. docvortex/analyzers/native/epub/constants.py +45 -0
  19. docvortex/analyzers/native/epub/converter.py +80 -0
  20. docvortex/analyzers/native/epub/errors.py +20 -0
  21. docvortex/analyzers/native/epub/metadata.py +31 -0
  22. docvortex/analyzers/native/epub/package.py +649 -0
  23. docvortex/analyzers/native/epub/xhtml.py +306 -0
  24. docvortex/analyzers/native/html/__init__.py +6 -0
  25. docvortex/analyzers/native/html/anchors.py +271 -0
  26. docvortex/analyzers/native/html/constants.py +25 -0
  27. docvortex/analyzers/native/html/contracts.py +7 -0
  28. docvortex/analyzers/native/html/converter.py +117 -0
  29. docvortex/analyzers/native/html/document.py +389 -0
  30. docvortex/analyzers/native/html/errors.py +12 -0
  31. docvortex/analyzers/native/html/resources.py +365 -0
  32. docvortex/analyzers/native/html/selector.py +421 -0
  33. docvortex/analyzers/native/models.py +209 -0
  34. docvortex/analyzers/native/ofd/__init__.py +14 -0
  35. docvortex/analyzers/native/ofd/constants.py +59 -0
  36. docvortex/analyzers/native/ofd/converter.py +31 -0
  37. docvortex/analyzers/native/ofd/errors.py +16 -0
  38. docvortex/analyzers/native/ofd/geometry.py +215 -0
  39. docvortex/analyzers/native/ofd/images.py +124 -0
  40. docvortex/analyzers/native/ofd/metadata.py +53 -0
  41. docvortex/analyzers/native/ofd/models.py +170 -0
  42. docvortex/analyzers/native/ofd/package.py +346 -0
  43. docvortex/analyzers/native/ofd/path.py +201 -0
  44. docvortex/analyzers/native/ofd/reading_order.py +293 -0
  45. docvortex/analyzers/native/ofd/resources.py +122 -0
  46. docvortex/analyzers/native/ofd/scene.py +347 -0
  47. docvortex/analyzers/native/ofd/table.py +240 -0
  48. docvortex/analyzers/native/ofd/text.py +521 -0
  49. docvortex/analyzers/native/office/__init__.py +3 -0
  50. docvortex/analyzers/native/office/doc/__init__.py +3 -0
  51. docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
  52. docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
  53. docvortex/analyzers/native/office/doc/fib.py +261 -0
  54. docvortex/analyzers/native/office/doc/fields.py +101 -0
  55. docvortex/analyzers/native/office/doc/formatting.py +171 -0
  56. docvortex/analyzers/native/office/doc/images.py +162 -0
  57. docvortex/analyzers/native/office/doc/lists.py +356 -0
  58. docvortex/analyzers/native/office/doc/models.py +164 -0
  59. docvortex/analyzers/native/office/doc/parser.py +851 -0
  60. docvortex/analyzers/native/office/doc/pieces.py +256 -0
  61. docvortex/analyzers/native/office/doc/records.py +40 -0
  62. docvortex/analyzers/native/office/doc/sprm.py +269 -0
  63. docvortex/analyzers/native/office/doc/styles.py +215 -0
  64. docvortex/analyzers/native/office/docx/__init__.py +3 -0
  65. docvortex/analyzers/native/office/docx/context.py +71 -0
  66. docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
  67. docvortex/analyzers/native/office/docx/equationxml.py +119 -0
  68. docvortex/analyzers/native/office/docx/fields.py +816 -0
  69. docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
  70. docvortex/analyzers/native/office/docx/main.py +50 -0
  71. docvortex/analyzers/native/office/docx/numbering.py +491 -0
  72. docvortex/analyzers/native/office/docx/office_xml.py +57 -0
  73. docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
  74. docvortex/analyzers/native/office/docx/resources.py +526 -0
  75. docvortex/analyzers/native/office/docx/styles.py +693 -0
  76. docvortex/analyzers/native/office/docx/tables.py +515 -0
  77. docvortex/analyzers/native/office/equation/__init__.py +3 -0
  78. docvortex/analyzers/native/office/equation/image.py +470 -0
  79. docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
  80. docvortex/analyzers/native/office/equation/mtef.py +885 -0
  81. docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
  82. docvortex/analyzers/native/office/equation/omml.py +561 -0
  83. docvortex/analyzers/native/office/equation/ooxml.py +62 -0
  84. docvortex/analyzers/native/office/errors.py +33 -0
  85. docvortex/analyzers/native/office/image.py +307 -0
  86. docvortex/analyzers/native/office/legacy/__init__.py +3 -0
  87. docvortex/analyzers/native/office/legacy/binary.py +43 -0
  88. docvortex/analyzers/native/office/legacy/officeart.py +362 -0
  89. docvortex/analyzers/native/office/legacy/ole.py +110 -0
  90. docvortex/analyzers/native/office/limits.py +12 -0
  91. docvortex/analyzers/native/office/odf/__init__.py +3 -0
  92. docvortex/analyzers/native/office/odf/chart.py +76 -0
  93. docvortex/analyzers/native/office/odf/constants.py +70 -0
  94. docvortex/analyzers/native/office/odf/converters.py +403 -0
  95. docvortex/analyzers/native/office/odf/errors.py +18 -0
  96. docvortex/analyzers/native/office/odf/metadata.py +91 -0
  97. docvortex/analyzers/native/office/odf/models.py +176 -0
  98. docvortex/analyzers/native/office/odf/package.py +270 -0
  99. docvortex/analyzers/native/office/odf/styles.py +329 -0
  100. docvortex/analyzers/native/office/odf/table.py +469 -0
  101. docvortex/analyzers/native/office/odf/text.py +1002 -0
  102. docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
  103. docvortex/analyzers/native/office/opc.py +38 -0
  104. docvortex/analyzers/native/office/ppt/__init__.py +3 -0
  105. docvortex/analyzers/native/office/ppt/models.py +118 -0
  106. docvortex/analyzers/native/office/ppt/parser.py +1895 -0
  107. docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
  108. docvortex/analyzers/native/office/ppt/records.py +131 -0
  109. docvortex/analyzers/native/office/ppt/style_text.py +247 -0
  110. docvortex/analyzers/native/office/pptx/__init__.py +3 -0
  111. docvortex/analyzers/native/office/pptx/context.py +113 -0
  112. docvortex/analyzers/native/office/pptx/lists.py +558 -0
  113. docvortex/analyzers/native/office/pptx/main.py +20 -0
  114. docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
  115. docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
  116. docvortex/analyzers/native/office/pptx/resources.py +329 -0
  117. docvortex/analyzers/native/office/pptx/shapes.py +393 -0
  118. docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
  119. docvortex/analyzers/native/office/pptx/titles.py +178 -0
  120. docvortex/analyzers/native/office/rich_text.py +420 -0
  121. docvortex/analyzers/native/office/rtf/__init__.py +3 -0
  122. docvortex/analyzers/native/office/rtf/converter.py +708 -0
  123. docvortex/analyzers/native/office/rtf/lexer.py +213 -0
  124. docvortex/analyzers/native/office/rtf/math.py +339 -0
  125. docvortex/analyzers/native/office/rtf/models.py +185 -0
  126. docvortex/analyzers/native/office/rtf/parser.py +1553 -0
  127. docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
  128. docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
  129. docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
  130. docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
  131. docvortex/analyzers/native/office/streams.py +18 -0
  132. docvortex/analyzers/native/office/xls/__init__.py +3 -0
  133. docvortex/analyzers/native/office/xls/chart.py +132 -0
  134. docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
  135. docvortex/analyzers/native/office/xls/models.py +109 -0
  136. docvortex/analyzers/native/office/xls/number_format.py +521 -0
  137. docvortex/analyzers/native/office/xls/parser.py +1145 -0
  138. docvortex/analyzers/native/office/xls/records.py +201 -0
  139. docvortex/analyzers/native/office/xls/strings.py +205 -0
  140. docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
  141. docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
  142. docvortex/analyzers/native/office/xlsx/main.py +20 -0
  143. docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
  144. docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
  145. docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
  146. docvortex/analyzers/native/pdf/__init__.py +3 -0
  147. docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
  148. docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
  149. docvortex/analyzers/native/pdf/code_blocks.py +535 -0
  150. docvortex/analyzers/native/pdf/formulas.py +1985 -0
  151. docvortex/analyzers/native/pdf/geometry.py +281 -0
  152. docvortex/analyzers/native/pdf/graphics.py +1501 -0
  153. docvortex/analyzers/native/pdf/index_blocks.py +268 -0
  154. docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
  155. docvortex/analyzers/native/pdf/inline/common.py +112 -0
  156. docvortex/analyzers/native/pdf/inline/detection.py +590 -0
  157. docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
  158. docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
  159. docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
  160. docvortex/analyzers/native/pdf/inline/types.py +385 -0
  161. docvortex/analyzers/native/pdf/line_layout.py +1106 -0
  162. docvortex/analyzers/native/pdf/line_merging.py +1223 -0
  163. docvortex/analyzers/native/pdf/models.py +246 -0
  164. docvortex/analyzers/native/pdf/native_text.py +1004 -0
  165. docvortex/analyzers/native/pdf/pipeline.py +1390 -0
  166. docvortex/analyzers/native/pdf/script_geometry.py +636 -0
  167. docvortex/analyzers/native/pdf/shared.py +30 -0
  168. docvortex/analyzers/native/pdf/spatial_text.py +383 -0
  169. docvortex/analyzers/native/pdf/table_annotations.py +446 -0
  170. docvortex/analyzers/native/pdf/table_constants.py +48 -0
  171. docvortex/analyzers/native/pdf/table_detection.py +227 -0
  172. docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
  173. docvortex/analyzers/native/pdf/table_geometry.py +40 -0
  174. docvortex/analyzers/native/pdf/table_materialization.py +444 -0
  175. docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
  176. docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
  177. docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
  178. docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
  179. docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
  180. docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
  181. docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
  182. docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
  183. docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
  184. docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
  185. docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
  186. docvortex/analyzers/native/pdf/table_rows.py +34 -0
  187. docvortex/analyzers/native/pdf/table_rules.py +1129 -0
  188. docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
  189. docvortex/analyzers/native/pdf/tables.py +147 -0
  190. docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
  191. docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
  192. docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
  193. docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
  194. docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
  195. docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
  196. docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
  197. docvortex/analyzers/native/pdf/text_blocks.py +82 -0
  198. docvortex/analyzers/native/pdf/text_styles.py +55 -0
  199. docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
  200. docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
  201. docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
  202. docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
  203. docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
  204. docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
  205. docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
  206. docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
  207. docvortex/analyzers/native/pdf/titles.py +75 -0
  208. docvortex/analyzers/native/pdf/typography.py +19 -0
  209. docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
  210. docvortex/api.py +180 -0
  211. docvortex/assets/__init__.py +5 -0
  212. docvortex/assets/store.py +51 -0
  213. docvortex/cli.py +54 -0
  214. docvortex/codecs/__init__.py +3 -0
  215. docvortex/codecs/html/__init__.py +22 -0
  216. docvortex/codecs/html/contracts.py +236 -0
  217. docvortex/codecs/html/materializer.py +331 -0
  218. docvortex/codecs/html/parser.py +763 -0
  219. docvortex/codecs/html/resources.py +34 -0
  220. docvortex/codecs/json.py +17 -0
  221. docvortex/content/__init__.py +5 -0
  222. docvortex/content/inline.py +248 -0
  223. docvortex/content/markup/__init__.py +44 -0
  224. docvortex/content/markup/anchors.py +188 -0
  225. docvortex/content/markup/formula.py +280 -0
  226. docvortex/content/markup/projector.py +1237 -0
  227. docvortex/content/markup/styles.py +327 -0
  228. docvortex/content/mathml.py +167 -0
  229. docvortex/content/normalization.py +188 -0
  230. docvortex/content/spans.py +183 -0
  231. docvortex/content/table/__init__.py +18 -0
  232. docvortex/content/table/blocks.py +152 -0
  233. docvortex/content/table/content.py +425 -0
  234. docvortex/content/table/document.py +104 -0
  235. docvortex/content/table/html.py +399 -0
  236. docvortex/content/table/models.py +76 -0
  237. docvortex/content/table/rules.py +42 -0
  238. docvortex/content/table/structure.py +221 -0
  239. docvortex/content/tree.py +5 -0
  240. docvortex/document/__init__.py +3 -0
  241. docvortex/document/contracts.py +22 -0
  242. docvortex/document/detection.py +389 -0
  243. docvortex/document/filetypes.py +175 -0
  244. docvortex/document/page_range.py +167 -0
  245. docvortex/document/pdf/__init__.py +19 -0
  246. docvortex/document/pdf/classify.py +1138 -0
  247. docvortex/document/pdf/constants.py +132 -0
  248. docvortex/document/pdf/diagnostics.py +350 -0
  249. docvortex/document/pdf/document.py +582 -0
  250. docvortex/document/pdf/font_runtime.py +335 -0
  251. docvortex/document/pdf/geometry.py +31 -0
  252. docvortex/document/pdf/images.py +654 -0
  253. docvortex/document/pdf/native_annotations.py +367 -0
  254. docvortex/document/pdf/native_contracts.py +169 -0
  255. docvortex/document/pdf/native_coordinates.py +216 -0
  256. docvortex/document/pdf/native_lifecycle.py +16 -0
  257. docvortex/document/pdf/native_objects.py +902 -0
  258. docvortex/document/pdf/native_text_geometry.py +315 -0
  259. docvortex/document/pdf/pdfium.py +325 -0
  260. docvortex/document/pdf/raster.py +46 -0
  261. docvortex/document/pdf/text/__init__.py +62 -0
  262. docvortex/document/pdf/text/contracts.py +211 -0
  263. docvortex/document/pdf/text/extract.py +165 -0
  264. docvortex/document/pdf/text/geometry.py +16 -0
  265. docvortex/document/pdf/text/groups.py +162 -0
  266. docvortex/document/pdf/visual_geometry.py +201 -0
  267. docvortex/document/pdf/visuals.py +343 -0
  268. docvortex/document/source.py +92 -0
  269. docvortex/errors.py +21 -0
  270. docvortex/export/__init__.py +3 -0
  271. docvortex/export/bundle.py +98 -0
  272. docvortex/export/files.py +66 -0
  273. docvortex/export/middle.py +208 -0
  274. docvortex/foundation/__init__.py +3 -0
  275. docvortex/foundation/geometry.py +125 -0
  276. docvortex/foundation/hyperlink.py +65 -0
  277. docvortex/foundation/image.py +48 -0
  278. docvortex/foundation/image_encoding.py +30 -0
  279. docvortex/foundation/image_payload.py +280 -0
  280. docvortex/foundation/language.py +92 -0
  281. docvortex/foundation/platform.py +38 -0
  282. docvortex/foundation/text.py +153 -0
  283. docvortex/foundation/type_identity.py +20 -0
  284. docvortex/foundation/xml_names.py +20 -0
  285. docvortex/options.py +30 -0
  286. docvortex/postprocess/__init__.py +3 -0
  287. docvortex/postprocess/content.py +53 -0
  288. docvortex/postprocess/document.py +19 -0
  289. docvortex/postprocess/lists.py +236 -0
  290. docvortex/postprocess/page_blocks.py +214 -0
  291. docvortex/postprocess/pages.py +95 -0
  292. docvortex/postprocess/paragraphs.py +580 -0
  293. docvortex/postprocess/visual.py +715 -0
  294. docvortex/render/__init__.py +48 -0
  295. docvortex/render/_internal/__init__.py +3 -0
  296. docvortex/render/_internal/common/__init__.py +3 -0
  297. docvortex/render/_internal/common/context.py +43 -0
  298. docvortex/render/_internal/common/html_table.py +178 -0
  299. docvortex/render/_internal/common/index.py +33 -0
  300. docvortex/render/_internal/common/list_items.py +158 -0
  301. docvortex/render/_internal/common/planner.py +140 -0
  302. docvortex/render/_internal/docx/__init__.py +3 -0
  303. docvortex/render/_internal/docx/assets.py +202 -0
  304. docvortex/render/_internal/docx/inline.py +434 -0
  305. docvortex/render/_internal/docx/math.py +220 -0
  306. docvortex/render/_internal/docx/renderer.py +905 -0
  307. docvortex/render/_internal/docx/styles.py +195 -0
  308. docvortex/render/_internal/docx/table.py +442 -0
  309. docvortex/render/_internal/epub/__init__.py +5 -0
  310. docvortex/render/_internal/epub/assets.py +173 -0
  311. docvortex/render/_internal/epub/package.py +249 -0
  312. docvortex/render/_internal/epub/renderer.py +1156 -0
  313. docvortex/render/_internal/html/__init__.py +3 -0
  314. docvortex/render/_internal/html/inline.py +346 -0
  315. docvortex/render/_internal/html/renderer.py +1041 -0
  316. docvortex/render/_internal/html/sanitizer.py +478 -0
  317. docvortex/render/_internal/html/table.py +121 -0
  318. docvortex/render/_internal/latex/__init__.py +1 -0
  319. docvortex/render/_internal/latex/assets.py +85 -0
  320. docvortex/render/_internal/latex/inline.py +145 -0
  321. docvortex/render/_internal/latex/renderer.py +506 -0
  322. docvortex/render/_internal/latex/table.py +347 -0
  323. docvortex/render/_internal/markdown/__init__.py +3 -0
  324. docvortex/render/_internal/markdown/assets.py +78 -0
  325. docvortex/render/_internal/markdown/blocks.py +635 -0
  326. docvortex/render/_internal/markdown/escaping.py +51 -0
  327. docvortex/render/_internal/markdown/inline.py +260 -0
  328. docvortex/render/_internal/markdown/renderer.py +93 -0
  329. docvortex/render/_internal/markdown/table.py +281 -0
  330. docvortex/render/_internal/pdf/__init__.py +3 -0
  331. docvortex/render/_internal/pdf/assets.py +197 -0
  332. docvortex/render/_internal/pdf/formula.py +417 -0
  333. docvortex/render/_internal/pdf/inline.py +343 -0
  334. docvortex/render/_internal/pdf/renderer.py +734 -0
  335. docvortex/render/_internal/pdf/styles.py +206 -0
  336. docvortex/render/_internal/pdf/table.py +272 -0
  337. docvortex/render/_internal/structured_content/__init__.py +3 -0
  338. docvortex/render/_internal/structured_content/renderer.py +193 -0
  339. docvortex/render/api.py +199 -0
  340. docvortex/render/contracts.py +205 -0
  341. docvortex/render/docx.py +38 -0
  342. docvortex/render/epub.py +35 -0
  343. docvortex/render/fragments.py +70 -0
  344. docvortex/render/html.py +29 -0
  345. docvortex/render/latex.py +24 -0
  346. docvortex/render/markdown.py +50 -0
  347. docvortex/render/pdf.py +25 -0
  348. docvortex/render/structured_content.py +23 -0
  349. docvortex/resources/epub/docvortex.css +91 -0
  350. docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
  351. docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
  352. docvortex/resources/fonts/NOTICE +190 -0
  353. docvortex/resources/fonts/manifest.json +9 -0
  354. docvortex/resources/html/docvortex.css +601 -0
  355. docvortex/resources/html/docvortex.min.css +1 -0
  356. docvortex/result.py +91 -0
  357. docvortex/schema.py +1141 -0
  358. docvortex/version.py +3 -0
  359. docvortex-0.2.1.dist-info/METADATA +193 -0
  360. docvortex-0.2.1.dist-info/RECORD +364 -0
  361. docvortex-0.2.1.dist-info/WHEEL +5 -0
  362. docvortex-0.2.1.dist-info/entry_points.txt +2 -0
  363. docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
  364. docvortex-0.2.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,535 @@
1
+ """识别带填充背景的等宽代码区域并投影其空间文本。"""
2
+
3
+ from __future__ import annotations
4
+
5
+ import math
6
+ import statistics
7
+ import unicodedata
8
+ from typing import Any
9
+
10
+ from .spatial_text import project_pdf_spatial_text
11
+ from ....schema import BBox
12
+ from ....document.pdf.document import PDFPathInfo
13
+
14
+ from .geometry import (
15
+ _bbox_area,
16
+ _bbox_center_x,
17
+ _bbox_center_y,
18
+ _bbox_overlap_in_first,
19
+ _bbox_overlap_in_smaller,
20
+ _bbox_union_many,
21
+ )
22
+ from .models import _CodeCandidate, _LineItem, _PageSource
23
+ from .native_text import _sanitize_pdf_control_text
24
+
25
+
26
+ _MONOSPACE_FONT_HINTS = (
27
+ "mono",
28
+ "courier",
29
+ "consolas",
30
+ "menlo",
31
+ "typewriter",
32
+ "fixed",
33
+ "code",
34
+ )
35
+
36
+
37
+ def _build_code_blocks(
38
+ source: _PageSource,
39
+ excluded_bboxes: list[BBox],
40
+ claimed_line_indices: set[int],
41
+ ) -> tuple[list[dict[str, Any]], set[int]]:
42
+ """检测代码背景、唯一认领其文本,并输出页内 code block。"""
43
+
44
+ candidates = _detect_code_candidates(
45
+ source,
46
+ excluded_bboxes,
47
+ claimed_line_indices,
48
+ )
49
+ return _materialize_code_candidates(source, candidates)
50
+
51
+
52
+ def _build_rule_delimited_code_blocks(
53
+ source: _PageSource,
54
+ excluded_bboxes: list[BBox],
55
+ claimed_line_indices: set[int] | None = None,
56
+ ) -> tuple[list[dict[str, Any]], set[int]]:
57
+ """在表格认领前物化外框或上下横线限定的代码清单。"""
58
+
59
+ candidates = _detect_rule_delimited_code_candidates(
60
+ source,
61
+ excluded_bboxes,
62
+ claimed_line_indices or set(),
63
+ )
64
+ return _materialize_code_candidates(source, candidates)
65
+
66
+
67
+ def _materialize_code_candidates(
68
+ source: _PageSource,
69
+ candidates: list[_CodeCandidate],
70
+ ) -> tuple[list[dict[str, Any]], set[int]]:
71
+ """统一投影代码候选,确保来源行只被一个 code block 认领。"""
72
+
73
+ if not candidates:
74
+ return [], set()
75
+ lines_by_index = {line.source_index: line for line in source.lines}
76
+ blocks: list[dict[str, Any]] = []
77
+ claimed: set[int] = set()
78
+ for candidate in candidates:
79
+ members = [
80
+ lines_by_index[source_index] for source_index in sorted(candidate.line_indices) if source_index in lines_by_index
81
+ ]
82
+ if not members:
83
+ continue
84
+ content = project_pdf_spatial_text(
85
+ _code_member_chars(members),
86
+ candidate.bbox,
87
+ candidate.angle,
88
+ preserve_blank_rows=True,
89
+ )
90
+ if not content:
91
+ content = _fallback_code_content(members)
92
+ if not content:
93
+ continue
94
+ blocks.append(
95
+ {
96
+ "type": "code",
97
+ "bbox": candidate.bbox,
98
+ "angle": candidate.angle,
99
+ "content": content,
100
+ }
101
+ )
102
+ claimed.update(candidate.line_indices)
103
+ blocks.sort(key=lambda block: (block["bbox"][1], block["bbox"][0]))
104
+ return blocks, claimed
105
+
106
+
107
+ def _vertical_rule_candidate_height_coverage(
108
+ rule_bbox: BBox,
109
+ candidate_bbox: BBox,
110
+ ) -> float:
111
+ """计算竖轨在候选高度方向上的实际覆盖比例,长轨超出候选时仍按交集计量。"""
112
+
113
+ candidate_height = max(0.0, candidate_bbox[3] - candidate_bbox[1])
114
+ if candidate_height <= 0:
115
+ return 0.0
116
+ overlap = max(
117
+ 0.0,
118
+ min(rule_bbox[3], candidate_bbox[3]) - max(rule_bbox[1], candidate_bbox[1]),
119
+ )
120
+ return overlap / candidate_height
121
+
122
+
123
+ def _detect_rule_delimited_code_candidates(
124
+ source: _PageSource,
125
+ excluded_bboxes: list[BBox],
126
+ claimed_line_indices: set[int],
127
+ ) -> list[_CodeCandidate]:
128
+ """按规则边界、稳定行距和缩进层次识别非等宽代码清单。"""
129
+
130
+ page_width, page_height = source.page_size
131
+ if page_width <= 0 or page_height <= 0:
132
+ return []
133
+ available = [line for line in source.lines if line.source_index not in claimed_line_indices and line.angle == 0]
134
+ if len(available) < 5:
135
+ return []
136
+ median_height = statistics.median(max(0.1, line.effective_height or line.bbox[3] - line.bbox[1]) for line in available)
137
+ horizontal_rules = sorted(
138
+ [
139
+ line
140
+ for line in source.drawing_lines
141
+ if line.orientation == "horizontal" and line.bbox[2] - line.bbox[0] >= 0.22 * page_width
142
+ ],
143
+ key=lambda line: (line.bbox[1], line.bbox[0]),
144
+ )
145
+ vertical_rules = [line for line in source.drawing_lines if line.orientation == "vertical"]
146
+ raw_candidates: list[_CodeCandidate] = []
147
+ endpoint_tolerance = max(2.0, 0.75 * median_height)
148
+ for top_index, top_rule in enumerate(horizontal_rules[:-1]):
149
+ for bottom_rule in horizontal_rules[top_index + 1 :]:
150
+ if (
151
+ abs(top_rule.bbox[0] - bottom_rule.bbox[0]) > endpoint_tolerance
152
+ or abs(top_rule.bbox[2] - bottom_rule.bbox[2]) > endpoint_tolerance
153
+ ):
154
+ continue
155
+ candidate_bbox = _bbox_union_many([top_rule.bbox, bottom_rule.bbox])
156
+ candidate_height = candidate_bbox[3] - candidate_bbox[1]
157
+ if not 6.0 * median_height <= candidate_height <= 0.5 * page_height:
158
+ continue
159
+ if any(_bbox_overlap_in_smaller(candidate_bbox, excluded_bbox) >= 0.5 for excluded_bbox in excluded_bboxes):
160
+ continue
161
+ interior_rules = [
162
+ rule
163
+ for rule in horizontal_rules
164
+ if top_rule.bbox[3] + 0.5 * median_height < rule.bbox[1] < bottom_rule.bbox[1] - 0.5 * median_height
165
+ and _bbox_overlap_in_first(rule.bbox, candidate_bbox) >= 0.8
166
+ and rule.bbox[2] - rule.bbox[0] >= 0.6 * (candidate_bbox[2] - candidate_bbox[0])
167
+ ]
168
+ if interior_rules:
169
+ continue
170
+ internal_vertical_rules = [
171
+ rule
172
+ for rule in vertical_rules
173
+ if candidate_bbox[0] + median_height < _bbox_center_x(rule.bbox) < candidate_bbox[2] - median_height
174
+ # 表格竖轨可能贯穿候选上下边界,必须相对候选高度计算覆盖率。
175
+ and _vertical_rule_candidate_height_coverage(
176
+ rule.bbox,
177
+ candidate_bbox,
178
+ )
179
+ >= 0.6
180
+ ]
181
+ if internal_vertical_rules:
182
+ continue
183
+ members = [
184
+ line
185
+ for line in available
186
+ if candidate_bbox[0] - 0.5 * median_height
187
+ <= _bbox_center_x(line.bbox)
188
+ <= candidate_bbox[2] + 0.5 * median_height
189
+ and top_rule.bbox[3] <= _bbox_center_y(line.bbox) <= bottom_rule.bbox[1]
190
+ ]
191
+ if not _rule_delimited_code_members_are_structured(
192
+ members,
193
+ candidate_bbox,
194
+ median_height,
195
+ ):
196
+ continue
197
+ raw_candidates.append(
198
+ _CodeCandidate(
199
+ bbox=candidate_bbox,
200
+ angle=0,
201
+ line_indices={line.source_index for line in members},
202
+ )
203
+ )
204
+
205
+ accepted: list[_CodeCandidate] = []
206
+ for candidate in sorted(raw_candidates, key=lambda item: _bbox_area(item.bbox)):
207
+ if any(_bbox_overlap_in_smaller(candidate.bbox, existing.bbox) >= 0.85 for existing in accepted):
208
+ continue
209
+ accepted.append(candidate)
210
+ return sorted(accepted, key=lambda item: (item.bbox[1], item.bbox[0]))
211
+
212
+
213
+ def _rule_delimited_code_members_are_structured(
214
+ members: list[_LineItem],
215
+ candidate_bbox: BBox,
216
+ median_height: float,
217
+ ) -> bool:
218
+ """验证候选具有稳定基线、代码缩进或窄行号槽,并排除规则多列表格。"""
219
+
220
+ if len(members) < 5:
221
+ return False
222
+ rows: dict[int, list[_LineItem]] = {}
223
+ fallback_row = 1_000_000
224
+ for line in members:
225
+ row_key = line.visual_row_id
226
+ if row_key is None:
227
+ row_key = fallback_row
228
+ fallback_row += 1
229
+ rows.setdefault(row_key, []).append(line)
230
+ if len(rows) < 5:
231
+ return False
232
+ ordered_rows = sorted(
233
+ rows.values(),
234
+ key=lambda row: min(_bbox_center_y(line.bbox) for line in row),
235
+ )
236
+ row_centers = [statistics.median(_bbox_center_y(line.bbox) for line in row) for row in ordered_rows]
237
+ row_gaps = [current - previous for previous, current in zip(row_centers, row_centers[1:]) if current > previous]
238
+ if not row_gaps:
239
+ return False
240
+ base_pitch = statistics.median(row_gaps)
241
+ if base_pitch <= 0 or sum(0.65 * base_pitch <= gap <= 1.8 * base_pitch for gap in row_gaps) / len(row_gaps) < 0.75:
242
+ return False
243
+ if sum(len(row) >= 3 for row in ordered_rows) / len(ordered_rows) >= 0.25:
244
+ return False
245
+
246
+ left_positions = sorted(line.bbox[0] for line in members)
247
+ indent_clusters: list[list[float]] = []
248
+ for position in left_positions:
249
+ if not indent_clusters or position - statistics.median(indent_clusters[-1]) > 0.75 * median_height:
250
+ indent_clusters.append([position])
251
+ else:
252
+ indent_clusters[-1].append(position)
253
+ narrow_gutter_rows = 0
254
+ for row in ordered_rows:
255
+ ordered = sorted(row, key=lambda line: line.bbox[0])
256
+ if (
257
+ len(ordered) >= 2
258
+ and ordered[0].bbox[2] - ordered[0].bbox[0] <= 2.0 * median_height
259
+ and ordered[1].bbox[0] - ordered[0].bbox[2] >= 0.5 * median_height
260
+ ):
261
+ narrow_gutter_rows += 1
262
+ has_line_number_gutter = narrow_gutter_rows / len(ordered_rows) >= 0.35
263
+ has_indent_hierarchy = len(indent_clusters) >= 3 and sum(len(cluster) >= 2 for cluster in indent_clusters) >= 2
264
+ if not has_line_number_gutter and not has_indent_hierarchy:
265
+ return False
266
+ # 稳定行号槽属于强代码证据,允许右侧长语句或注释自然触及清单边界。
267
+ if has_line_number_gutter:
268
+ return True
269
+
270
+ occupied_width = max(line.bbox[2] for line in members) - min(line.bbox[0] for line in members)
271
+ return occupied_width <= 0.95 * max(
272
+ 0.1,
273
+ candidate_bbox[2] - candidate_bbox[0],
274
+ )
275
+
276
+
277
+ def _code_member_chars(lines: list[_LineItem]) -> list[dict[str, Any]]:
278
+ """按 char_idx 去重代码成员字符,避免区域内斜向水印混入空间投影。"""
279
+
280
+ output: list[dict[str, Any]] = []
281
+ seen: set[tuple[str, int]] = set()
282
+ for line in lines:
283
+ for fallback_index, char in enumerate(line.chars):
284
+ if not isinstance(char, dict):
285
+ continue
286
+ try:
287
+ identity = ("source", int(char.get("char_idx")))
288
+ except (TypeError, ValueError):
289
+ identity = ("fallback", fallback_index + line.source_index * 1_000_000)
290
+ if identity in seen:
291
+ continue
292
+ seen.add(identity)
293
+ output.append(char)
294
+ return output
295
+
296
+
297
+ def _detect_code_candidates(
298
+ source: _PageSource,
299
+ excluded_bboxes: list[BBox],
300
+ claimed_line_indices: set[int],
301
+ ) -> list[_CodeCandidate]:
302
+ """以非白填充矩形、等宽字体和规则空间栅格筛选代码候选。"""
303
+
304
+ page_width, page_height = source.page_size
305
+ page_area = max(0.0, page_width) * max(0.0, page_height)
306
+ if page_area <= 0 or not source.path_infos:
307
+ return []
308
+ raw_candidates: list[_CodeCandidate] = []
309
+ for path_info in source.path_infos:
310
+ bbox = path_info.bbox
311
+ width = bbox[2] - bbox[0]
312
+ height = bbox[3] - bbox[1]
313
+ if (
314
+ path_info.form_depth != 0
315
+ or not path_info.fill_visible
316
+ or path_info.segment_count < 4
317
+ or not _path_has_visible_nonwhite_fill(path_info)
318
+ or width < 0.5 * page_width
319
+ or height < 0.008 * page_height
320
+ or _bbox_area(bbox) >= 0.8 * page_area
321
+ or any(_bbox_overlap_in_smaller(bbox, excluded_bbox) >= 0.5 for excluded_bbox in excluded_bboxes)
322
+ ):
323
+ continue
324
+ members = [
325
+ line
326
+ for line in source.lines
327
+ if line.source_index not in claimed_line_indices and _bbox_overlap_in_first(line.bbox, bbox) >= 0.8
328
+ ]
329
+ if not members:
330
+ continue
331
+ dominant_angle = _dominant_code_angle(members)
332
+ angle_members = [line for line in members if line.angle == dominant_angle]
333
+ total_support = sum(_estimated_line_character_count(line) for line in members)
334
+ angle_support = sum(_estimated_line_character_count(line) for line in angle_members)
335
+ if total_support <= 0 or angle_support / total_support < 0.8:
336
+ continue
337
+ monospace_ratio, cell_widths = _monospace_character_support(angle_members)
338
+ if monospace_ratio < 0.8 or not _monospace_advances_are_stable(cell_widths):
339
+ continue
340
+ median_cell_width = statistics.median(width_value for values in cell_widths.values() for width_value in values)
341
+ if not _code_rows_have_spatial_structure(
342
+ angle_members,
343
+ bbox,
344
+ median_cell_width,
345
+ ):
346
+ continue
347
+ raw_candidates.append(
348
+ _CodeCandidate(
349
+ bbox=bbox,
350
+ angle=dominant_angle,
351
+ line_indices={line.source_index for line in angle_members},
352
+ )
353
+ )
354
+
355
+ accepted: list[_CodeCandidate] = []
356
+ for candidate in sorted(raw_candidates, key=lambda item: _bbox_area(item.bbox)):
357
+ if any(_bbox_overlap_in_smaller(candidate.bbox, existing.bbox) >= 0.9 for existing in accepted):
358
+ continue
359
+ accepted.append(candidate)
360
+ return sorted(accepted, key=lambda item: (item.bbox[1], item.bbox[0]))
361
+
362
+
363
+ def _path_has_visible_nonwhite_fill(path_info: PDFPathInfo) -> bool:
364
+ """检查填充色是否可见且与白色背景存在最小颜色差。"""
365
+
366
+ if path_info.fill_rgba is None:
367
+ return False
368
+ red, green, blue, alpha = path_info.fill_rgba
369
+ return alpha > 0 and max(255 - red, 255 - green, 255 - blue) >= 5
370
+
371
+
372
+ def _dominant_code_angle(lines: list[_LineItem]) -> int:
373
+ """按估算字符数选择代码区域的主文本方向。"""
374
+
375
+ support: dict[int, float] = {}
376
+ for line in lines:
377
+ support[line.angle] = support.get(line.angle, 0.0) + _estimated_line_character_count(line)
378
+ return max(sorted(support), key=lambda angle: support[angle])
379
+
380
+
381
+ def _estimated_line_character_count(line: _LineItem) -> float:
382
+ """优先按字符对象计数,缺失时用行宽和缓存字宽估算字符支持。"""
383
+
384
+ valid_chars = [char for char in line.chars if isinstance(char, dict) and str(char.get("char") or "").strip()]
385
+ if valid_chars:
386
+ return float(len(valid_chars))
387
+ line_width = max(0.1, line.bbox[2] - line.bbox[0])
388
+ glyph_width = max(0.1, line.median_glyph_width or line_width)
389
+ return max(1.0, line_width / glyph_width)
390
+
391
+
392
+ def _font_name_looks_monospaced(name: str | None) -> bool:
393
+ """按字体元数据中的通用等宽提示判断字体族,不匹配文档内容。"""
394
+
395
+ normalized = (name or "").replace("-", "").replace("_", "").casefold()
396
+ return any(hint in normalized for hint in _MONOSPACE_FONT_HINTS)
397
+
398
+
399
+ def _monospace_character_support(
400
+ lines: list[_LineItem],
401
+ ) -> tuple[float, dict[str, list[float]]]:
402
+ """统计等宽字体字符占比,并按东西文宽度组收集字符 advance。"""
403
+
404
+ supported = 0.0
405
+ total = 0.0
406
+ widths: dict[str, list[float]] = {"narrow": [], "wide": []}
407
+ for line in lines:
408
+ fallback_monospace = _font_name_looks_monospaced(line.font_signature[0] if line.font_signature is not None else None)
409
+ for char in line.chars:
410
+ if not isinstance(char, dict):
411
+ continue
412
+ value = str(char.get("char") or "")
413
+ if not value.strip():
414
+ continue
415
+ total += 1.0
416
+ font = char.get("font")
417
+ font_name = font.get("name") if isinstance(font, dict) else None
418
+ if _font_name_looks_monospaced(font_name) or fallback_monospace:
419
+ supported += 1.0
420
+ try:
421
+ x0, _y0, x1, _y1 = [float(item) for item in char.get("bbox", [])]
422
+ except (TypeError, ValueError):
423
+ continue
424
+ width = x1 - x0
425
+ if not math.isfinite(width) or width <= 0.1:
426
+ continue
427
+ width_group = "wide" if unicodedata.east_asian_width(value[0]) in {"W", "F"} else "narrow"
428
+ widths[width_group].append(width)
429
+
430
+ if total <= 0:
431
+ fallback_support = sum(_estimated_line_character_count(line) for line in lines)
432
+ if fallback_support <= 0:
433
+ return 0.0, widths
434
+ supported_support = sum(
435
+ _estimated_line_character_count(line)
436
+ for line in lines
437
+ if _font_name_looks_monospaced(line.font_signature[0] if line.font_signature is not None else None)
438
+ )
439
+ fallback_widths = [
440
+ line.median_glyph_width for line in lines if line.median_glyph_width is not None and line.median_glyph_width > 0
441
+ ]
442
+ widths["narrow"].extend(fallback_widths)
443
+ return supported_support / fallback_support, widths
444
+ return supported / total, widths
445
+
446
+
447
+ def _monospace_advances_are_stable(
448
+ widths: dict[str, list[float]],
449
+ ) -> bool:
450
+ """验证各字符宽度组的中位绝对偏差足够小,并校验中西文宽度关系。"""
451
+
452
+ populated = [values for values in widths.values() if values]
453
+ if not populated or sum(len(values) for values in populated) < 3:
454
+ return False
455
+ medians: dict[str, float] = {}
456
+ for group_name, values in widths.items():
457
+ if not values:
458
+ continue
459
+ median_width = statistics.median(values)
460
+ mad = statistics.median(abs(value - median_width) for value in values)
461
+ if mad / max(0.1, median_width) > 0.2:
462
+ return False
463
+ medians[group_name] = median_width
464
+ if "narrow" in medians and "wide" in medians:
465
+ ratio = medians["wide"] / medians["narrow"]
466
+ if not 1.25 <= ratio <= 2.25:
467
+ return False
468
+ return True
469
+
470
+
471
+ def _code_rows_have_spatial_structure(
472
+ lines: list[_LineItem],
473
+ candidate_bbox: BBox,
474
+ median_cell_width: float,
475
+ ) -> bool:
476
+ """验证代码行具有规则基线,并且左缘落在一致的等宽字符槽。"""
477
+
478
+ rows: list[list[_LineItem]] = []
479
+ for line in sorted(lines, key=lambda item: (_bbox_center_y(item.bbox), item.bbox[0])):
480
+ target = next(
481
+ (
482
+ row
483
+ for row in rows
484
+ if any(
485
+ min(member.bbox[3], line.bbox[3]) - max(member.bbox[1], line.bbox[1])
486
+ >= 0.5
487
+ * min(
488
+ member.bbox[3] - member.bbox[1],
489
+ line.bbox[3] - line.bbox[1],
490
+ )
491
+ for member in row
492
+ )
493
+ ),
494
+ None,
495
+ )
496
+ if target is None:
497
+ rows.append([line])
498
+ else:
499
+ target.append(line)
500
+ if not rows:
501
+ return False
502
+
503
+ residuals = []
504
+ for line in lines:
505
+ slot = (line.bbox[0] - candidate_bbox[0]) / max(0.1, median_cell_width)
506
+ residuals.append(abs(slot - round(slot)))
507
+ if sum(residual <= 0.35 for residual in residuals) / len(residuals) < 0.8:
508
+ return False
509
+ if len(rows) == 1:
510
+ return True
511
+
512
+ row_tops = sorted(min(line.bbox[1] for line in row) for row in rows)
513
+ deltas = [current - previous for previous, current in zip(row_tops, row_tops[1:]) if current > previous]
514
+ if not deltas:
515
+ return False
516
+ lower_count = max(1, math.ceil(0.6 * len(deltas)))
517
+ base_pitch = statistics.median(sorted(deltas)[:lower_count])
518
+ if base_pitch <= 0:
519
+ return False
520
+ return all(abs(delta / base_pitch - round(delta / base_pitch)) <= 0.35 for delta in deltas)
521
+
522
+
523
+ def _fallback_code_content(lines: list[_LineItem]) -> str:
524
+ """空间投影失败时按视觉行和水平位置保留代码文本的最小结构。"""
525
+
526
+ ordered = sorted(
527
+ lines,
528
+ key=lambda line: (
529
+ round(_bbox_center_y(line.bbox), 1),
530
+ _bbox_center_x(line.bbox),
531
+ line.source_index,
532
+ ),
533
+ )
534
+ content = "\n".join(line.text for line in ordered if line.text)
535
+ return _sanitize_pdf_control_text(content, preserve_newlines=True).strip()