docvortex 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (364) hide show
  1. docvortex/__init__.py +21 -0
  2. docvortex/analyzers/__init__.py +3 -0
  3. docvortex/analyzers/native/__init__.py +37 -0
  4. docvortex/analyzers/native/_shared/__init__.py +3 -0
  5. docvortex/analyzers/native/_shared/hyperlink.py +16 -0
  6. docvortex/analyzers/native/_shared/image.py +8 -0
  7. docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
  8. docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
  9. docvortex/analyzers/native/_shared/markup/formula.py +37 -0
  10. docvortex/analyzers/native/_shared/markup/projector.py +57 -0
  11. docvortex/analyzers/native/_shared/markup/styles.py +19 -0
  12. docvortex/analyzers/native/_shared/mathml.py +17 -0
  13. docvortex/analyzers/native/_shared/names.py +7 -0
  14. docvortex/analyzers/native/_shared/xycut.py +414 -0
  15. docvortex/analyzers/native/contracts.py +44 -0
  16. docvortex/analyzers/native/csv.py +351 -0
  17. docvortex/analyzers/native/epub/__init__.py +18 -0
  18. docvortex/analyzers/native/epub/constants.py +45 -0
  19. docvortex/analyzers/native/epub/converter.py +80 -0
  20. docvortex/analyzers/native/epub/errors.py +20 -0
  21. docvortex/analyzers/native/epub/metadata.py +31 -0
  22. docvortex/analyzers/native/epub/package.py +649 -0
  23. docvortex/analyzers/native/epub/xhtml.py +306 -0
  24. docvortex/analyzers/native/html/__init__.py +6 -0
  25. docvortex/analyzers/native/html/anchors.py +271 -0
  26. docvortex/analyzers/native/html/constants.py +25 -0
  27. docvortex/analyzers/native/html/contracts.py +7 -0
  28. docvortex/analyzers/native/html/converter.py +117 -0
  29. docvortex/analyzers/native/html/document.py +389 -0
  30. docvortex/analyzers/native/html/errors.py +12 -0
  31. docvortex/analyzers/native/html/resources.py +365 -0
  32. docvortex/analyzers/native/html/selector.py +421 -0
  33. docvortex/analyzers/native/models.py +209 -0
  34. docvortex/analyzers/native/ofd/__init__.py +14 -0
  35. docvortex/analyzers/native/ofd/constants.py +59 -0
  36. docvortex/analyzers/native/ofd/converter.py +31 -0
  37. docvortex/analyzers/native/ofd/errors.py +16 -0
  38. docvortex/analyzers/native/ofd/geometry.py +215 -0
  39. docvortex/analyzers/native/ofd/images.py +124 -0
  40. docvortex/analyzers/native/ofd/metadata.py +53 -0
  41. docvortex/analyzers/native/ofd/models.py +170 -0
  42. docvortex/analyzers/native/ofd/package.py +346 -0
  43. docvortex/analyzers/native/ofd/path.py +201 -0
  44. docvortex/analyzers/native/ofd/reading_order.py +293 -0
  45. docvortex/analyzers/native/ofd/resources.py +122 -0
  46. docvortex/analyzers/native/ofd/scene.py +347 -0
  47. docvortex/analyzers/native/ofd/table.py +240 -0
  48. docvortex/analyzers/native/ofd/text.py +521 -0
  49. docvortex/analyzers/native/office/__init__.py +3 -0
  50. docvortex/analyzers/native/office/doc/__init__.py +3 -0
  51. docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
  52. docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
  53. docvortex/analyzers/native/office/doc/fib.py +261 -0
  54. docvortex/analyzers/native/office/doc/fields.py +101 -0
  55. docvortex/analyzers/native/office/doc/formatting.py +171 -0
  56. docvortex/analyzers/native/office/doc/images.py +162 -0
  57. docvortex/analyzers/native/office/doc/lists.py +356 -0
  58. docvortex/analyzers/native/office/doc/models.py +164 -0
  59. docvortex/analyzers/native/office/doc/parser.py +851 -0
  60. docvortex/analyzers/native/office/doc/pieces.py +256 -0
  61. docvortex/analyzers/native/office/doc/records.py +40 -0
  62. docvortex/analyzers/native/office/doc/sprm.py +269 -0
  63. docvortex/analyzers/native/office/doc/styles.py +215 -0
  64. docvortex/analyzers/native/office/docx/__init__.py +3 -0
  65. docvortex/analyzers/native/office/docx/context.py +71 -0
  66. docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
  67. docvortex/analyzers/native/office/docx/equationxml.py +119 -0
  68. docvortex/analyzers/native/office/docx/fields.py +816 -0
  69. docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
  70. docvortex/analyzers/native/office/docx/main.py +50 -0
  71. docvortex/analyzers/native/office/docx/numbering.py +491 -0
  72. docvortex/analyzers/native/office/docx/office_xml.py +57 -0
  73. docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
  74. docvortex/analyzers/native/office/docx/resources.py +526 -0
  75. docvortex/analyzers/native/office/docx/styles.py +693 -0
  76. docvortex/analyzers/native/office/docx/tables.py +515 -0
  77. docvortex/analyzers/native/office/equation/__init__.py +3 -0
  78. docvortex/analyzers/native/office/equation/image.py +470 -0
  79. docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
  80. docvortex/analyzers/native/office/equation/mtef.py +885 -0
  81. docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
  82. docvortex/analyzers/native/office/equation/omml.py +561 -0
  83. docvortex/analyzers/native/office/equation/ooxml.py +62 -0
  84. docvortex/analyzers/native/office/errors.py +33 -0
  85. docvortex/analyzers/native/office/image.py +307 -0
  86. docvortex/analyzers/native/office/legacy/__init__.py +3 -0
  87. docvortex/analyzers/native/office/legacy/binary.py +43 -0
  88. docvortex/analyzers/native/office/legacy/officeart.py +362 -0
  89. docvortex/analyzers/native/office/legacy/ole.py +110 -0
  90. docvortex/analyzers/native/office/limits.py +12 -0
  91. docvortex/analyzers/native/office/odf/__init__.py +3 -0
  92. docvortex/analyzers/native/office/odf/chart.py +76 -0
  93. docvortex/analyzers/native/office/odf/constants.py +70 -0
  94. docvortex/analyzers/native/office/odf/converters.py +403 -0
  95. docvortex/analyzers/native/office/odf/errors.py +18 -0
  96. docvortex/analyzers/native/office/odf/metadata.py +91 -0
  97. docvortex/analyzers/native/office/odf/models.py +176 -0
  98. docvortex/analyzers/native/office/odf/package.py +270 -0
  99. docvortex/analyzers/native/office/odf/styles.py +329 -0
  100. docvortex/analyzers/native/office/odf/table.py +469 -0
  101. docvortex/analyzers/native/office/odf/text.py +1002 -0
  102. docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
  103. docvortex/analyzers/native/office/opc.py +38 -0
  104. docvortex/analyzers/native/office/ppt/__init__.py +3 -0
  105. docvortex/analyzers/native/office/ppt/models.py +118 -0
  106. docvortex/analyzers/native/office/ppt/parser.py +1895 -0
  107. docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
  108. docvortex/analyzers/native/office/ppt/records.py +131 -0
  109. docvortex/analyzers/native/office/ppt/style_text.py +247 -0
  110. docvortex/analyzers/native/office/pptx/__init__.py +3 -0
  111. docvortex/analyzers/native/office/pptx/context.py +113 -0
  112. docvortex/analyzers/native/office/pptx/lists.py +558 -0
  113. docvortex/analyzers/native/office/pptx/main.py +20 -0
  114. docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
  115. docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
  116. docvortex/analyzers/native/office/pptx/resources.py +329 -0
  117. docvortex/analyzers/native/office/pptx/shapes.py +393 -0
  118. docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
  119. docvortex/analyzers/native/office/pptx/titles.py +178 -0
  120. docvortex/analyzers/native/office/rich_text.py +420 -0
  121. docvortex/analyzers/native/office/rtf/__init__.py +3 -0
  122. docvortex/analyzers/native/office/rtf/converter.py +708 -0
  123. docvortex/analyzers/native/office/rtf/lexer.py +213 -0
  124. docvortex/analyzers/native/office/rtf/math.py +339 -0
  125. docvortex/analyzers/native/office/rtf/models.py +185 -0
  126. docvortex/analyzers/native/office/rtf/parser.py +1553 -0
  127. docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
  128. docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
  129. docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
  130. docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
  131. docvortex/analyzers/native/office/streams.py +18 -0
  132. docvortex/analyzers/native/office/xls/__init__.py +3 -0
  133. docvortex/analyzers/native/office/xls/chart.py +132 -0
  134. docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
  135. docvortex/analyzers/native/office/xls/models.py +109 -0
  136. docvortex/analyzers/native/office/xls/number_format.py +521 -0
  137. docvortex/analyzers/native/office/xls/parser.py +1145 -0
  138. docvortex/analyzers/native/office/xls/records.py +201 -0
  139. docvortex/analyzers/native/office/xls/strings.py +205 -0
  140. docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
  141. docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
  142. docvortex/analyzers/native/office/xlsx/main.py +20 -0
  143. docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
  144. docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
  145. docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
  146. docvortex/analyzers/native/pdf/__init__.py +3 -0
  147. docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
  148. docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
  149. docvortex/analyzers/native/pdf/code_blocks.py +535 -0
  150. docvortex/analyzers/native/pdf/formulas.py +1985 -0
  151. docvortex/analyzers/native/pdf/geometry.py +281 -0
  152. docvortex/analyzers/native/pdf/graphics.py +1501 -0
  153. docvortex/analyzers/native/pdf/index_blocks.py +268 -0
  154. docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
  155. docvortex/analyzers/native/pdf/inline/common.py +112 -0
  156. docvortex/analyzers/native/pdf/inline/detection.py +590 -0
  157. docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
  158. docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
  159. docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
  160. docvortex/analyzers/native/pdf/inline/types.py +385 -0
  161. docvortex/analyzers/native/pdf/line_layout.py +1106 -0
  162. docvortex/analyzers/native/pdf/line_merging.py +1223 -0
  163. docvortex/analyzers/native/pdf/models.py +246 -0
  164. docvortex/analyzers/native/pdf/native_text.py +1004 -0
  165. docvortex/analyzers/native/pdf/pipeline.py +1390 -0
  166. docvortex/analyzers/native/pdf/script_geometry.py +636 -0
  167. docvortex/analyzers/native/pdf/shared.py +30 -0
  168. docvortex/analyzers/native/pdf/spatial_text.py +383 -0
  169. docvortex/analyzers/native/pdf/table_annotations.py +446 -0
  170. docvortex/analyzers/native/pdf/table_constants.py +48 -0
  171. docvortex/analyzers/native/pdf/table_detection.py +227 -0
  172. docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
  173. docvortex/analyzers/native/pdf/table_geometry.py +40 -0
  174. docvortex/analyzers/native/pdf/table_materialization.py +444 -0
  175. docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
  176. docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
  177. docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
  178. docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
  179. docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
  180. docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
  181. docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
  182. docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
  183. docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
  184. docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
  185. docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
  186. docvortex/analyzers/native/pdf/table_rows.py +34 -0
  187. docvortex/analyzers/native/pdf/table_rules.py +1129 -0
  188. docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
  189. docvortex/analyzers/native/pdf/tables.py +147 -0
  190. docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
  191. docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
  192. docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
  193. docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
  194. docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
  195. docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
  196. docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
  197. docvortex/analyzers/native/pdf/text_blocks.py +82 -0
  198. docvortex/analyzers/native/pdf/text_styles.py +55 -0
  199. docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
  200. docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
  201. docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
  202. docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
  203. docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
  204. docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
  205. docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
  206. docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
  207. docvortex/analyzers/native/pdf/titles.py +75 -0
  208. docvortex/analyzers/native/pdf/typography.py +19 -0
  209. docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
  210. docvortex/api.py +180 -0
  211. docvortex/assets/__init__.py +5 -0
  212. docvortex/assets/store.py +51 -0
  213. docvortex/cli.py +54 -0
  214. docvortex/codecs/__init__.py +3 -0
  215. docvortex/codecs/html/__init__.py +22 -0
  216. docvortex/codecs/html/contracts.py +236 -0
  217. docvortex/codecs/html/materializer.py +331 -0
  218. docvortex/codecs/html/parser.py +763 -0
  219. docvortex/codecs/html/resources.py +34 -0
  220. docvortex/codecs/json.py +17 -0
  221. docvortex/content/__init__.py +5 -0
  222. docvortex/content/inline.py +248 -0
  223. docvortex/content/markup/__init__.py +44 -0
  224. docvortex/content/markup/anchors.py +188 -0
  225. docvortex/content/markup/formula.py +280 -0
  226. docvortex/content/markup/projector.py +1237 -0
  227. docvortex/content/markup/styles.py +327 -0
  228. docvortex/content/mathml.py +167 -0
  229. docvortex/content/normalization.py +188 -0
  230. docvortex/content/spans.py +183 -0
  231. docvortex/content/table/__init__.py +18 -0
  232. docvortex/content/table/blocks.py +152 -0
  233. docvortex/content/table/content.py +425 -0
  234. docvortex/content/table/document.py +104 -0
  235. docvortex/content/table/html.py +399 -0
  236. docvortex/content/table/models.py +76 -0
  237. docvortex/content/table/rules.py +42 -0
  238. docvortex/content/table/structure.py +221 -0
  239. docvortex/content/tree.py +5 -0
  240. docvortex/document/__init__.py +3 -0
  241. docvortex/document/contracts.py +22 -0
  242. docvortex/document/detection.py +389 -0
  243. docvortex/document/filetypes.py +175 -0
  244. docvortex/document/page_range.py +167 -0
  245. docvortex/document/pdf/__init__.py +19 -0
  246. docvortex/document/pdf/classify.py +1138 -0
  247. docvortex/document/pdf/constants.py +132 -0
  248. docvortex/document/pdf/diagnostics.py +350 -0
  249. docvortex/document/pdf/document.py +582 -0
  250. docvortex/document/pdf/font_runtime.py +335 -0
  251. docvortex/document/pdf/geometry.py +31 -0
  252. docvortex/document/pdf/images.py +654 -0
  253. docvortex/document/pdf/native_annotations.py +367 -0
  254. docvortex/document/pdf/native_contracts.py +169 -0
  255. docvortex/document/pdf/native_coordinates.py +216 -0
  256. docvortex/document/pdf/native_lifecycle.py +16 -0
  257. docvortex/document/pdf/native_objects.py +902 -0
  258. docvortex/document/pdf/native_text_geometry.py +315 -0
  259. docvortex/document/pdf/pdfium.py +325 -0
  260. docvortex/document/pdf/raster.py +46 -0
  261. docvortex/document/pdf/text/__init__.py +62 -0
  262. docvortex/document/pdf/text/contracts.py +211 -0
  263. docvortex/document/pdf/text/extract.py +165 -0
  264. docvortex/document/pdf/text/geometry.py +16 -0
  265. docvortex/document/pdf/text/groups.py +162 -0
  266. docvortex/document/pdf/visual_geometry.py +201 -0
  267. docvortex/document/pdf/visuals.py +343 -0
  268. docvortex/document/source.py +92 -0
  269. docvortex/errors.py +21 -0
  270. docvortex/export/__init__.py +3 -0
  271. docvortex/export/bundle.py +98 -0
  272. docvortex/export/files.py +66 -0
  273. docvortex/export/middle.py +208 -0
  274. docvortex/foundation/__init__.py +3 -0
  275. docvortex/foundation/geometry.py +125 -0
  276. docvortex/foundation/hyperlink.py +65 -0
  277. docvortex/foundation/image.py +48 -0
  278. docvortex/foundation/image_encoding.py +30 -0
  279. docvortex/foundation/image_payload.py +280 -0
  280. docvortex/foundation/language.py +92 -0
  281. docvortex/foundation/platform.py +38 -0
  282. docvortex/foundation/text.py +153 -0
  283. docvortex/foundation/type_identity.py +20 -0
  284. docvortex/foundation/xml_names.py +20 -0
  285. docvortex/options.py +30 -0
  286. docvortex/postprocess/__init__.py +3 -0
  287. docvortex/postprocess/content.py +53 -0
  288. docvortex/postprocess/document.py +19 -0
  289. docvortex/postprocess/lists.py +236 -0
  290. docvortex/postprocess/page_blocks.py +214 -0
  291. docvortex/postprocess/pages.py +95 -0
  292. docvortex/postprocess/paragraphs.py +580 -0
  293. docvortex/postprocess/visual.py +715 -0
  294. docvortex/render/__init__.py +48 -0
  295. docvortex/render/_internal/__init__.py +3 -0
  296. docvortex/render/_internal/common/__init__.py +3 -0
  297. docvortex/render/_internal/common/context.py +43 -0
  298. docvortex/render/_internal/common/html_table.py +178 -0
  299. docvortex/render/_internal/common/index.py +33 -0
  300. docvortex/render/_internal/common/list_items.py +158 -0
  301. docvortex/render/_internal/common/planner.py +140 -0
  302. docvortex/render/_internal/docx/__init__.py +3 -0
  303. docvortex/render/_internal/docx/assets.py +202 -0
  304. docvortex/render/_internal/docx/inline.py +434 -0
  305. docvortex/render/_internal/docx/math.py +220 -0
  306. docvortex/render/_internal/docx/renderer.py +905 -0
  307. docvortex/render/_internal/docx/styles.py +195 -0
  308. docvortex/render/_internal/docx/table.py +442 -0
  309. docvortex/render/_internal/epub/__init__.py +5 -0
  310. docvortex/render/_internal/epub/assets.py +173 -0
  311. docvortex/render/_internal/epub/package.py +249 -0
  312. docvortex/render/_internal/epub/renderer.py +1156 -0
  313. docvortex/render/_internal/html/__init__.py +3 -0
  314. docvortex/render/_internal/html/inline.py +346 -0
  315. docvortex/render/_internal/html/renderer.py +1041 -0
  316. docvortex/render/_internal/html/sanitizer.py +478 -0
  317. docvortex/render/_internal/html/table.py +121 -0
  318. docvortex/render/_internal/latex/__init__.py +1 -0
  319. docvortex/render/_internal/latex/assets.py +85 -0
  320. docvortex/render/_internal/latex/inline.py +145 -0
  321. docvortex/render/_internal/latex/renderer.py +506 -0
  322. docvortex/render/_internal/latex/table.py +347 -0
  323. docvortex/render/_internal/markdown/__init__.py +3 -0
  324. docvortex/render/_internal/markdown/assets.py +78 -0
  325. docvortex/render/_internal/markdown/blocks.py +635 -0
  326. docvortex/render/_internal/markdown/escaping.py +51 -0
  327. docvortex/render/_internal/markdown/inline.py +260 -0
  328. docvortex/render/_internal/markdown/renderer.py +93 -0
  329. docvortex/render/_internal/markdown/table.py +281 -0
  330. docvortex/render/_internal/pdf/__init__.py +3 -0
  331. docvortex/render/_internal/pdf/assets.py +197 -0
  332. docvortex/render/_internal/pdf/formula.py +417 -0
  333. docvortex/render/_internal/pdf/inline.py +343 -0
  334. docvortex/render/_internal/pdf/renderer.py +734 -0
  335. docvortex/render/_internal/pdf/styles.py +206 -0
  336. docvortex/render/_internal/pdf/table.py +272 -0
  337. docvortex/render/_internal/structured_content/__init__.py +3 -0
  338. docvortex/render/_internal/structured_content/renderer.py +193 -0
  339. docvortex/render/api.py +199 -0
  340. docvortex/render/contracts.py +205 -0
  341. docvortex/render/docx.py +38 -0
  342. docvortex/render/epub.py +35 -0
  343. docvortex/render/fragments.py +70 -0
  344. docvortex/render/html.py +29 -0
  345. docvortex/render/latex.py +24 -0
  346. docvortex/render/markdown.py +50 -0
  347. docvortex/render/pdf.py +25 -0
  348. docvortex/render/structured_content.py +23 -0
  349. docvortex/resources/epub/docvortex.css +91 -0
  350. docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
  351. docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
  352. docvortex/resources/fonts/NOTICE +190 -0
  353. docvortex/resources/fonts/manifest.json +9 -0
  354. docvortex/resources/html/docvortex.css +601 -0
  355. docvortex/resources/html/docvortex.min.css +1 -0
  356. docvortex/result.py +91 -0
  357. docvortex/schema.py +1141 -0
  358. docvortex/version.py +3 -0
  359. docvortex-0.2.1.dist-info/METADATA +193 -0
  360. docvortex-0.2.1.dist-info/RECORD +364 -0
  361. docvortex-0.2.1.dist-info/WHEEL +5 -0
  362. docvortex-0.2.1.dist-info/entry_points.txt +2 -0
  363. docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
  364. docvortex-0.2.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,1024 @@
1
+ """编排页面文档标题、跨栏标题及误判回退。"""
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+ import statistics
7
+
8
+ from .....schema import BBox
9
+ from ..geometry import _bbox_axis_overlap_ratio, _bbox_center_x, _bbox_center_y, _bbox_union_many, _rotate_bbox_to_upright
10
+ from ..inline.types import PDF_FONT_ITALIC_FLAG
11
+ from ..line_layout import (
12
+ _effective_text_row_gap,
13
+ _font_signatures_share_family,
14
+ _infer_text_lanes,
15
+ _line_effective_height,
16
+ _title_fonts_compatible,
17
+ )
18
+ from ..models import _DocumentBodyProfile, _DocumentTitleProfile, _LineItem, _TextLane
19
+ from .body_profile import _infer_lane_body_profile, _line_uses_document_regular_font
20
+ from .common import _build_physical_title_gap_map, _line_inside_visual_container, _line_near_visual_container
21
+ from .lane_titles import _classify_paragraph_titles_in_lane
22
+
23
+
24
+ def _classify_page_titles(
25
+ lines: list[_LineItem],
26
+ page_size: tuple[float, float],
27
+ *,
28
+ page_index: int,
29
+ container_bboxes: list[BBox],
30
+ caption_container_bboxes: list[BBox] | None = None,
31
+ document_body_profile: _DocumentBodyProfile | None = None,
32
+ document_title_profile: _DocumentTitleProfile | None = None,
33
+ ) -> None:
34
+ """只用页面几何与字体排版标注首页文档标题和各页段落标题。"""
35
+
36
+ for angle in sorted({line.angle for line in lines if line.semantic_type is None and not line.title_suppressed}):
37
+ line_geometry = [
38
+ (line, _rotate_bbox_to_upright(line.bbox, page_size, angle))
39
+ for line in lines
40
+ if line.angle == angle and (line.semantic_type is None or line.explicit_section_title) and not line.title_suppressed
41
+ ]
42
+ if not line_geometry:
43
+ continue
44
+ median_height = statistics.median(_line_effective_height(line, bbox) for line, bbox in line_geometry)
45
+ local_page_width = page_size[1] if angle in {90, 270} else page_size[0]
46
+ local_page_height = page_size[0] if angle in {90, 270} else page_size[1]
47
+ lanes = _infer_text_lanes(line_geometry, local_page_width, median_height)
48
+ for lane in lanes:
49
+ lane.lines.sort(key=lambda item: (item[1][1], item[1][0], item[0].source_index))
50
+ physical_gaps = _build_physical_title_gap_map(line_geometry)
51
+ grid_title_suppressions = _find_repeated_grid_title_suppressions(
52
+ lanes,
53
+ median_height,
54
+ )
55
+ local_container_bboxes = [_rotate_bbox_to_upright(bbox, page_size, angle) for bbox in container_bboxes]
56
+ grid_title_suppressions.update(
57
+ _find_container_visual_row_title_suppressions(
58
+ line_geometry,
59
+ local_container_bboxes,
60
+ median_height,
61
+ )
62
+ )
63
+
64
+ document_title_bottom: float | None = None
65
+ if page_index == 0:
66
+ document_title_bottom = _classify_document_title(
67
+ lanes,
68
+ local_page_height,
69
+ local_page_width,
70
+ document_body_profile=document_body_profile,
71
+ )
72
+ document_title_bottom = _expand_document_title_across_lanes(
73
+ line_geometry,
74
+ local_page_width,
75
+ local_page_height,
76
+ document_title_bottom,
77
+ )
78
+ document_title_bottom = _classify_additional_document_title_bands(
79
+ line_geometry,
80
+ local_page_width,
81
+ local_page_height,
82
+ document_title_bottom,
83
+ document_body_profile=document_body_profile,
84
+ )
85
+ preserve_front_matter_boundaries = _document_title_uses_page_fallback(
86
+ lanes,
87
+ document_body_profile=document_body_profile,
88
+ )
89
+
90
+ _classify_cross_lane_centered_section_titles(
91
+ line_geometry,
92
+ lanes,
93
+ local_page_width,
94
+ local_page_height,
95
+ local_container_bboxes,
96
+ page_index=page_index,
97
+ document_title_bottom=document_title_bottom,
98
+ )
99
+ _classify_cross_lane_emphasized_section_titles(
100
+ line_geometry,
101
+ lanes,
102
+ local_page_width,
103
+ local_page_height,
104
+ local_container_bboxes,
105
+ document_title_bottom=document_title_bottom,
106
+ document_body_profile=document_body_profile,
107
+ )
108
+
109
+ for lane in lanes:
110
+ profile = _infer_lane_body_profile(lane)
111
+ _classify_paragraph_titles_in_lane(
112
+ lane,
113
+ profile,
114
+ local_page_width,
115
+ local_page_height,
116
+ local_container_bboxes,
117
+ document_title_bottom=document_title_bottom,
118
+ preserve_front_matter_boundaries=preserve_front_matter_boundaries,
119
+ physical_gaps=physical_gaps,
120
+ grid_title_suppressions=grid_title_suppressions,
121
+ document_body_profile=document_body_profile,
122
+ document_title_profile=document_title_profile,
123
+ page_index=page_index,
124
+ )
125
+ _expand_cross_lane_paragraph_title_neighbors(
126
+ line_geometry,
127
+ )
128
+ _demote_hanging_multiline_text_titles(
129
+ lanes,
130
+ document_body_profile,
131
+ page_index=page_index,
132
+ )
133
+ _demote_cross_lane_body_continuation_titles(
134
+ line_geometry,
135
+ lanes,
136
+ )
137
+ _demote_visual_container_caption_titles(
138
+ line_geometry,
139
+ [_rotate_bbox_to_upright(bbox, page_size, angle) for bbox in (caption_container_bboxes or [])],
140
+ )
141
+ _demote_sentence_tail_titles(
142
+ [
143
+ (
144
+ line,
145
+ _rotate_bbox_to_upright(
146
+ line.bbox,
147
+ page_size,
148
+ angle,
149
+ ),
150
+ )
151
+ for line in lines
152
+ if line.angle == angle and line.semantic_type in {None, "paragraph_title"} and not line.title_suppressed
153
+ ],
154
+ )
155
+ if document_body_profile is not None and document_body_profile.has_style_scale_repairs:
156
+ _demote_non_structural_anomaly_titles(line_geometry)
157
+
158
+
159
+ def _demote_non_structural_anomaly_titles(
160
+ line_geometry: list[tuple[_LineItem, BBox]],
161
+ ) -> None:
162
+ """在 loose 高度异常文档中只保留预先通过结构转折校验的段落标题。"""
163
+
164
+ for line, _bbox in line_geometry:
165
+ if line.semantic_type == "paragraph_title" and not line.structural_title:
166
+ line.semantic_type = None
167
+
168
+
169
+ def _find_repeated_grid_title_suppressions(
170
+ lanes: list[_TextLane],
171
+ median_height: float,
172
+ ) -> set[int]:
173
+ """识别重复双栏信息网格中的短首行,避免把城市等记录头标成标题。"""
174
+
175
+ candidates: list[tuple[int, int, float]] = []
176
+ for lane_index, lane in enumerate(lanes):
177
+ if lane.is_span:
178
+ continue
179
+ lane_width = max(0.1, lane.right - lane.left)
180
+ for row_index, (line, bbox) in enumerate(lane.lines[:-1]):
181
+ next_line, next_bbox = lane.lines[row_index + 1]
182
+ width = bbox[2] - bbox[0]
183
+ next_width = next_bbox[2] - next_bbox[0]
184
+ gap = _effective_text_row_gap(
185
+ (line, bbox),
186
+ (next_line, next_bbox),
187
+ )
188
+ if (
189
+ line.semantic_type is not None
190
+ or next_line.semantic_type is not None
191
+ or width > 0.35 * lane_width
192
+ or next_width < max(0.6 * lane_width, 1.8 * width)
193
+ or abs(next_bbox[0] - bbox[0]) > 0.75 * median_height
194
+ or not -0.25 * median_height <= gap <= 0.75 * median_height
195
+ ):
196
+ continue
197
+ candidates.append((lane_index, line.source_index, _bbox_center_y(bbox)))
198
+
199
+ bands: list[list[tuple[int, int, float]]] = []
200
+ for candidate in sorted(candidates, key=lambda item: item[2]):
201
+ target = next(
202
+ (band for band in bands if abs(candidate[2] - statistics.median(item[2] for item in band)) <= 0.5 * median_height),
203
+ None,
204
+ )
205
+ if target is None:
206
+ bands.append([candidate])
207
+ else:
208
+ target.append(candidate)
209
+ paired_bands = [
210
+ {source_index for _lane_index, source_index, _center_y in band}
211
+ for band in bands
212
+ if len({lane_index for lane_index, _source_index, _center_y in band}) >= 2
213
+ ]
214
+ if len(paired_bands) < 2:
215
+ return set()
216
+ return set().union(*paired_bands)
217
+
218
+
219
+ def _find_container_visual_row_title_suppressions(
220
+ line_geometry: list[tuple[_LineItem, BBox]],
221
+ container_bboxes: list[BBox],
222
+ median_height: float,
223
+ ) -> set[int]:
224
+ """用完整视觉行与图表容器的邻接关系抑制拆分 caption 标题误报。"""
225
+
226
+ visual_rows: dict[int, list[tuple[_LineItem, BBox]]] = {}
227
+ for item in line_geometry:
228
+ row_id = item[0].visual_row_id
229
+ if row_id is not None:
230
+ visual_rows.setdefault(row_id, []).append(item)
231
+ suppressed: set[int] = set()
232
+ for members in visual_rows.values():
233
+ if len(members) < 2:
234
+ continue
235
+ row_bbox = _bbox_union_many([bbox for _line, bbox in members])
236
+ if any(
237
+ _bbox_axis_overlap_ratio(row_bbox, container_bbox, axis="x") >= 0.35
238
+ and max(
239
+ row_bbox[1] - container_bbox[3],
240
+ container_bbox[1] - row_bbox[3],
241
+ 0.0,
242
+ )
243
+ <= 1.5 * median_height
244
+ for container_bbox in container_bboxes
245
+ ):
246
+ suppressed.update(line.source_index for line, _bbox in members)
247
+ return suppressed
248
+
249
+
250
+ def _classify_document_title(
251
+ lanes: list[_TextLane],
252
+ local_page_height: float,
253
+ local_page_width: float,
254
+ *,
255
+ document_body_profile: _DocumentBodyProfile | None = None,
256
+ ) -> float | None:
257
+ """从首页上部选取显著大字号锚点,并用同版式邻行扩展多行文档标题。"""
258
+
259
+ candidates: list[tuple[float, _TextLane, int, tuple[_LineItem, BBox]]] = []
260
+ lane_profiles = [(lane, _infer_lane_body_profile(lane)) for lane in lanes]
261
+ column_body_heights = [profile.body_height for lane, profile in lane_profiles if not lane.is_span]
262
+ if document_body_profile is not None:
263
+ document_body_height = document_body_profile.body_height
264
+ else:
265
+ document_body_height = (
266
+ statistics.median(column_body_heights)
267
+ if column_body_heights
268
+ else min(
269
+ (profile.body_height for _lane, profile in lane_profiles),
270
+ default=1.0,
271
+ )
272
+ )
273
+ for lane, profile in lane_profiles:
274
+ lane_width = max(0.1, lane.right - lane.left)
275
+ available = [item for item in lane.lines if item[0].semantic_type is None]
276
+ for row_index, item in enumerate(available):
277
+ line, bbox = item
278
+ reference_height = min(profile.body_height, 1.25 * document_body_height)
279
+ height_ratio = _line_effective_height(line, bbox) / max(0.1, reference_height)
280
+ document_height_ratio = _line_effective_height(line, bbox) / max(
281
+ 0.1,
282
+ document_body_height,
283
+ )
284
+ width_ratio = (bbox[2] - bbox[0]) / lane_width
285
+ centered = abs(_bbox_center_x(bbox) - (lane.left + lane.right) / 2.0) <= 0.15 * lane_width
286
+ page_centered = abs(_bbox_center_x(bbox) - 0.5 * local_page_width) <= 0.15 * local_page_width
287
+ page_width_ratio = (bbox[2] - bbox[0]) / max(0.1, local_page_width)
288
+ spans_columns_fallback = lane.is_span and centered and width_ratio >= 0.65 and document_height_ratio >= 1.3
289
+ if (
290
+ _bbox_center_y(bbox) > 0.45 * local_page_height
291
+ or (height_ratio < 1.4 and not spans_columns_fallback)
292
+ or width_ratio < 0.2
293
+ or (not centered and height_ratio < 1.7)
294
+ or (not page_centered and page_width_ratio < 0.45 and height_ratio < 1.8)
295
+ ):
296
+ continue
297
+ top_preference = max(0.0, 0.45 - _bbox_center_y(bbox) / local_page_height)
298
+ top_preference_weight = (
299
+ 4.0 if document_body_profile is not None and document_body_profile.has_style_scale_repairs else 1.0
300
+ )
301
+ score = (
302
+ height_ratio
303
+ + (0.75 if centered else 0.0)
304
+ + (1.25 if page_centered else 0.0)
305
+ + (0.75 if page_width_ratio >= 0.55 else 0.0)
306
+ + top_preference_weight * top_preference
307
+ - 0.02 * row_index
308
+ )
309
+ candidates.append((score, lane, row_index, item))
310
+ if not candidates:
311
+ return None
312
+
313
+ _score, lane, _row_index, anchor = max(candidates, key=lambda item: item[0])
314
+ anchor_line, anchor_bbox = anchor
315
+ anchor_height = _line_effective_height(anchor_line, anchor_bbox)
316
+ anchor_line.semantic_type = "doc_title"
317
+ selected = [anchor]
318
+ ordered = lane.lines
319
+ anchor_index = ordered.index(anchor)
320
+ for direction in (-1, 1):
321
+ index = anchor_index + direction
322
+ previous_bbox = anchor_bbox
323
+ while 0 <= index < len(ordered):
324
+ candidate_line, candidate_bbox = ordered[index]
325
+ if candidate_line.semantic_type is not None:
326
+ break
327
+ candidate_height = _line_effective_height(candidate_line, candidate_bbox)
328
+ if not 0.8 <= candidate_height / anchor_height <= 1.25:
329
+ break
330
+ if not _document_title_fonts_compatible(anchor_line, candidate_line):
331
+ break
332
+ vertical_gap = max(candidate_bbox[1] - previous_bbox[3], previous_bbox[1] - candidate_bbox[3], 0.0)
333
+ lane_width = max(0.1, lane.right - lane.left)
334
+ centered = abs(_bbox_center_x(candidate_bbox) - _bbox_center_x(anchor_bbox)) <= 0.18 * lane_width
335
+ aligned = abs(candidate_bbox[0] - anchor_bbox[0]) <= 0.75 * anchor_height
336
+ if vertical_gap > 1.1 * anchor_height or not (centered or aligned):
337
+ break
338
+ candidate_line.semantic_type = "doc_title"
339
+ selected.append((candidate_line, candidate_bbox))
340
+ previous_bbox = candidate_bbox
341
+ index += direction
342
+ return max(bbox[3] for _line, bbox in selected)
343
+
344
+
345
+ def _expand_document_title_across_lanes(
346
+ line_geometry: list[tuple[_LineItem, BBox]],
347
+ local_page_width: float,
348
+ local_page_height: float,
349
+ document_title_bottom: float | None,
350
+ ) -> float | None:
351
+ """跨错误推断栏扩展紧邻、同字号且对齐的多行文档标题。"""
352
+
353
+ title_items = [item for item in line_geometry if item[0].semantic_type == "doc_title"]
354
+ if not title_items:
355
+ return document_title_bottom
356
+ selected_ids = {id(line) for line, _bbox in title_items}
357
+ changed = True
358
+ while changed:
359
+ changed = False
360
+ title_bbox = _bbox_union_many([bbox for _line, bbox in title_items])
361
+ title_height = statistics.median(_line_effective_height(line, bbox) for line, bbox in title_items)
362
+ anchors = [line for line, _bbox in title_items]
363
+ candidates = [
364
+ item
365
+ for item in line_geometry
366
+ if id(item[0]) not in selected_ids
367
+ and item[0].semantic_type is None
368
+ and _bbox_center_y(item[1]) <= 0.45 * local_page_height
369
+ ]
370
+ candidates.sort(
371
+ key=lambda item: min(
372
+ abs(item[1][1] - title_bbox[3]),
373
+ abs(title_bbox[1] - item[1][3]),
374
+ )
375
+ )
376
+ for candidate_line, candidate_bbox in candidates:
377
+ candidate_height = _line_effective_height(
378
+ candidate_line,
379
+ candidate_bbox,
380
+ )
381
+ if not 0.8 <= candidate_height / max(0.1, title_height) <= 1.25:
382
+ continue
383
+ if not any(_document_title_fonts_compatible(anchor, candidate_line) for anchor in anchors):
384
+ continue
385
+ vertical_gap = max(
386
+ candidate_bbox[1] - title_bbox[3],
387
+ title_bbox[1] - candidate_bbox[3],
388
+ 0.0,
389
+ )
390
+ centered = abs(_bbox_center_x(candidate_bbox) - _bbox_center_x(title_bbox)) <= 0.18 * local_page_width
391
+ aligned = abs(candidate_bbox[0] - title_bbox[0]) <= 0.75 * title_height
392
+ title_width = max(0.1, title_bbox[2] - title_bbox[0])
393
+ candidate_width = candidate_bbox[2] - candidate_bbox[0]
394
+ aligned_continuation = aligned and candidate_width >= 0.25 * title_width
395
+ centered_continuation = centered and candidate_width >= 0.45 * title_width
396
+ if vertical_gap > 1.1 * title_height or not (aligned_continuation or centered_continuation):
397
+ continue
398
+ candidate_line.semantic_type = "doc_title"
399
+ title_items.append((candidate_line, candidate_bbox))
400
+ selected_ids.add(id(candidate_line))
401
+ changed = True
402
+ break
403
+ return max(bbox[3] for _line, bbox in title_items)
404
+
405
+
406
+ def _classify_additional_document_title_bands(
407
+ line_geometry: list[tuple[_LineItem, BBox]],
408
+ local_page_width: float,
409
+ local_page_height: float,
410
+ document_title_bottom: float | None,
411
+ *,
412
+ document_body_profile: _DocumentBodyProfile | None,
413
+ ) -> float | None:
414
+ """用居中译题与后续作者行结构补充首页第二文档标题带。"""
415
+
416
+ if document_title_bottom is None or document_body_profile is None or not document_body_profile.has_style_scale_repairs:
417
+ return document_title_bottom
418
+ body_height = max(0.1, document_body_profile.body_height)
419
+ ordered = sorted(
420
+ line_geometry,
421
+ key=lambda item: (item[1][1], item[1][0], item[0].source_index),
422
+ )
423
+ selected_bottoms = [bbox[3] for line, bbox in ordered if line.semantic_type == "doc_title"]
424
+ for index, (line, bbox) in enumerate(ordered):
425
+ if line.semantic_type is not None:
426
+ continue
427
+ center_y = _bbox_center_y(bbox)
428
+ width_ratio = (bbox[2] - bbox[0]) / max(0.1, local_page_width)
429
+ if not (document_title_bottom + 4.0 * body_height <= center_y <= 0.58 * local_page_height):
430
+ continue
431
+ if not 0.4 <= width_ratio <= 0.8:
432
+ continue
433
+ if abs(_bbox_center_x(bbox) - 0.5 * local_page_width) > 0.08 * local_page_width:
434
+ continue
435
+ if _line_uses_document_regular_font(line, document_body_profile):
436
+ continue
437
+ following = [
438
+ item for item in ordered[index + 1 :] if item[0].semantic_type is None and _bbox_center_y(item[1]) > center_y
439
+ ]
440
+ if not following:
441
+ continue
442
+ _next_line, next_bbox = following[0]
443
+ next_width_ratio = (next_bbox[2] - next_bbox[0]) / max(0.1, local_page_width)
444
+ next_centered = abs(_bbox_center_x(next_bbox) - 0.5 * local_page_width) <= 0.12 * local_page_width
445
+ vertical_gap = max(0.0, next_bbox[1] - bbox[3])
446
+ if not next_centered or next_width_ratio > 0.45 or vertical_gap > 2.0 * body_height:
447
+ continue
448
+ line.semantic_type = "doc_title"
449
+ selected_bottoms.append(bbox[3])
450
+ return max(selected_bottoms, default=document_title_bottom)
451
+
452
+
453
+ def _classify_cross_lane_centered_section_titles(
454
+ line_geometry: list[tuple[_LineItem, BBox]],
455
+ lanes: list[_TextLane],
456
+ local_page_width: float,
457
+ local_page_height: float,
458
+ container_bboxes: list[BBox],
459
+ *,
460
+ page_index: int,
461
+ document_title_bottom: float | None,
462
+ ) -> None:
463
+ """用正文栏中心、上下留白和正文邻行补标被单独推成窄栏的标题。"""
464
+
465
+ stable_lanes = [
466
+ lane for lane in lanes if not lane.is_span and len(lane.lines) >= 5 and lane.right - lane.left >= 0.2 * local_page_width
467
+ ]
468
+ for line, bbox in line_geometry:
469
+ if line.semantic_type is not None:
470
+ continue
471
+ if not 0.07 * local_page_height <= _bbox_center_y(bbox) <= 0.93 * local_page_height:
472
+ continue
473
+ candidate_lanes = [lane for lane in stable_lanes if lane.left <= _bbox_center_x(bbox) <= lane.right]
474
+ if not candidate_lanes:
475
+ continue
476
+ lane = min(
477
+ candidate_lanes,
478
+ key=lambda candidate: abs(_bbox_center_x(bbox) - 0.5 * (candidate.left + candidate.right)),
479
+ )
480
+ profile = _infer_lane_body_profile(lane)
481
+ lane_width = max(0.1, lane.right - lane.left)
482
+ line_width = bbox[2] - bbox[0]
483
+ line_height = _line_effective_height(line, bbox)
484
+ if not 0.08 * lane_width <= line_width <= 0.7 * lane_width:
485
+ continue
486
+ if abs(_bbox_center_x(bbox) - 0.5 * (lane.left + lane.right)) > 0.05 * lane_width:
487
+ continue
488
+ if not 0.75 <= line_height / max(0.1, profile.body_height) <= 1.3:
489
+ continue
490
+ uses_regular_font = (
491
+ profile.body_font is not None
492
+ and line.font_signature is not None
493
+ and line.font_coverage >= 0.75
494
+ and _font_signatures_share_family(
495
+ line.font_signature,
496
+ profile.body_font,
497
+ )
498
+ )
499
+ weight_emphasized = (
500
+ profile.body_weight is not None
501
+ and line.dominant_font_weight is not None
502
+ and line.dominant_font_weight
503
+ >= max(
504
+ profile.body_weight + 100.0,
505
+ 1.15 * profile.body_weight,
506
+ )
507
+ )
508
+ if page_index == 0 and line_height < 0.9 * profile.body_height and uses_regular_font and not weight_emphasized:
509
+ continue
510
+ if document_title_bottom is not None and bbox[1] <= document_title_bottom + profile.body_height:
511
+ continue
512
+ if _line_inside_visual_container(bbox, container_bboxes) or _line_near_visual_container(
513
+ bbox,
514
+ container_bboxes,
515
+ profile.body_height,
516
+ ):
517
+ continue
518
+ body_rows = [
519
+ item
520
+ for item in lane.lines
521
+ if item[0] is not line
522
+ and item[0].semantic_type is None
523
+ and item[1][2] - item[1][0] >= 0.5 * lane_width
524
+ and 0.75 <= _line_effective_height(*item) / max(0.1, profile.body_height) <= 1.3
525
+ ]
526
+ rows_above = [item for item in body_rows if _bbox_center_y(item[1]) < _bbox_center_y(bbox)]
527
+ rows_below = [item for item in body_rows if _bbox_center_y(item[1]) > _bbox_center_y(bbox)]
528
+ if not rows_above or not rows_below:
529
+ continue
530
+ previous = max(rows_above, key=lambda item: _bbox_center_y(item[1]))
531
+ following_rows = sorted(
532
+ rows_below,
533
+ key=lambda item: _bbox_center_y(item[1]),
534
+ )
535
+ if len(following_rows) < 3 or any(
536
+ _effective_text_row_gap(previous_row, current_row) > profile.regular_gap + 0.75 * profile.body_height
537
+ for previous_row, current_row in zip(
538
+ following_rows[:3],
539
+ following_rows[1:3],
540
+ )
541
+ ):
542
+ continue
543
+ following = following_rows[0]
544
+ gap_above = bbox[1] - previous[1][3]
545
+ gap_below = following[1][1] - bbox[3]
546
+ if (
547
+ gap_above < 0.3 * profile.body_height
548
+ or gap_below < -0.1 * profile.body_height
549
+ or max(gap_above, gap_below) < profile.regular_gap + 0.2 * profile.body_height
550
+ ):
551
+ continue
552
+ line.semantic_type = "paragraph_title"
553
+
554
+
555
+ def _demote_cross_lane_body_continuation_titles(
556
+ line_geometry: list[tuple[_LineItem, BBox]],
557
+ lanes: list[_TextLane],
558
+ ) -> None:
559
+ """把紧接上一正文行、同字号同字体的短续行从标题降回正文。"""
560
+
561
+ stable_lanes = [lane for lane in lanes if not lane.is_span and len(lane.lines) >= 4]
562
+ for line, bbox in line_geometry:
563
+ if line.semantic_type != "paragraph_title":
564
+ continue
565
+ line_height = _line_effective_height(line, bbox)
566
+ global_preceding = [
567
+ item
568
+ for item in line_geometry
569
+ if item[0] is not line
570
+ and item[0].semantic_type is None
571
+ and item[0].angle == line.angle
572
+ and _bbox_center_y(item[1]) < _bbox_center_y(bbox)
573
+ and abs(item[1][0] - bbox[0]) <= line_height
574
+ ]
575
+ if global_preceding:
576
+ previous_line, previous_bbox = max(
577
+ global_preceding,
578
+ key=lambda item: _bbox_center_y(item[1]),
579
+ )
580
+ previous_height = _line_effective_height(
581
+ previous_line,
582
+ previous_bbox,
583
+ )
584
+ same_family = (
585
+ previous_line.font_signature is None
586
+ or line.font_signature is None
587
+ or _font_signatures_share_family(
588
+ previous_line.font_signature,
589
+ line.font_signature,
590
+ )
591
+ )
592
+ if (
593
+ same_family
594
+ and 0.9 <= line_height / max(0.1, previous_height) <= 1.1
595
+ and bbox[1] - previous_bbox[3] <= 0.2 * max(line_height, previous_height)
596
+ and abs(bbox[0] - previous_bbox[0]) <= 0.75 * max(line_height, previous_height)
597
+ and previous_bbox[2] - previous_bbox[0] >= 2.0 * (bbox[2] - bbox[0])
598
+ ):
599
+ line.semantic_type = None
600
+ continue
601
+ candidate_lanes = [lane for lane in stable_lanes if lane.left <= _bbox_center_x(bbox) <= lane.right]
602
+ if not candidate_lanes:
603
+ continue
604
+ lane = min(
605
+ candidate_lanes,
606
+ key=lambda candidate: abs(_bbox_center_x(bbox) - 0.5 * (candidate.left + candidate.right)),
607
+ )
608
+ lane_width = max(0.1, lane.right - lane.left)
609
+ lane_profile = _infer_lane_body_profile(lane)
610
+ if bbox[2] - bbox[0] > 0.55 * lane_width:
611
+ continue
612
+ preceding = [
613
+ item
614
+ for item in line_geometry
615
+ if item[0] is not line
616
+ and item[0].semantic_type is None
617
+ and _bbox_center_y(item[1]) < _bbox_center_y(bbox)
618
+ and lane.left <= _bbox_center_x(item[1]) <= lane.right
619
+ ]
620
+ if not preceding:
621
+ continue
622
+ previous_line, previous_bbox = max(
623
+ preceding,
624
+ key=lambda item: _bbox_center_y(item[1]),
625
+ )
626
+ previous_height = _line_effective_height(
627
+ previous_line,
628
+ previous_bbox,
629
+ )
630
+ line_height = _line_effective_height(line, bbox)
631
+ if not 0.9 <= line_height / max(0.1, previous_height) <= 1.1:
632
+ continue
633
+ if (
634
+ previous_line.font_signature is not None
635
+ and line.font_signature is not None
636
+ and not _font_signatures_share_family(
637
+ previous_line.font_signature,
638
+ line.font_signature,
639
+ )
640
+ ):
641
+ continue
642
+ if previous_bbox[2] - previous_bbox[0] < 0.5 * lane_width:
643
+ continue
644
+ maximum_gap = 0.15 * max(line_height, previous_height)
645
+ if (
646
+ re.search(
647
+ r"[.!?。!?][\]\)})】》”’'\"]*$",
648
+ line.text.rstrip(),
649
+ )
650
+ is not None
651
+ ):
652
+ maximum_gap = max(
653
+ maximum_gap,
654
+ lane_profile.regular_gap + 0.35 * lane_profile.body_height,
655
+ )
656
+ if bbox[1] - previous_bbox[3] > maximum_gap:
657
+ continue
658
+ if abs(bbox[0] - lane.left) > max(1.0, 0.75 * line_height):
659
+ continue
660
+ line.semantic_type = None
661
+
662
+
663
+ def _classify_cross_lane_emphasized_section_titles(
664
+ line_geometry: list[tuple[_LineItem, BBox]],
665
+ lanes: list[_TextLane],
666
+ local_page_width: float,
667
+ local_page_height: float,
668
+ container_bboxes: list[BBox],
669
+ *,
670
+ document_title_bottom: float | None,
671
+ document_body_profile: _DocumentBodyProfile | None,
672
+ ) -> None:
673
+ """用正文栏左缘、强调字体和段间留白补标跨栏推断失败的小节标题。"""
674
+
675
+ stable_lanes = [
676
+ lane for lane in lanes if not lane.is_span and len(lane.lines) >= 5 and lane.right - lane.left >= 0.2 * local_page_width
677
+ ]
678
+ for line, bbox in line_geometry:
679
+ if line.semantic_type is not None or line.font_signature is None:
680
+ continue
681
+ if not line.font_signature[1] & PDF_FONT_ITALIC_FLAG:
682
+ continue
683
+ if _line_uses_document_regular_font(line, document_body_profile):
684
+ continue
685
+ if not 0.07 * local_page_height <= _bbox_center_y(bbox) <= 0.93 * local_page_height:
686
+ continue
687
+ candidate_lanes = [lane for lane in stable_lanes if lane.left <= _bbox_center_x(bbox) <= lane.right]
688
+ if not candidate_lanes:
689
+ continue
690
+ lane = min(
691
+ candidate_lanes,
692
+ key=lambda candidate: abs(bbox[0] - candidate.left),
693
+ )
694
+ profile = _infer_lane_body_profile(lane)
695
+ if profile.body_font is None or line.font_signature == profile.body_font or line.font_coverage < 0.75:
696
+ continue
697
+ lane_width = max(0.1, lane.right - lane.left)
698
+ line_height = _line_effective_height(line, bbox)
699
+ if bbox[2] - bbox[0] > 0.9 * lane_width:
700
+ continue
701
+ if abs(bbox[0] - lane.left) > 2.0 * profile.body_height:
702
+ continue
703
+ if not 0.8 <= line_height / max(0.1, profile.body_height) <= 1.35:
704
+ continue
705
+ if document_title_bottom is not None and bbox[1] <= document_title_bottom + 2.0 * profile.body_height:
706
+ continue
707
+ if _line_inside_visual_container(bbox, container_bboxes) or _line_near_visual_container(
708
+ bbox,
709
+ container_bboxes,
710
+ profile.body_height,
711
+ ):
712
+ continue
713
+ body_rows = [
714
+ item
715
+ for item in lane.lines
716
+ if item[0] is not line
717
+ and item[0].semantic_type is None
718
+ and item[0].font_signature == profile.body_font
719
+ and item[1][2] - item[1][0] >= 0.45 * lane_width
720
+ ]
721
+ rows_above = [item for item in body_rows if _bbox_center_y(item[1]) < _bbox_center_y(bbox)]
722
+ rows_below = [item for item in body_rows if _bbox_center_y(item[1]) > _bbox_center_y(bbox)]
723
+ if not rows_above or not rows_below:
724
+ continue
725
+ previous = max(rows_above, key=lambda item: _bbox_center_y(item[1]))
726
+ following = min(rows_below, key=lambda item: _bbox_center_y(item[1]))
727
+ gap_above = bbox[1] - previous[1][3]
728
+ gap_below = following[1][1] - bbox[3]
729
+ if (
730
+ gap_above < 0.35 * profile.body_height
731
+ or gap_below > 3.0 * profile.body_height
732
+ or gap_below < -0.1 * profile.body_height
733
+ ):
734
+ continue
735
+ line.semantic_type = "paragraph_title"
736
+
737
+
738
+ def _is_wide_leading_title_continuation(
739
+ title_bbox: BBox,
740
+ candidate_bbox: BBox,
741
+ title_height: float,
742
+ candidate_height: float,
743
+ ) -> bool:
744
+ """识别紧贴在窄标题锚点上方、同中心的较宽首行。"""
745
+
746
+ pair_height = max(0.1, title_height, candidate_height)
747
+ title_width = max(0.1, title_bbox[2] - title_bbox[0])
748
+ width_ratio = (candidate_bbox[2] - candidate_bbox[0]) / title_width
749
+ vertical_gap = title_bbox[1] - candidate_bbox[3]
750
+ return (
751
+ 2.0 < width_ratio <= 2.5
752
+ and _bbox_center_y(candidate_bbox) < _bbox_center_y(title_bbox)
753
+ and -0.1 * pair_height <= vertical_gap <= 0.15 * pair_height
754
+ and _bbox_axis_overlap_ratio(title_bbox, candidate_bbox, axis="x") >= 0.8
755
+ and abs(_bbox_center_x(candidate_bbox) - _bbox_center_x(title_bbox)) <= 0.5 * pair_height
756
+ )
757
+
758
+
759
+ def _expand_cross_lane_paragraph_title_neighbors(
760
+ line_geometry: list[tuple[_LineItem, BBox]],
761
+ ) -> None:
762
+ """把紧贴标题锚点的同字体相邻行跨栏补标为同一标题。"""
763
+
764
+ changed = True
765
+ while changed:
766
+ changed = False
767
+ title_items = [
768
+ item for item in line_geometry if item[0].semantic_type == "paragraph_title" and not item[0].explicit_section_title
769
+ ]
770
+ for title_line, title_bbox in title_items:
771
+ title_height = _line_effective_height(title_line, title_bbox)
772
+ for candidate_line, candidate_bbox in line_geometry:
773
+ if candidate_line.semantic_type is not None:
774
+ continue
775
+ candidate_height = _line_effective_height(
776
+ candidate_line,
777
+ candidate_bbox,
778
+ )
779
+ if not 0.85 <= candidate_height / max(0.1, title_height) <= 1.2:
780
+ continue
781
+ title_width = max(0.1, title_bbox[2] - title_bbox[0])
782
+ candidate_width = candidate_bbox[2] - candidate_bbox[0]
783
+ if not (
784
+ 0.25 <= candidate_width / title_width <= 2.0
785
+ or _is_wide_leading_title_continuation(
786
+ title_bbox,
787
+ candidate_bbox,
788
+ title_height,
789
+ candidate_height,
790
+ )
791
+ ):
792
+ continue
793
+ if (
794
+ title_line.font_signature is not None
795
+ and candidate_line.font_signature is not None
796
+ and title_line.font_signature != candidate_line.font_signature
797
+ ):
798
+ continue
799
+ vertical_gap = max(
800
+ candidate_bbox[1] - title_bbox[3],
801
+ title_bbox[1] - candidate_bbox[3],
802
+ 0.0,
803
+ )
804
+ if vertical_gap > 0.35 * max(title_height, candidate_height):
805
+ continue
806
+ if (
807
+ _bbox_axis_overlap_ratio(title_bbox, candidate_bbox, axis="x") < 0.2
808
+ and abs(candidate_bbox[0] - title_bbox[0]) > 2.0 * title_height
809
+ ):
810
+ continue
811
+ candidate_line.semantic_type = "paragraph_title"
812
+ changed = True
813
+ break
814
+ if changed:
815
+ break
816
+
817
+
818
+ def _demote_hanging_multiline_text_titles(
819
+ lanes: list[_TextLane],
820
+ document_body_profile: _DocumentBodyProfile | None,
821
+ *,
822
+ page_index: int,
823
+ ) -> None:
824
+ """把缩进满行后回到栏左缘的紧邻标题行组降回正文。"""
825
+
826
+ if page_index != 0 or document_body_profile is None:
827
+ return
828
+ body_height = max(0.1, document_body_profile.body_height)
829
+ for lane in lanes:
830
+ rows = sorted(
831
+ lane.lines,
832
+ key=lambda item: (
833
+ item[1][1],
834
+ item[1][0],
835
+ item[0].source_index,
836
+ ),
837
+ )
838
+ lane_width = max(0.1, lane.right - lane.left)
839
+ for first, second in zip(rows, rows[1:]):
840
+ first_line, first_bbox = first
841
+ second_line, second_bbox = second
842
+ if (
843
+ first_line.semantic_type != "paragraph_title"
844
+ or second_line.semantic_type != "paragraph_title"
845
+ or first_line.font_signature is None
846
+ or second_line.font_signature is None
847
+ or first_line.font_coverage < 0.75
848
+ or second_line.font_coverage < 0.75
849
+ or not _font_signatures_share_family(
850
+ first_line.font_signature,
851
+ second_line.font_signature,
852
+ )
853
+ ):
854
+ continue
855
+ first_height = _line_effective_height(*first)
856
+ second_height = _line_effective_height(*second)
857
+ pair_height = max(first_height, second_height)
858
+ if not (
859
+ max(first_height, second_height) <= 1.1 * body_height
860
+ and min(first_height, second_height) >= 0.85 * pair_height
861
+ and first_bbox[2] - first_bbox[0] >= 0.9 * lane_width
862
+ and first_bbox[0] - lane.left >= 0.75 * pair_height
863
+ and abs(second_bbox[0] - lane.left) <= 0.5 * pair_height
864
+ and -0.25 * pair_height <= _effective_text_row_gap(first, second) <= 0.5 * pair_height
865
+ ):
866
+ continue
867
+ first_line.semantic_type = None
868
+ second_line.semantic_type = None
869
+ first_line.title_suppressed = True
870
+ second_line.title_suppressed = True
871
+
872
+
873
+ def _demote_visual_container_caption_titles(
874
+ line_geometry: list[tuple[_LineItem, BBox]],
875
+ container_bboxes: list[BBox],
876
+ ) -> None:
877
+ """把紧贴视觉容器下缘且水平居中的标题候选降回普通图注文本。"""
878
+
879
+ for line, bbox in line_geometry:
880
+ if line.semantic_type != "paragraph_title":
881
+ continue
882
+ line_height = _line_effective_height(line, bbox)
883
+ for container_bbox in container_bboxes:
884
+ vertical_gap = bbox[1] - container_bbox[3]
885
+ if not -0.25 * line_height <= vertical_gap <= 1.5 * line_height:
886
+ continue
887
+ if _bbox_axis_overlap_ratio(bbox, container_bbox, axis="x") < 0.35:
888
+ continue
889
+ if abs(_bbox_center_x(bbox) - _bbox_center_x(container_bbox)) > 0.35 * max(
890
+ container_bbox[2] - container_bbox[0],
891
+ bbox[2] - bbox[0],
892
+ ):
893
+ continue
894
+ line.semantic_type = None
895
+ break
896
+
897
+
898
+ def _demote_sentence_tail_titles(
899
+ line_geometry: list[tuple[_LineItem, BBox]],
900
+ ) -> None:
901
+ """把紧接未完正文、以句末标点结束的短行标题降回正文。"""
902
+
903
+ for line, bbox in line_geometry:
904
+ if (
905
+ line.semantic_type != "paragraph_title"
906
+ or re.search(
907
+ r"[.!?。!?][\]\)})】》”’'\"]*$",
908
+ line.text.rstrip(),
909
+ )
910
+ is None
911
+ ):
912
+ continue
913
+ line_height = _line_effective_height(line, bbox)
914
+ preceding = [
915
+ item
916
+ for item in line_geometry
917
+ if item[0] is not line
918
+ and item[0].semantic_type is None
919
+ and item[1][3] <= bbox[1]
920
+ and abs(item[1][0] - bbox[0]) <= 0.75 * line_height
921
+ ]
922
+ if not preceding:
923
+ continue
924
+ previous_line, previous_bbox = max(
925
+ preceding,
926
+ key=lambda item: item[1][3],
927
+ )
928
+ previous_height = _line_effective_height(
929
+ previous_line,
930
+ previous_bbox,
931
+ )
932
+ same_family = (
933
+ previous_line.font_signature is None
934
+ or line.font_signature is None
935
+ or _font_signatures_share_family(
936
+ previous_line.font_signature,
937
+ line.font_signature,
938
+ )
939
+ )
940
+ bbox_height = max(0.1, bbox[3] - bbox[1])
941
+ previous_bbox_height = max(
942
+ 0.1,
943
+ previous_bbox[3] - previous_bbox[1],
944
+ )
945
+ compatible_height = (same_family and 0.75 <= line_height / max(0.1, previous_height) <= 1.25) or (
946
+ 0.8 <= bbox_height / previous_bbox_height <= 1.2
947
+ )
948
+ if (
949
+ compatible_height
950
+ and bbox[1] - previous_bbox[3]
951
+ <= 0.9
952
+ * max(
953
+ line_height,
954
+ previous_height,
955
+ bbox_height,
956
+ previous_bbox_height,
957
+ )
958
+ and previous_bbox[2] - previous_bbox[0] >= 1.5 * (bbox[2] - bbox[0])
959
+ and re.search(
960
+ r"[.!?。!?][\]\)})】》”’'\"]*$",
961
+ previous_line.text.rstrip(),
962
+ )
963
+ is None
964
+ ):
965
+ line.semantic_type = None
966
+
967
+
968
+ def _document_title_fonts_compatible(
969
+ first: _LineItem,
970
+ second: _LineItem,
971
+ ) -> bool:
972
+ """允许混排标题因主字体覆盖不足而切换字体,同时保留可靠字重屏障。"""
973
+
974
+ if _title_fonts_compatible(first, second):
975
+ return True
976
+ uncertain_dominant_font = min(first.font_coverage, second.font_coverage) < 0.85
977
+ weights_compatible = (
978
+ first.dominant_font_weight is None
979
+ or second.dominant_font_weight is None
980
+ or abs(first.dominant_font_weight - second.dominant_font_weight) < 100.0
981
+ or max(first.dominant_font_weight, second.dominant_font_weight)
982
+ < 1.15 * min(first.dominant_font_weight, second.dominant_font_weight)
983
+ )
984
+ return uncertain_dominant_font and weights_compatible
985
+
986
+
987
+ def _document_title_uses_page_fallback(
988
+ lanes: list[_TextLane],
989
+ *,
990
+ document_body_profile: _DocumentBodyProfile | None = None,
991
+ ) -> bool:
992
+ """判断首页标题是否依赖跨栏 1.30 倍全文正文行高兜底。"""
993
+
994
+ title_heights = [
995
+ _line_effective_height(line, bbox) for lane in lanes for line, bbox in lane.lines if line.semantic_type == "doc_title"
996
+ ]
997
+ column_body_heights = [_infer_lane_body_profile(lane).body_height for lane in lanes if not lane.is_span]
998
+ if not title_heights or not column_body_heights:
999
+ return False
1000
+ document_body_height = (
1001
+ document_body_profile.body_height if document_body_profile is not None else statistics.median(column_body_heights)
1002
+ )
1003
+ return any(1.3 <= title_height / max(0.1, document_body_height) < 1.4 for title_height in title_heights)
1004
+
1005
+
1006
+ __all__ = [
1007
+ "_classify_page_titles",
1008
+ "_demote_non_structural_anomaly_titles",
1009
+ "_find_repeated_grid_title_suppressions",
1010
+ "_find_container_visual_row_title_suppressions",
1011
+ "_classify_document_title",
1012
+ "_expand_document_title_across_lanes",
1013
+ "_classify_additional_document_title_bands",
1014
+ "_classify_cross_lane_centered_section_titles",
1015
+ "_demote_cross_lane_body_continuation_titles",
1016
+ "_classify_cross_lane_emphasized_section_titles",
1017
+ "_is_wide_leading_title_continuation",
1018
+ "_expand_cross_lane_paragraph_title_neighbors",
1019
+ "_demote_hanging_multiline_text_titles",
1020
+ "_demote_visual_container_caption_titles",
1021
+ "_demote_sentence_tail_titles",
1022
+ "_document_title_fonts_compatible",
1023
+ "_document_title_uses_page_fallback",
1024
+ ]