docvortex 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (364) hide show
  1. docvortex/__init__.py +21 -0
  2. docvortex/analyzers/__init__.py +3 -0
  3. docvortex/analyzers/native/__init__.py +37 -0
  4. docvortex/analyzers/native/_shared/__init__.py +3 -0
  5. docvortex/analyzers/native/_shared/hyperlink.py +16 -0
  6. docvortex/analyzers/native/_shared/image.py +8 -0
  7. docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
  8. docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
  9. docvortex/analyzers/native/_shared/markup/formula.py +37 -0
  10. docvortex/analyzers/native/_shared/markup/projector.py +57 -0
  11. docvortex/analyzers/native/_shared/markup/styles.py +19 -0
  12. docvortex/analyzers/native/_shared/mathml.py +17 -0
  13. docvortex/analyzers/native/_shared/names.py +7 -0
  14. docvortex/analyzers/native/_shared/xycut.py +414 -0
  15. docvortex/analyzers/native/contracts.py +44 -0
  16. docvortex/analyzers/native/csv.py +351 -0
  17. docvortex/analyzers/native/epub/__init__.py +18 -0
  18. docvortex/analyzers/native/epub/constants.py +45 -0
  19. docvortex/analyzers/native/epub/converter.py +80 -0
  20. docvortex/analyzers/native/epub/errors.py +20 -0
  21. docvortex/analyzers/native/epub/metadata.py +31 -0
  22. docvortex/analyzers/native/epub/package.py +649 -0
  23. docvortex/analyzers/native/epub/xhtml.py +306 -0
  24. docvortex/analyzers/native/html/__init__.py +6 -0
  25. docvortex/analyzers/native/html/anchors.py +271 -0
  26. docvortex/analyzers/native/html/constants.py +25 -0
  27. docvortex/analyzers/native/html/contracts.py +7 -0
  28. docvortex/analyzers/native/html/converter.py +117 -0
  29. docvortex/analyzers/native/html/document.py +389 -0
  30. docvortex/analyzers/native/html/errors.py +12 -0
  31. docvortex/analyzers/native/html/resources.py +365 -0
  32. docvortex/analyzers/native/html/selector.py +421 -0
  33. docvortex/analyzers/native/models.py +209 -0
  34. docvortex/analyzers/native/ofd/__init__.py +14 -0
  35. docvortex/analyzers/native/ofd/constants.py +59 -0
  36. docvortex/analyzers/native/ofd/converter.py +31 -0
  37. docvortex/analyzers/native/ofd/errors.py +16 -0
  38. docvortex/analyzers/native/ofd/geometry.py +215 -0
  39. docvortex/analyzers/native/ofd/images.py +124 -0
  40. docvortex/analyzers/native/ofd/metadata.py +53 -0
  41. docvortex/analyzers/native/ofd/models.py +170 -0
  42. docvortex/analyzers/native/ofd/package.py +346 -0
  43. docvortex/analyzers/native/ofd/path.py +201 -0
  44. docvortex/analyzers/native/ofd/reading_order.py +293 -0
  45. docvortex/analyzers/native/ofd/resources.py +122 -0
  46. docvortex/analyzers/native/ofd/scene.py +347 -0
  47. docvortex/analyzers/native/ofd/table.py +240 -0
  48. docvortex/analyzers/native/ofd/text.py +521 -0
  49. docvortex/analyzers/native/office/__init__.py +3 -0
  50. docvortex/analyzers/native/office/doc/__init__.py +3 -0
  51. docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
  52. docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
  53. docvortex/analyzers/native/office/doc/fib.py +261 -0
  54. docvortex/analyzers/native/office/doc/fields.py +101 -0
  55. docvortex/analyzers/native/office/doc/formatting.py +171 -0
  56. docvortex/analyzers/native/office/doc/images.py +162 -0
  57. docvortex/analyzers/native/office/doc/lists.py +356 -0
  58. docvortex/analyzers/native/office/doc/models.py +164 -0
  59. docvortex/analyzers/native/office/doc/parser.py +851 -0
  60. docvortex/analyzers/native/office/doc/pieces.py +256 -0
  61. docvortex/analyzers/native/office/doc/records.py +40 -0
  62. docvortex/analyzers/native/office/doc/sprm.py +269 -0
  63. docvortex/analyzers/native/office/doc/styles.py +215 -0
  64. docvortex/analyzers/native/office/docx/__init__.py +3 -0
  65. docvortex/analyzers/native/office/docx/context.py +71 -0
  66. docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
  67. docvortex/analyzers/native/office/docx/equationxml.py +119 -0
  68. docvortex/analyzers/native/office/docx/fields.py +816 -0
  69. docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
  70. docvortex/analyzers/native/office/docx/main.py +50 -0
  71. docvortex/analyzers/native/office/docx/numbering.py +491 -0
  72. docvortex/analyzers/native/office/docx/office_xml.py +57 -0
  73. docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
  74. docvortex/analyzers/native/office/docx/resources.py +526 -0
  75. docvortex/analyzers/native/office/docx/styles.py +693 -0
  76. docvortex/analyzers/native/office/docx/tables.py +515 -0
  77. docvortex/analyzers/native/office/equation/__init__.py +3 -0
  78. docvortex/analyzers/native/office/equation/image.py +470 -0
  79. docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
  80. docvortex/analyzers/native/office/equation/mtef.py +885 -0
  81. docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
  82. docvortex/analyzers/native/office/equation/omml.py +561 -0
  83. docvortex/analyzers/native/office/equation/ooxml.py +62 -0
  84. docvortex/analyzers/native/office/errors.py +33 -0
  85. docvortex/analyzers/native/office/image.py +307 -0
  86. docvortex/analyzers/native/office/legacy/__init__.py +3 -0
  87. docvortex/analyzers/native/office/legacy/binary.py +43 -0
  88. docvortex/analyzers/native/office/legacy/officeart.py +362 -0
  89. docvortex/analyzers/native/office/legacy/ole.py +110 -0
  90. docvortex/analyzers/native/office/limits.py +12 -0
  91. docvortex/analyzers/native/office/odf/__init__.py +3 -0
  92. docvortex/analyzers/native/office/odf/chart.py +76 -0
  93. docvortex/analyzers/native/office/odf/constants.py +70 -0
  94. docvortex/analyzers/native/office/odf/converters.py +403 -0
  95. docvortex/analyzers/native/office/odf/errors.py +18 -0
  96. docvortex/analyzers/native/office/odf/metadata.py +91 -0
  97. docvortex/analyzers/native/office/odf/models.py +176 -0
  98. docvortex/analyzers/native/office/odf/package.py +270 -0
  99. docvortex/analyzers/native/office/odf/styles.py +329 -0
  100. docvortex/analyzers/native/office/odf/table.py +469 -0
  101. docvortex/analyzers/native/office/odf/text.py +1002 -0
  102. docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
  103. docvortex/analyzers/native/office/opc.py +38 -0
  104. docvortex/analyzers/native/office/ppt/__init__.py +3 -0
  105. docvortex/analyzers/native/office/ppt/models.py +118 -0
  106. docvortex/analyzers/native/office/ppt/parser.py +1895 -0
  107. docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
  108. docvortex/analyzers/native/office/ppt/records.py +131 -0
  109. docvortex/analyzers/native/office/ppt/style_text.py +247 -0
  110. docvortex/analyzers/native/office/pptx/__init__.py +3 -0
  111. docvortex/analyzers/native/office/pptx/context.py +113 -0
  112. docvortex/analyzers/native/office/pptx/lists.py +558 -0
  113. docvortex/analyzers/native/office/pptx/main.py +20 -0
  114. docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
  115. docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
  116. docvortex/analyzers/native/office/pptx/resources.py +329 -0
  117. docvortex/analyzers/native/office/pptx/shapes.py +393 -0
  118. docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
  119. docvortex/analyzers/native/office/pptx/titles.py +178 -0
  120. docvortex/analyzers/native/office/rich_text.py +420 -0
  121. docvortex/analyzers/native/office/rtf/__init__.py +3 -0
  122. docvortex/analyzers/native/office/rtf/converter.py +708 -0
  123. docvortex/analyzers/native/office/rtf/lexer.py +213 -0
  124. docvortex/analyzers/native/office/rtf/math.py +339 -0
  125. docvortex/analyzers/native/office/rtf/models.py +185 -0
  126. docvortex/analyzers/native/office/rtf/parser.py +1553 -0
  127. docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
  128. docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
  129. docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
  130. docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
  131. docvortex/analyzers/native/office/streams.py +18 -0
  132. docvortex/analyzers/native/office/xls/__init__.py +3 -0
  133. docvortex/analyzers/native/office/xls/chart.py +132 -0
  134. docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
  135. docvortex/analyzers/native/office/xls/models.py +109 -0
  136. docvortex/analyzers/native/office/xls/number_format.py +521 -0
  137. docvortex/analyzers/native/office/xls/parser.py +1145 -0
  138. docvortex/analyzers/native/office/xls/records.py +201 -0
  139. docvortex/analyzers/native/office/xls/strings.py +205 -0
  140. docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
  141. docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
  142. docvortex/analyzers/native/office/xlsx/main.py +20 -0
  143. docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
  144. docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
  145. docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
  146. docvortex/analyzers/native/pdf/__init__.py +3 -0
  147. docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
  148. docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
  149. docvortex/analyzers/native/pdf/code_blocks.py +535 -0
  150. docvortex/analyzers/native/pdf/formulas.py +1985 -0
  151. docvortex/analyzers/native/pdf/geometry.py +281 -0
  152. docvortex/analyzers/native/pdf/graphics.py +1501 -0
  153. docvortex/analyzers/native/pdf/index_blocks.py +268 -0
  154. docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
  155. docvortex/analyzers/native/pdf/inline/common.py +112 -0
  156. docvortex/analyzers/native/pdf/inline/detection.py +590 -0
  157. docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
  158. docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
  159. docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
  160. docvortex/analyzers/native/pdf/inline/types.py +385 -0
  161. docvortex/analyzers/native/pdf/line_layout.py +1106 -0
  162. docvortex/analyzers/native/pdf/line_merging.py +1223 -0
  163. docvortex/analyzers/native/pdf/models.py +246 -0
  164. docvortex/analyzers/native/pdf/native_text.py +1004 -0
  165. docvortex/analyzers/native/pdf/pipeline.py +1390 -0
  166. docvortex/analyzers/native/pdf/script_geometry.py +636 -0
  167. docvortex/analyzers/native/pdf/shared.py +30 -0
  168. docvortex/analyzers/native/pdf/spatial_text.py +383 -0
  169. docvortex/analyzers/native/pdf/table_annotations.py +446 -0
  170. docvortex/analyzers/native/pdf/table_constants.py +48 -0
  171. docvortex/analyzers/native/pdf/table_detection.py +227 -0
  172. docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
  173. docvortex/analyzers/native/pdf/table_geometry.py +40 -0
  174. docvortex/analyzers/native/pdf/table_materialization.py +444 -0
  175. docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
  176. docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
  177. docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
  178. docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
  179. docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
  180. docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
  181. docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
  182. docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
  183. docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
  184. docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
  185. docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
  186. docvortex/analyzers/native/pdf/table_rows.py +34 -0
  187. docvortex/analyzers/native/pdf/table_rules.py +1129 -0
  188. docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
  189. docvortex/analyzers/native/pdf/tables.py +147 -0
  190. docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
  191. docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
  192. docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
  193. docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
  194. docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
  195. docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
  196. docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
  197. docvortex/analyzers/native/pdf/text_blocks.py +82 -0
  198. docvortex/analyzers/native/pdf/text_styles.py +55 -0
  199. docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
  200. docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
  201. docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
  202. docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
  203. docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
  204. docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
  205. docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
  206. docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
  207. docvortex/analyzers/native/pdf/titles.py +75 -0
  208. docvortex/analyzers/native/pdf/typography.py +19 -0
  209. docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
  210. docvortex/api.py +180 -0
  211. docvortex/assets/__init__.py +5 -0
  212. docvortex/assets/store.py +51 -0
  213. docvortex/cli.py +54 -0
  214. docvortex/codecs/__init__.py +3 -0
  215. docvortex/codecs/html/__init__.py +22 -0
  216. docvortex/codecs/html/contracts.py +236 -0
  217. docvortex/codecs/html/materializer.py +331 -0
  218. docvortex/codecs/html/parser.py +763 -0
  219. docvortex/codecs/html/resources.py +34 -0
  220. docvortex/codecs/json.py +17 -0
  221. docvortex/content/__init__.py +5 -0
  222. docvortex/content/inline.py +248 -0
  223. docvortex/content/markup/__init__.py +44 -0
  224. docvortex/content/markup/anchors.py +188 -0
  225. docvortex/content/markup/formula.py +280 -0
  226. docvortex/content/markup/projector.py +1237 -0
  227. docvortex/content/markup/styles.py +327 -0
  228. docvortex/content/mathml.py +167 -0
  229. docvortex/content/normalization.py +188 -0
  230. docvortex/content/spans.py +183 -0
  231. docvortex/content/table/__init__.py +18 -0
  232. docvortex/content/table/blocks.py +152 -0
  233. docvortex/content/table/content.py +425 -0
  234. docvortex/content/table/document.py +104 -0
  235. docvortex/content/table/html.py +399 -0
  236. docvortex/content/table/models.py +76 -0
  237. docvortex/content/table/rules.py +42 -0
  238. docvortex/content/table/structure.py +221 -0
  239. docvortex/content/tree.py +5 -0
  240. docvortex/document/__init__.py +3 -0
  241. docvortex/document/contracts.py +22 -0
  242. docvortex/document/detection.py +389 -0
  243. docvortex/document/filetypes.py +175 -0
  244. docvortex/document/page_range.py +167 -0
  245. docvortex/document/pdf/__init__.py +19 -0
  246. docvortex/document/pdf/classify.py +1138 -0
  247. docvortex/document/pdf/constants.py +132 -0
  248. docvortex/document/pdf/diagnostics.py +350 -0
  249. docvortex/document/pdf/document.py +582 -0
  250. docvortex/document/pdf/font_runtime.py +335 -0
  251. docvortex/document/pdf/geometry.py +31 -0
  252. docvortex/document/pdf/images.py +654 -0
  253. docvortex/document/pdf/native_annotations.py +367 -0
  254. docvortex/document/pdf/native_contracts.py +169 -0
  255. docvortex/document/pdf/native_coordinates.py +216 -0
  256. docvortex/document/pdf/native_lifecycle.py +16 -0
  257. docvortex/document/pdf/native_objects.py +902 -0
  258. docvortex/document/pdf/native_text_geometry.py +315 -0
  259. docvortex/document/pdf/pdfium.py +325 -0
  260. docvortex/document/pdf/raster.py +46 -0
  261. docvortex/document/pdf/text/__init__.py +62 -0
  262. docvortex/document/pdf/text/contracts.py +211 -0
  263. docvortex/document/pdf/text/extract.py +165 -0
  264. docvortex/document/pdf/text/geometry.py +16 -0
  265. docvortex/document/pdf/text/groups.py +162 -0
  266. docvortex/document/pdf/visual_geometry.py +201 -0
  267. docvortex/document/pdf/visuals.py +343 -0
  268. docvortex/document/source.py +92 -0
  269. docvortex/errors.py +21 -0
  270. docvortex/export/__init__.py +3 -0
  271. docvortex/export/bundle.py +98 -0
  272. docvortex/export/files.py +66 -0
  273. docvortex/export/middle.py +208 -0
  274. docvortex/foundation/__init__.py +3 -0
  275. docvortex/foundation/geometry.py +125 -0
  276. docvortex/foundation/hyperlink.py +65 -0
  277. docvortex/foundation/image.py +48 -0
  278. docvortex/foundation/image_encoding.py +30 -0
  279. docvortex/foundation/image_payload.py +280 -0
  280. docvortex/foundation/language.py +92 -0
  281. docvortex/foundation/platform.py +38 -0
  282. docvortex/foundation/text.py +153 -0
  283. docvortex/foundation/type_identity.py +20 -0
  284. docvortex/foundation/xml_names.py +20 -0
  285. docvortex/options.py +30 -0
  286. docvortex/postprocess/__init__.py +3 -0
  287. docvortex/postprocess/content.py +53 -0
  288. docvortex/postprocess/document.py +19 -0
  289. docvortex/postprocess/lists.py +236 -0
  290. docvortex/postprocess/page_blocks.py +214 -0
  291. docvortex/postprocess/pages.py +95 -0
  292. docvortex/postprocess/paragraphs.py +580 -0
  293. docvortex/postprocess/visual.py +715 -0
  294. docvortex/render/__init__.py +48 -0
  295. docvortex/render/_internal/__init__.py +3 -0
  296. docvortex/render/_internal/common/__init__.py +3 -0
  297. docvortex/render/_internal/common/context.py +43 -0
  298. docvortex/render/_internal/common/html_table.py +178 -0
  299. docvortex/render/_internal/common/index.py +33 -0
  300. docvortex/render/_internal/common/list_items.py +158 -0
  301. docvortex/render/_internal/common/planner.py +140 -0
  302. docvortex/render/_internal/docx/__init__.py +3 -0
  303. docvortex/render/_internal/docx/assets.py +202 -0
  304. docvortex/render/_internal/docx/inline.py +434 -0
  305. docvortex/render/_internal/docx/math.py +220 -0
  306. docvortex/render/_internal/docx/renderer.py +905 -0
  307. docvortex/render/_internal/docx/styles.py +195 -0
  308. docvortex/render/_internal/docx/table.py +442 -0
  309. docvortex/render/_internal/epub/__init__.py +5 -0
  310. docvortex/render/_internal/epub/assets.py +173 -0
  311. docvortex/render/_internal/epub/package.py +249 -0
  312. docvortex/render/_internal/epub/renderer.py +1156 -0
  313. docvortex/render/_internal/html/__init__.py +3 -0
  314. docvortex/render/_internal/html/inline.py +346 -0
  315. docvortex/render/_internal/html/renderer.py +1041 -0
  316. docvortex/render/_internal/html/sanitizer.py +478 -0
  317. docvortex/render/_internal/html/table.py +121 -0
  318. docvortex/render/_internal/latex/__init__.py +1 -0
  319. docvortex/render/_internal/latex/assets.py +85 -0
  320. docvortex/render/_internal/latex/inline.py +145 -0
  321. docvortex/render/_internal/latex/renderer.py +506 -0
  322. docvortex/render/_internal/latex/table.py +347 -0
  323. docvortex/render/_internal/markdown/__init__.py +3 -0
  324. docvortex/render/_internal/markdown/assets.py +78 -0
  325. docvortex/render/_internal/markdown/blocks.py +635 -0
  326. docvortex/render/_internal/markdown/escaping.py +51 -0
  327. docvortex/render/_internal/markdown/inline.py +260 -0
  328. docvortex/render/_internal/markdown/renderer.py +93 -0
  329. docvortex/render/_internal/markdown/table.py +281 -0
  330. docvortex/render/_internal/pdf/__init__.py +3 -0
  331. docvortex/render/_internal/pdf/assets.py +197 -0
  332. docvortex/render/_internal/pdf/formula.py +417 -0
  333. docvortex/render/_internal/pdf/inline.py +343 -0
  334. docvortex/render/_internal/pdf/renderer.py +734 -0
  335. docvortex/render/_internal/pdf/styles.py +206 -0
  336. docvortex/render/_internal/pdf/table.py +272 -0
  337. docvortex/render/_internal/structured_content/__init__.py +3 -0
  338. docvortex/render/_internal/structured_content/renderer.py +193 -0
  339. docvortex/render/api.py +199 -0
  340. docvortex/render/contracts.py +205 -0
  341. docvortex/render/docx.py +38 -0
  342. docvortex/render/epub.py +35 -0
  343. docvortex/render/fragments.py +70 -0
  344. docvortex/render/html.py +29 -0
  345. docvortex/render/latex.py +24 -0
  346. docvortex/render/markdown.py +50 -0
  347. docvortex/render/pdf.py +25 -0
  348. docvortex/render/structured_content.py +23 -0
  349. docvortex/resources/epub/docvortex.css +91 -0
  350. docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
  351. docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
  352. docvortex/resources/fonts/NOTICE +190 -0
  353. docvortex/resources/fonts/manifest.json +9 -0
  354. docvortex/resources/html/docvortex.css +601 -0
  355. docvortex/resources/html/docvortex.min.css +1 -0
  356. docvortex/result.py +91 -0
  357. docvortex/schema.py +1141 -0
  358. docvortex/version.py +3 -0
  359. docvortex-0.2.1.dist-info/METADATA +193 -0
  360. docvortex-0.2.1.dist-info/RECORD +364 -0
  361. docvortex-0.2.1.dist-info/WHEEL +5 -0
  362. docvortex-0.2.1.dist-info/entry_points.txt +2 -0
  363. docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
  364. docvortex-0.2.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,221 @@
1
+ """跨页表格的表头、宽度和边界行结构判定。"""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Any
6
+
7
+ from bs4 import Tag
8
+
9
+ from ...schema import BlockType
10
+
11
+ from .blocks import _bbox_for_calculation, _is_continuation_caption, _is_post_table_non_continuation_caption, _table_children
12
+ from .html import _colspan, _rowspan, calculate_row_rendered_segments
13
+ from .models import MAX_HEADER_ROWS, TableMergeState
14
+
15
+
16
+ def detect_table_headers(
17
+ state1: TableMergeState, state2: TableMergeState, max_header_rows: int = MAX_HEADER_ROWS
18
+ ) -> tuple[int, bool, list[list[str]]]:
19
+ """检测并比较两个表格的表头,仅扫描前几行."""
20
+ front_rows1 = state1.front_header_info[:max_header_rows]
21
+ front_rows2 = state2.front_header_info[:max_header_rows]
22
+
23
+ min_rows = min(len(front_rows1), len(front_rows2), max_header_rows)
24
+ header_rows = 0
25
+ headers_match = True
26
+ header_texts = []
27
+
28
+ for row_idx in range(min_rows):
29
+ row1 = front_rows1[row_idx]
30
+ row2 = front_rows2[row_idx]
31
+ structure_match = (
32
+ row1.cell_count == row2.cell_count
33
+ and row1.effective_cols == row2.effective_cols
34
+ and row1.colspans == row2.colspans
35
+ and row1.rowspans == row2.rowspans
36
+ and row1.normalized_texts == row2.normalized_texts
37
+ )
38
+
39
+ if structure_match:
40
+ header_rows += 1
41
+ header_texts.append(list(row1.display_texts))
42
+ else:
43
+ headers_match = header_rows > 0
44
+ break
45
+
46
+ if header_rows == 0:
47
+ header_rows, headers_match, header_texts = _detect_table_headers_visual(state1, state2, max_header_rows=max_header_rows)
48
+
49
+ return header_rows, headers_match, header_texts
50
+
51
+
52
+ def _detect_table_headers_visual(
53
+ state1: TableMergeState,
54
+ state2: TableMergeState,
55
+ max_header_rows: int = MAX_HEADER_ROWS,
56
+ ) -> tuple[int, bool, list[list[str]]]:
57
+ """基于视觉一致性检测表头(只比较文本内容,忽略colspan/rowspan差异)."""
58
+ front_rows1 = state1.front_header_info[:max_header_rows]
59
+ front_rows2 = state2.front_header_info[:max_header_rows]
60
+
61
+ min_rows = min(len(front_rows1), len(front_rows2), max_header_rows)
62
+ header_rows = 0
63
+ headers_match = True
64
+ header_texts = []
65
+
66
+ for row_idx in range(min_rows):
67
+ row1 = front_rows1[row_idx]
68
+ row2 = front_rows2[row_idx]
69
+ # OCR 识别表头时可能丢失 colspan/rowspan,这里用渲染段数约束视觉一致性。
70
+ rendered_segments1 = calculate_row_rendered_segments(state1.rows, row_idx)
71
+ rendered_segments2 = calculate_row_rendered_segments(state2.rows, row_idx)
72
+ if row1.normalized_texts == row2.normalized_texts and rendered_segments1 == rendered_segments2:
73
+ header_rows += 1
74
+ header_texts.append(list(row1.display_texts))
75
+ else:
76
+ headers_match = header_rows > 0
77
+ break
78
+
79
+ if header_rows == 0:
80
+ headers_match = False
81
+
82
+ return header_rows, headers_match, header_texts
83
+
84
+
85
+ def _expand_header_count_by_rowspan(rows: list[Tag], header_count: int) -> int:
86
+ """按表头 rowspan 覆盖范围扩展跳过行数。
87
+
88
+ 跨页续表的第一行表头可能包含 rowspan。如果只跳过已匹配的首行,
89
+ 被该 rowspan 覆盖的后续表头行会失去占位来源,合并后形成半截表头。
90
+ 因此跳过重复表头时,需要覆盖所有由已跳过表头行跨行占据的行。
91
+ """
92
+ if header_count <= 0 or not rows:
93
+ return header_count
94
+
95
+ expanded_header_count = min(header_count, len(rows))
96
+ row_idx = 0
97
+ while row_idx < expanded_header_count:
98
+ row = rows[row_idx]
99
+ for cell in row.find_all(["td", "th"]):
100
+ rowspan = _rowspan(cell)
101
+ if rowspan > 1:
102
+ expanded_header_count = max(expanded_header_count, row_idx + rowspan)
103
+ expanded_header_count = min(expanded_header_count, len(rows))
104
+ row_idx += 1
105
+
106
+ return expanded_header_count
107
+
108
+
109
+ def can_merge_by_structure(
110
+ current_state: TableMergeState,
111
+ previous_state: TableMergeState,
112
+ current_bbox: Any = None,
113
+ previous_bbox: Any = None,
114
+ ) -> bool:
115
+ """仅基于表格结构判断是否可合并(不检查 caption/footnote)。
116
+
117
+ 供外部工具调用,忽略 caption 和 footnote 检查。
118
+ """
119
+ if (
120
+ current_bbox is not None
121
+ and previous_bbox is not None
122
+ and not _table_widths_are_compatible(
123
+ current_bbox,
124
+ previous_bbox,
125
+ )
126
+ ):
127
+ return False
128
+
129
+ if (
130
+ previous_state.total_cols <= 0
131
+ or current_state.total_cols <= 0
132
+ or previous_state.last_data_row_metrics is None
133
+ or current_state.last_data_row_metrics is None
134
+ ):
135
+ return False
136
+
137
+ if previous_state.total_cols == current_state.total_cols:
138
+ return True
139
+
140
+ return check_rows_match(previous_state, current_state)
141
+
142
+
143
+ def _table_widths_are_compatible(current_bbox: Any, previous_bbox: Any) -> bool:
144
+ """使用千分位 bbox 判断两张表的宽度相对差是否小于百分之十。"""
145
+ current_calc_bbox = _bbox_for_calculation(current_bbox)
146
+ previous_calc_bbox = _bbox_for_calculation(previous_bbox)
147
+ if current_calc_bbox is None or previous_calc_bbox is None:
148
+ return False
149
+
150
+ current_width = current_calc_bbox[2] - current_calc_bbox[0]
151
+ previous_width = previous_calc_bbox[2] - previous_calc_bbox[0]
152
+ min_width = min(current_width, previous_width)
153
+ return min_width > 0 and abs(current_width - previous_width) / min_width < 0.1
154
+
155
+
156
+ def can_merge_tables(current_state: TableMergeState, previous_state: TableMergeState) -> bool:
157
+ """根据 dict 表格的辅助文本、宽度和 HTML 结构判断是否可合并。"""
158
+ current_table_block = current_state.owner_block
159
+ previous_table_block = previous_state.owner_block
160
+
161
+ if not isinstance(previous_table_block, dict) or not isinstance(current_table_block, dict):
162
+ return False
163
+
164
+ previous_children = _table_children(previous_table_block)
165
+ current_children = _table_children(current_table_block)
166
+ footnote_count = sum(1 for block in previous_children if block.get("type") == BlockType.TABLE_FOOTNOTE)
167
+ caption_blocks = [block for block in current_children if block.get("type") == BlockType.TABLE_CAPTION]
168
+ merge_caption_blocks = [
169
+ block for block in caption_blocks if not _is_post_table_non_continuation_caption(current_table_block, block)
170
+ ]
171
+ if merge_caption_blocks:
172
+ has_continuation_marker = any(_is_continuation_caption(block) for block in merge_caption_blocks)
173
+
174
+ if not has_continuation_marker:
175
+ return False
176
+
177
+ if footnote_count > 1:
178
+ return False
179
+ elif footnote_count > 0:
180
+ return False
181
+
182
+ if not _table_widths_are_compatible(current_table_block.get("bbox"), previous_table_block.get("bbox")):
183
+ return False
184
+
185
+ return can_merge_by_structure(current_state, previous_state)
186
+
187
+
188
+ def check_rows_match(previous_state: TableMergeState, current_state: TableMergeState) -> bool:
189
+ """检查表格边界行是否匹配."""
190
+ last_row_metrics = previous_state.last_data_row_metrics
191
+ if last_row_metrics is None:
192
+ return False
193
+
194
+ header_count, _, _ = detect_table_headers(previous_state, current_state)
195
+ header_count = _expand_header_count_by_rowspan(current_state.rows, header_count)
196
+ first_data_row_metrics = current_state.front_first_data_row_metrics.get(header_count)
197
+ if first_data_row_metrics is None:
198
+ return False
199
+
200
+ previous_rendered_segments = calculate_row_rendered_segments(previous_state.rows, last_row_metrics.row_idx)
201
+ current_rendered_segments = calculate_row_rendered_segments(current_state.rows, first_data_row_metrics.row_idx)
202
+
203
+ return (
204
+ last_row_metrics.effective_cols == first_data_row_metrics.effective_cols
205
+ or last_row_metrics.actual_cols == first_data_row_metrics.actual_cols
206
+ or previous_rendered_segments == current_rendered_segments
207
+ )
208
+
209
+
210
+ def check_row_columns_match(row1: Tag, row2: Tag) -> bool:
211
+ """判断两行显式单元格数量与 colspan 结构是否一致。"""
212
+ cells1 = row1.find_all(["td", "th"])
213
+ cells2 = row2.find_all(["td", "th"])
214
+ if len(cells1) != len(cells2):
215
+ return False
216
+ for cell1, cell2 in zip(cells1, cells2):
217
+ colspan1 = _colspan(cell1)
218
+ colspan2 = _colspan(cell2)
219
+ if colspan1 != colspan2:
220
+ return False
221
+ return True
@@ -0,0 +1,5 @@
1
+ """严格文档树的公共遍历能力。"""
2
+
3
+ from ..schema import _iter_child_blocks as iter_child_blocks
4
+
5
+ __all__ = ["iter_child_blocks"]
@@ -0,0 +1,3 @@
1
+ """DocVortex 模块边界。"""
2
+
3
+ __all__ = []
@@ -0,0 +1,22 @@
1
+ """HTML Flash 解析使用的来源上下文契约。"""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import dataclass
6
+ from pathlib import Path
7
+ from docvortex.foundation.type_identity import preserve_type_module
8
+
9
+
10
+ @dataclass(frozen=True, slots=True)
11
+ class HtmlSourceContext:
12
+ """保存相对链接解析及 HTML 解码所需的来源上下文。"""
13
+
14
+ source_uri: str | None = None
15
+ local_resource_root: Path | None = None
16
+ transport_encoding: str | None = None
17
+
18
+
19
+ __all__ = ["HtmlSourceContext"]
20
+
21
+ # 保持既有公开类型的 pickle 路径,所有旧、新入口指向同一个类。
22
+ preserve_type_module(HtmlSourceContext, "docvortex.analyzers.native.html.contracts")
@@ -0,0 +1,389 @@
1
+ """根据文件内容和容器结构识别 DocVortex 支持的输入后缀。"""
2
+
3
+ from __future__ import annotations
4
+
5
+ from io import BytesIO
6
+ from functools import lru_cache
7
+ from pathlib import Path
8
+ from xml.etree import ElementTree
9
+ from zipfile import BadZipFile, ZipFile
10
+
11
+ from loguru import logger
12
+ from typing import TYPE_CHECKING
13
+
14
+ if TYPE_CHECKING:
15
+ from magika import Magika
16
+
17
+ from .filetypes import CSV_EXTENSIONS, HTML_EXTENSIONS, IMAGE_EXTENSIONS, rtf_header_offset
18
+
19
+ PDF_SIG_BYTES = b"%PDF"
20
+ OLE2_SIG_BYTES = b"\xd0\xcf\x11\xe0\xa1\xb1\x1a\xe1"
21
+ OOXML_ROOT_RELS = "_rels/.rels"
22
+ OOXML_CONTENT_TYPES = "[Content_Types].xml"
23
+ OOXML_PACKAGE_REL_NS = "http://schemas.openxmlformats.org/package/2006/relationships"
24
+ OOXML_CONTENT_TYPES_NS = "http://schemas.openxmlformats.org/package/2006/content-types"
25
+ OOXML_OFFICE_DOCUMENT_REL = "http://schemas.openxmlformats.org/officeDocument/2006/relationships/officeDocument"
26
+ OOXML_MAIN_CONTENT_TYPES = {
27
+ ("application/vnd.openxmlformats-officedocument.wordprocessingml.document.main+xml"): "docx",
28
+ ("application/vnd.openxmlformats-officedocument.presentationml.presentation.main+xml"): "pptx",
29
+ ("application/vnd.openxmlformats-officedocument.spreadsheetml.sheet.main+xml"): "xlsx",
30
+ }
31
+ ODF_MIMETYPE_SUFFIXES = {
32
+ "application/vnd.oasis.opendocument.text": "odt",
33
+ "application/vnd.oasis.opendocument.spreadsheet": "ods",
34
+ "application/vnd.oasis.opendocument.presentation": "odp",
35
+ }
36
+ ODF_MANIFEST_PATH = "META-INF/manifest.xml"
37
+ ODF_MANIFEST_NS = "urn:oasis:names:tc:opendocument:xmlns:manifest:1.0"
38
+ # OLE2 compound file 内部 stream 名 → 旧 Office 格式后缀
39
+ # doc: WordDocument stream;xls: Workbook 或 Book stream;ppt: PowerPoint Document stream
40
+ OLE2_STREAM_SUFFIX_MAP: dict[str, str] = {
41
+ "WordDocument": "doc",
42
+ "Workbook": "xls",
43
+ "Book": "xls",
44
+ "PowerPoint Document": "ppt",
45
+ }
46
+ _STRONG_CONTENT_SUFFIXES = frozenset(
47
+ {
48
+ "pdf",
49
+ "doc",
50
+ "docx",
51
+ "ppt",
52
+ "pptx",
53
+ "xls",
54
+ "xlsx",
55
+ "rtf",
56
+ "epub",
57
+ "ofd",
58
+ "odt",
59
+ "ods",
60
+ "odp",
61
+ *IMAGE_EXTENSIONS,
62
+ }
63
+ )
64
+
65
+
66
+ @lru_cache(maxsize=1)
67
+ def _magika() -> Magika:
68
+ """惰性创建文件类型识别器,避免导入 parser 时加载模型。"""
69
+ from magika import Magika
70
+
71
+ return Magika()
72
+
73
+
74
+ def _strip_package_part_name(part_name: str | None) -> str:
75
+ """规范化 OPC part 路径,方便匹配 Content_Types 中的 PartName。"""
76
+ if not part_name:
77
+ return ""
78
+ return part_name.replace("\\", "/").lstrip("/")
79
+
80
+
81
+ def _ooxml_relationship_targets(root: ElementTree.Element) -> list[str]:
82
+ """从根关系文件中提取 Office 主文档关系目标。"""
83
+ targets = []
84
+ for relationship in root:
85
+ if relationship.tag not in {
86
+ f"{{{OOXML_PACKAGE_REL_NS}}}Relationship",
87
+ "Relationship",
88
+ }:
89
+ continue
90
+ if relationship.get("TargetMode") == "External":
91
+ continue
92
+ if relationship.get("Type") != OOXML_OFFICE_DOCUMENT_REL:
93
+ continue
94
+ target = _strip_package_part_name(relationship.get("Target"))
95
+ if target:
96
+ targets.append(target)
97
+ return targets
98
+
99
+
100
+ def _ooxml_content_type_overrides(root: ElementTree.Element) -> dict[str, str]:
101
+ """读取 Content_Types 中每个显式 part 的 ContentType 映射。"""
102
+ overrides = {}
103
+ for override in root:
104
+ if override.tag not in {
105
+ f"{{{OOXML_CONTENT_TYPES_NS}}}Override",
106
+ "Override",
107
+ }:
108
+ continue
109
+ part_name = _strip_package_part_name(override.get("PartName"))
110
+ content_type = override.get("ContentType")
111
+ if part_name and content_type:
112
+ overrides[part_name] = content_type
113
+ return overrides
114
+
115
+
116
+ def _guess_ooxml_suffix_from_zip(package: ZipFile) -> str | None:
117
+ """根据 OOXML 包内标准主文档关系和主内容类型判断 Office 子类型。"""
118
+ rels_root = ElementTree.fromstring(package.read(OOXML_ROOT_RELS))
119
+ content_types_root = ElementTree.fromstring(package.read(OOXML_CONTENT_TYPES))
120
+
121
+ overrides = _ooxml_content_type_overrides(content_types_root)
122
+ for target in _ooxml_relationship_targets(rels_root):
123
+ suffix = OOXML_MAIN_CONTENT_TYPES.get(overrides.get(target, ""))
124
+ if suffix:
125
+ return suffix
126
+ return None
127
+
128
+
129
+ def _guess_ooxml_suffix_by_bytes(file_bytes: bytes) -> str | None:
130
+ """优先用 OOXML 包结构识别 docx/pptx/xlsx,避免 Magika 被内嵌对象误导。"""
131
+ try:
132
+ with ZipFile(BytesIO(file_bytes)) as package:
133
+ return _guess_ooxml_suffix_from_zip(package)
134
+ except (
135
+ BadZipFile,
136
+ KeyError,
137
+ ElementTree.ParseError,
138
+ RuntimeError,
139
+ OSError,
140
+ ValueError,
141
+ ):
142
+ return None
143
+
144
+
145
+ def _guess_ooxml_suffix_by_path(file_path: Path) -> str | None:
146
+ """从文件路径读取 OOXML 包结构;失败时交给 Magika 原有逻辑兜底。"""
147
+ try:
148
+ with ZipFile(file_path) as package:
149
+ return _guess_ooxml_suffix_from_zip(package)
150
+ except (
151
+ BadZipFile,
152
+ KeyError,
153
+ ElementTree.ParseError,
154
+ RuntimeError,
155
+ OSError,
156
+ ValueError,
157
+ ):
158
+ return None
159
+
160
+
161
+ def _guess_odf_suffix_from_zip(package: ZipFile) -> str | None:
162
+ """按 ODF mimetype、manifest 根条目依次识别 odt/ods/odp。"""
163
+ try:
164
+ mimetype_info = package.getinfo("mimetype")
165
+ if mimetype_info.file_size <= 256:
166
+ mimetype = package.read(mimetype_info).decode("ascii", errors="strict").strip()
167
+ if suffix := ODF_MIMETYPE_SUFFIXES.get(mimetype):
168
+ return suffix
169
+ except (KeyError, UnicodeDecodeError, RuntimeError, OSError, ValueError):
170
+ pass
171
+ try:
172
+ manifest_info = package.getinfo(ODF_MANIFEST_PATH)
173
+ if manifest_info.file_size > 1024 * 1024:
174
+ return None
175
+ root = ElementTree.fromstring(package.read(manifest_info))
176
+ except (KeyError, ElementTree.ParseError, RuntimeError, OSError, ValueError):
177
+ return None
178
+ for entry in root.iter(f"{{{ODF_MANIFEST_NS}}}file-entry"):
179
+ if entry.get(f"{{{ODF_MANIFEST_NS}}}full-path") != "/":
180
+ continue
181
+ media_type = entry.get(f"{{{ODF_MANIFEST_NS}}}media-type", "").strip()
182
+ return ODF_MIMETYPE_SUFFIXES.get(media_type)
183
+ return None
184
+
185
+
186
+ def _guess_odf_suffix_by_bytes(file_bytes: bytes) -> str | None:
187
+ """从内存 ZIP 包识别 ODF,失败时不影响后续 OLE/Magika/CSV 路由。"""
188
+ try:
189
+ with ZipFile(BytesIO(file_bytes)) as package:
190
+ return _guess_odf_suffix_from_zip(package)
191
+ except (BadZipFile, RuntimeError, OSError, ValueError):
192
+ return None
193
+
194
+
195
+ def _guess_odf_suffix_by_path(file_path: Path) -> str | None:
196
+ """从路径 ZIP 包识别 ODF,保持现有 OOXML 检测优先级。"""
197
+ try:
198
+ with ZipFile(file_path) as package:
199
+ return _guess_odf_suffix_from_zip(package)
200
+ except (BadZipFile, RuntimeError, OSError, ValueError):
201
+ return None
202
+
203
+
204
+ def _guess_epub_suffix_by_bytes(file_bytes: bytes) -> str | None:
205
+ """从内存 ZIP 包验证 EPUB 强内容身份。"""
206
+ from ..analyzers.native.epub import detect_epub
207
+
208
+ return "epub" if detect_epub(file_bytes) else None
209
+
210
+
211
+ def _guess_epub_suffix_by_path(file_path: Path) -> str | None:
212
+ """从路径 ZIP 包验证 EPUB 强内容身份。"""
213
+ from ..analyzers.native.epub import detect_epub_path
214
+
215
+ return "epub" if detect_epub_path(file_path) else None
216
+
217
+
218
+ def _guess_ofd_suffix_by_bytes(file_bytes: bytes) -> str | None:
219
+ """从内存 ZIP 包验证 OFD 强内容身份。"""
220
+ from ..analyzers.native.ofd import detect_ofd
221
+
222
+ return "ofd" if detect_ofd(file_bytes) else None
223
+
224
+
225
+ def _guess_ofd_suffix_by_path(file_path: Path) -> str | None:
226
+ """从路径 ZIP 包验证 OFD 强内容身份。"""
227
+ from ..analyzers.native.ofd import detect_ofd_path
228
+
229
+ return "ofd" if detect_ofd_path(file_path) else None
230
+
231
+
232
+ def _guess_ole2_suffix_by_bytes(file_bytes: bytes) -> str | None:
233
+ """用 OLE2 magic + olefile 内部 stream 区分 doc/xls/ppt。
234
+
235
+ olefile 是纯 Python 库且已是核心依赖(mineru.model.flash.office.legacy 使用)。
236
+ 在 OOXML 识别失败后、Magika 兜底前插入此层,避免 Magika 对 OLE2 返回 unknown。
237
+ """
238
+ if len(file_bytes) < 8 or file_bytes[:8] != OLE2_SIG_BYTES:
239
+ return None
240
+ try:
241
+ import olefile # type: ignore[import-untyped]
242
+
243
+ with olefile.OleFileIO(BytesIO(file_bytes)) as ole:
244
+ for stream_name in ole.listdir(streams=True):
245
+ name = "/".join(stream_name)
246
+ suffix = OLE2_STREAM_SUFFIX_MAP.get(name)
247
+ if suffix:
248
+ return suffix
249
+ except Exception:
250
+ return None
251
+ return None
252
+
253
+
254
+ def _guess_ole2_suffix_by_path(file_path: Path) -> str | None:
255
+ """从文件路径读取 OLE2 容器并识别旧 Office 格式。"""
256
+ try:
257
+ with open(file_path, "rb") as f:
258
+ return _guess_ole2_suffix_by_bytes(f.read())
259
+ except OSError:
260
+ return None
261
+
262
+
263
+ def _has_pdf_signature_by_path(file_path: Path) -> bool:
264
+ """读取文件头判断路径指向的内容是否具有 PDF 强签名。"""
265
+ try:
266
+ with open(file_path, "rb") as file:
267
+ return file.read(len(PDF_SIG_BYTES)) == PDF_SIG_BYTES
268
+ except OSError:
269
+ return False
270
+
271
+
272
+ def _has_rtf_signature_by_path(file_path: Path) -> bool:
273
+ """读取有限文件头并按共享规则识别 RTF 根组。"""
274
+ try:
275
+ with open(file_path, "rb") as file:
276
+ return rtf_header_offset(file.read(128)) is not None
277
+ except OSError:
278
+ return False
279
+
280
+
281
+ def _resolve_signatureless_csv_suffix(detected_suffix: str, file_path: str | Path | None) -> str:
282
+ """以 .csv/.tsv 扩展名兜底无签名分隔文本,并保留强内容类型的优先级。"""
283
+ extension = Path(file_path).suffix.lower().lstrip(".") if file_path else ""
284
+ if extension in CSV_EXTENSIONS:
285
+ if detected_suffix in _STRONG_CONTENT_SUFFIXES:
286
+ return detected_suffix
287
+ return "csv"
288
+ if detected_suffix == "csv":
289
+ if extension in ODF_MIMETYPE_SUFFIXES.values():
290
+ return "txt"
291
+ return extension or "txt"
292
+ return detected_suffix
293
+
294
+
295
+ def _resolve_signatureless_html_suffix(detected_suffix: str, file_path: str | Path | None) -> str:
296
+ """用 HTML_EXTENSIONS 兜底短文本,并把 Magika 的 HTML 结果统一规范为 html。"""
297
+ extension = Path(file_path).suffix.lower().lstrip(".") if file_path else ""
298
+ if extension in HTML_EXTENSIONS and detected_suffix not in _STRONG_CONTENT_SUFFIXES:
299
+ return "html"
300
+ return "html" if detected_suffix == "html" else detected_suffix
301
+
302
+
303
+ def _reject_unverified_package_suffix(detected_suffix: str) -> str:
304
+ """拒绝未通过包身份验证、仅由启发式工具猜出的 ODF/EPUB/OFD 类型。"""
305
+ package_suffixes = {*ODF_MIMETYPE_SUFFIXES.values(), "epub", "ofd"}
306
+ return "unknown" if detected_suffix in package_suffixes else detected_suffix
307
+
308
+
309
+ def guess_suffix_by_bytes(file_bytes: bytes, file_path: str | None = None) -> str:
310
+ if file_bytes[: len(PDF_SIG_BYTES)] == PDF_SIG_BYTES:
311
+ return "pdf"
312
+ if rtf_header_offset(file_bytes[:128]) is not None:
313
+ return "rtf"
314
+
315
+ ofd_suffix = _guess_ofd_suffix_by_bytes(file_bytes)
316
+ if ofd_suffix:
317
+ return ofd_suffix
318
+
319
+ epub_suffix = _guess_epub_suffix_by_bytes(file_bytes)
320
+ if epub_suffix:
321
+ return epub_suffix
322
+
323
+ ooxml_suffix = _guess_ooxml_suffix_by_bytes(file_bytes)
324
+ if ooxml_suffix:
325
+ return ooxml_suffix
326
+
327
+ odf_suffix = _guess_odf_suffix_by_bytes(file_bytes)
328
+ if odf_suffix:
329
+ return odf_suffix
330
+
331
+ ole2_suffix = _guess_ole2_suffix_by_bytes(file_bytes)
332
+ if ole2_suffix:
333
+ return ole2_suffix
334
+
335
+ suffix = _magika().identify_bytes(file_bytes).prediction.output.label
336
+ if (
337
+ file_path
338
+ and suffix in ["ai", "html"]
339
+ and Path(file_path).suffix.lower() in [".pdf"]
340
+ and file_bytes[:4] == PDF_SIG_BYTES
341
+ ):
342
+ suffix = "pdf"
343
+ suffix = _resolve_signatureless_csv_suffix(_reject_unverified_package_suffix(suffix), file_path)
344
+ return _resolve_signatureless_html_suffix(suffix, file_path)
345
+
346
+
347
+ def guess_suffix_by_path(file_path: str | Path) -> str:
348
+ if not isinstance(file_path, Path):
349
+ file_path = Path(file_path)
350
+
351
+ if _has_rtf_signature_by_path(file_path):
352
+ return "rtf"
353
+
354
+ ofd_suffix = _guess_ofd_suffix_by_path(file_path)
355
+ if ofd_suffix:
356
+ return ofd_suffix
357
+
358
+ epub_suffix = _guess_epub_suffix_by_path(file_path)
359
+ if epub_suffix:
360
+ return epub_suffix
361
+
362
+ ooxml_suffix = _guess_ooxml_suffix_by_path(file_path)
363
+ if ooxml_suffix:
364
+ return ooxml_suffix
365
+
366
+ odf_suffix = _guess_odf_suffix_by_path(file_path)
367
+ if odf_suffix:
368
+ return odf_suffix
369
+
370
+ ole2_suffix = _guess_ole2_suffix_by_path(file_path)
371
+ if ole2_suffix:
372
+ return ole2_suffix
373
+
374
+ if _has_pdf_signature_by_path(file_path):
375
+ return "pdf"
376
+
377
+ suffix = _magika().identify_path(file_path).prediction.output.label
378
+ if suffix in ["ai", "html"] and file_path.suffix.lower() in [".pdf"]:
379
+ try:
380
+ with open(file_path, "rb") as f:
381
+ if f.read(4) == PDF_SIG_BYTES:
382
+ suffix = "pdf"
383
+ except Exception as e:
384
+ logger.warning(f"Failed to read file {file_path} for PDF signature check: {e}")
385
+ suffix = _resolve_signatureless_csv_suffix(_reject_unverified_package_suffix(suffix), file_path)
386
+ return _resolve_signatureless_html_suffix(suffix, file_path)
387
+
388
+
389
+ __all__ = ["guess_suffix_by_bytes", "guess_suffix_by_path"]