docvortex 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (364) hide show
  1. docvortex/__init__.py +21 -0
  2. docvortex/analyzers/__init__.py +3 -0
  3. docvortex/analyzers/native/__init__.py +37 -0
  4. docvortex/analyzers/native/_shared/__init__.py +3 -0
  5. docvortex/analyzers/native/_shared/hyperlink.py +16 -0
  6. docvortex/analyzers/native/_shared/image.py +8 -0
  7. docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
  8. docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
  9. docvortex/analyzers/native/_shared/markup/formula.py +37 -0
  10. docvortex/analyzers/native/_shared/markup/projector.py +57 -0
  11. docvortex/analyzers/native/_shared/markup/styles.py +19 -0
  12. docvortex/analyzers/native/_shared/mathml.py +17 -0
  13. docvortex/analyzers/native/_shared/names.py +7 -0
  14. docvortex/analyzers/native/_shared/xycut.py +414 -0
  15. docvortex/analyzers/native/contracts.py +44 -0
  16. docvortex/analyzers/native/csv.py +351 -0
  17. docvortex/analyzers/native/epub/__init__.py +18 -0
  18. docvortex/analyzers/native/epub/constants.py +45 -0
  19. docvortex/analyzers/native/epub/converter.py +80 -0
  20. docvortex/analyzers/native/epub/errors.py +20 -0
  21. docvortex/analyzers/native/epub/metadata.py +31 -0
  22. docvortex/analyzers/native/epub/package.py +649 -0
  23. docvortex/analyzers/native/epub/xhtml.py +306 -0
  24. docvortex/analyzers/native/html/__init__.py +6 -0
  25. docvortex/analyzers/native/html/anchors.py +271 -0
  26. docvortex/analyzers/native/html/constants.py +25 -0
  27. docvortex/analyzers/native/html/contracts.py +7 -0
  28. docvortex/analyzers/native/html/converter.py +117 -0
  29. docvortex/analyzers/native/html/document.py +389 -0
  30. docvortex/analyzers/native/html/errors.py +12 -0
  31. docvortex/analyzers/native/html/resources.py +365 -0
  32. docvortex/analyzers/native/html/selector.py +421 -0
  33. docvortex/analyzers/native/models.py +209 -0
  34. docvortex/analyzers/native/ofd/__init__.py +14 -0
  35. docvortex/analyzers/native/ofd/constants.py +59 -0
  36. docvortex/analyzers/native/ofd/converter.py +31 -0
  37. docvortex/analyzers/native/ofd/errors.py +16 -0
  38. docvortex/analyzers/native/ofd/geometry.py +215 -0
  39. docvortex/analyzers/native/ofd/images.py +124 -0
  40. docvortex/analyzers/native/ofd/metadata.py +53 -0
  41. docvortex/analyzers/native/ofd/models.py +170 -0
  42. docvortex/analyzers/native/ofd/package.py +346 -0
  43. docvortex/analyzers/native/ofd/path.py +201 -0
  44. docvortex/analyzers/native/ofd/reading_order.py +293 -0
  45. docvortex/analyzers/native/ofd/resources.py +122 -0
  46. docvortex/analyzers/native/ofd/scene.py +347 -0
  47. docvortex/analyzers/native/ofd/table.py +240 -0
  48. docvortex/analyzers/native/ofd/text.py +521 -0
  49. docvortex/analyzers/native/office/__init__.py +3 -0
  50. docvortex/analyzers/native/office/doc/__init__.py +3 -0
  51. docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
  52. docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
  53. docvortex/analyzers/native/office/doc/fib.py +261 -0
  54. docvortex/analyzers/native/office/doc/fields.py +101 -0
  55. docvortex/analyzers/native/office/doc/formatting.py +171 -0
  56. docvortex/analyzers/native/office/doc/images.py +162 -0
  57. docvortex/analyzers/native/office/doc/lists.py +356 -0
  58. docvortex/analyzers/native/office/doc/models.py +164 -0
  59. docvortex/analyzers/native/office/doc/parser.py +851 -0
  60. docvortex/analyzers/native/office/doc/pieces.py +256 -0
  61. docvortex/analyzers/native/office/doc/records.py +40 -0
  62. docvortex/analyzers/native/office/doc/sprm.py +269 -0
  63. docvortex/analyzers/native/office/doc/styles.py +215 -0
  64. docvortex/analyzers/native/office/docx/__init__.py +3 -0
  65. docvortex/analyzers/native/office/docx/context.py +71 -0
  66. docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
  67. docvortex/analyzers/native/office/docx/equationxml.py +119 -0
  68. docvortex/analyzers/native/office/docx/fields.py +816 -0
  69. docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
  70. docvortex/analyzers/native/office/docx/main.py +50 -0
  71. docvortex/analyzers/native/office/docx/numbering.py +491 -0
  72. docvortex/analyzers/native/office/docx/office_xml.py +57 -0
  73. docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
  74. docvortex/analyzers/native/office/docx/resources.py +526 -0
  75. docvortex/analyzers/native/office/docx/styles.py +693 -0
  76. docvortex/analyzers/native/office/docx/tables.py +515 -0
  77. docvortex/analyzers/native/office/equation/__init__.py +3 -0
  78. docvortex/analyzers/native/office/equation/image.py +470 -0
  79. docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
  80. docvortex/analyzers/native/office/equation/mtef.py +885 -0
  81. docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
  82. docvortex/analyzers/native/office/equation/omml.py +561 -0
  83. docvortex/analyzers/native/office/equation/ooxml.py +62 -0
  84. docvortex/analyzers/native/office/errors.py +33 -0
  85. docvortex/analyzers/native/office/image.py +307 -0
  86. docvortex/analyzers/native/office/legacy/__init__.py +3 -0
  87. docvortex/analyzers/native/office/legacy/binary.py +43 -0
  88. docvortex/analyzers/native/office/legacy/officeart.py +362 -0
  89. docvortex/analyzers/native/office/legacy/ole.py +110 -0
  90. docvortex/analyzers/native/office/limits.py +12 -0
  91. docvortex/analyzers/native/office/odf/__init__.py +3 -0
  92. docvortex/analyzers/native/office/odf/chart.py +76 -0
  93. docvortex/analyzers/native/office/odf/constants.py +70 -0
  94. docvortex/analyzers/native/office/odf/converters.py +403 -0
  95. docvortex/analyzers/native/office/odf/errors.py +18 -0
  96. docvortex/analyzers/native/office/odf/metadata.py +91 -0
  97. docvortex/analyzers/native/office/odf/models.py +176 -0
  98. docvortex/analyzers/native/office/odf/package.py +270 -0
  99. docvortex/analyzers/native/office/odf/styles.py +329 -0
  100. docvortex/analyzers/native/office/odf/table.py +469 -0
  101. docvortex/analyzers/native/office/odf/text.py +1002 -0
  102. docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
  103. docvortex/analyzers/native/office/opc.py +38 -0
  104. docvortex/analyzers/native/office/ppt/__init__.py +3 -0
  105. docvortex/analyzers/native/office/ppt/models.py +118 -0
  106. docvortex/analyzers/native/office/ppt/parser.py +1895 -0
  107. docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
  108. docvortex/analyzers/native/office/ppt/records.py +131 -0
  109. docvortex/analyzers/native/office/ppt/style_text.py +247 -0
  110. docvortex/analyzers/native/office/pptx/__init__.py +3 -0
  111. docvortex/analyzers/native/office/pptx/context.py +113 -0
  112. docvortex/analyzers/native/office/pptx/lists.py +558 -0
  113. docvortex/analyzers/native/office/pptx/main.py +20 -0
  114. docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
  115. docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
  116. docvortex/analyzers/native/office/pptx/resources.py +329 -0
  117. docvortex/analyzers/native/office/pptx/shapes.py +393 -0
  118. docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
  119. docvortex/analyzers/native/office/pptx/titles.py +178 -0
  120. docvortex/analyzers/native/office/rich_text.py +420 -0
  121. docvortex/analyzers/native/office/rtf/__init__.py +3 -0
  122. docvortex/analyzers/native/office/rtf/converter.py +708 -0
  123. docvortex/analyzers/native/office/rtf/lexer.py +213 -0
  124. docvortex/analyzers/native/office/rtf/math.py +339 -0
  125. docvortex/analyzers/native/office/rtf/models.py +185 -0
  126. docvortex/analyzers/native/office/rtf/parser.py +1553 -0
  127. docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
  128. docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
  129. docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
  130. docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
  131. docvortex/analyzers/native/office/streams.py +18 -0
  132. docvortex/analyzers/native/office/xls/__init__.py +3 -0
  133. docvortex/analyzers/native/office/xls/chart.py +132 -0
  134. docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
  135. docvortex/analyzers/native/office/xls/models.py +109 -0
  136. docvortex/analyzers/native/office/xls/number_format.py +521 -0
  137. docvortex/analyzers/native/office/xls/parser.py +1145 -0
  138. docvortex/analyzers/native/office/xls/records.py +201 -0
  139. docvortex/analyzers/native/office/xls/strings.py +205 -0
  140. docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
  141. docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
  142. docvortex/analyzers/native/office/xlsx/main.py +20 -0
  143. docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
  144. docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
  145. docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
  146. docvortex/analyzers/native/pdf/__init__.py +3 -0
  147. docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
  148. docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
  149. docvortex/analyzers/native/pdf/code_blocks.py +535 -0
  150. docvortex/analyzers/native/pdf/formulas.py +1985 -0
  151. docvortex/analyzers/native/pdf/geometry.py +281 -0
  152. docvortex/analyzers/native/pdf/graphics.py +1501 -0
  153. docvortex/analyzers/native/pdf/index_blocks.py +268 -0
  154. docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
  155. docvortex/analyzers/native/pdf/inline/common.py +112 -0
  156. docvortex/analyzers/native/pdf/inline/detection.py +590 -0
  157. docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
  158. docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
  159. docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
  160. docvortex/analyzers/native/pdf/inline/types.py +385 -0
  161. docvortex/analyzers/native/pdf/line_layout.py +1106 -0
  162. docvortex/analyzers/native/pdf/line_merging.py +1223 -0
  163. docvortex/analyzers/native/pdf/models.py +246 -0
  164. docvortex/analyzers/native/pdf/native_text.py +1004 -0
  165. docvortex/analyzers/native/pdf/pipeline.py +1390 -0
  166. docvortex/analyzers/native/pdf/script_geometry.py +636 -0
  167. docvortex/analyzers/native/pdf/shared.py +30 -0
  168. docvortex/analyzers/native/pdf/spatial_text.py +383 -0
  169. docvortex/analyzers/native/pdf/table_annotations.py +446 -0
  170. docvortex/analyzers/native/pdf/table_constants.py +48 -0
  171. docvortex/analyzers/native/pdf/table_detection.py +227 -0
  172. docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
  173. docvortex/analyzers/native/pdf/table_geometry.py +40 -0
  174. docvortex/analyzers/native/pdf/table_materialization.py +444 -0
  175. docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
  176. docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
  177. docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
  178. docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
  179. docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
  180. docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
  181. docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
  182. docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
  183. docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
  184. docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
  185. docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
  186. docvortex/analyzers/native/pdf/table_rows.py +34 -0
  187. docvortex/analyzers/native/pdf/table_rules.py +1129 -0
  188. docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
  189. docvortex/analyzers/native/pdf/tables.py +147 -0
  190. docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
  191. docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
  192. docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
  193. docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
  194. docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
  195. docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
  196. docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
  197. docvortex/analyzers/native/pdf/text_blocks.py +82 -0
  198. docvortex/analyzers/native/pdf/text_styles.py +55 -0
  199. docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
  200. docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
  201. docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
  202. docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
  203. docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
  204. docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
  205. docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
  206. docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
  207. docvortex/analyzers/native/pdf/titles.py +75 -0
  208. docvortex/analyzers/native/pdf/typography.py +19 -0
  209. docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
  210. docvortex/api.py +180 -0
  211. docvortex/assets/__init__.py +5 -0
  212. docvortex/assets/store.py +51 -0
  213. docvortex/cli.py +54 -0
  214. docvortex/codecs/__init__.py +3 -0
  215. docvortex/codecs/html/__init__.py +22 -0
  216. docvortex/codecs/html/contracts.py +236 -0
  217. docvortex/codecs/html/materializer.py +331 -0
  218. docvortex/codecs/html/parser.py +763 -0
  219. docvortex/codecs/html/resources.py +34 -0
  220. docvortex/codecs/json.py +17 -0
  221. docvortex/content/__init__.py +5 -0
  222. docvortex/content/inline.py +248 -0
  223. docvortex/content/markup/__init__.py +44 -0
  224. docvortex/content/markup/anchors.py +188 -0
  225. docvortex/content/markup/formula.py +280 -0
  226. docvortex/content/markup/projector.py +1237 -0
  227. docvortex/content/markup/styles.py +327 -0
  228. docvortex/content/mathml.py +167 -0
  229. docvortex/content/normalization.py +188 -0
  230. docvortex/content/spans.py +183 -0
  231. docvortex/content/table/__init__.py +18 -0
  232. docvortex/content/table/blocks.py +152 -0
  233. docvortex/content/table/content.py +425 -0
  234. docvortex/content/table/document.py +104 -0
  235. docvortex/content/table/html.py +399 -0
  236. docvortex/content/table/models.py +76 -0
  237. docvortex/content/table/rules.py +42 -0
  238. docvortex/content/table/structure.py +221 -0
  239. docvortex/content/tree.py +5 -0
  240. docvortex/document/__init__.py +3 -0
  241. docvortex/document/contracts.py +22 -0
  242. docvortex/document/detection.py +389 -0
  243. docvortex/document/filetypes.py +175 -0
  244. docvortex/document/page_range.py +167 -0
  245. docvortex/document/pdf/__init__.py +19 -0
  246. docvortex/document/pdf/classify.py +1138 -0
  247. docvortex/document/pdf/constants.py +132 -0
  248. docvortex/document/pdf/diagnostics.py +350 -0
  249. docvortex/document/pdf/document.py +582 -0
  250. docvortex/document/pdf/font_runtime.py +335 -0
  251. docvortex/document/pdf/geometry.py +31 -0
  252. docvortex/document/pdf/images.py +654 -0
  253. docvortex/document/pdf/native_annotations.py +367 -0
  254. docvortex/document/pdf/native_contracts.py +169 -0
  255. docvortex/document/pdf/native_coordinates.py +216 -0
  256. docvortex/document/pdf/native_lifecycle.py +16 -0
  257. docvortex/document/pdf/native_objects.py +902 -0
  258. docvortex/document/pdf/native_text_geometry.py +315 -0
  259. docvortex/document/pdf/pdfium.py +325 -0
  260. docvortex/document/pdf/raster.py +46 -0
  261. docvortex/document/pdf/text/__init__.py +62 -0
  262. docvortex/document/pdf/text/contracts.py +211 -0
  263. docvortex/document/pdf/text/extract.py +165 -0
  264. docvortex/document/pdf/text/geometry.py +16 -0
  265. docvortex/document/pdf/text/groups.py +162 -0
  266. docvortex/document/pdf/visual_geometry.py +201 -0
  267. docvortex/document/pdf/visuals.py +343 -0
  268. docvortex/document/source.py +92 -0
  269. docvortex/errors.py +21 -0
  270. docvortex/export/__init__.py +3 -0
  271. docvortex/export/bundle.py +98 -0
  272. docvortex/export/files.py +66 -0
  273. docvortex/export/middle.py +208 -0
  274. docvortex/foundation/__init__.py +3 -0
  275. docvortex/foundation/geometry.py +125 -0
  276. docvortex/foundation/hyperlink.py +65 -0
  277. docvortex/foundation/image.py +48 -0
  278. docvortex/foundation/image_encoding.py +30 -0
  279. docvortex/foundation/image_payload.py +280 -0
  280. docvortex/foundation/language.py +92 -0
  281. docvortex/foundation/platform.py +38 -0
  282. docvortex/foundation/text.py +153 -0
  283. docvortex/foundation/type_identity.py +20 -0
  284. docvortex/foundation/xml_names.py +20 -0
  285. docvortex/options.py +30 -0
  286. docvortex/postprocess/__init__.py +3 -0
  287. docvortex/postprocess/content.py +53 -0
  288. docvortex/postprocess/document.py +19 -0
  289. docvortex/postprocess/lists.py +236 -0
  290. docvortex/postprocess/page_blocks.py +214 -0
  291. docvortex/postprocess/pages.py +95 -0
  292. docvortex/postprocess/paragraphs.py +580 -0
  293. docvortex/postprocess/visual.py +715 -0
  294. docvortex/render/__init__.py +48 -0
  295. docvortex/render/_internal/__init__.py +3 -0
  296. docvortex/render/_internal/common/__init__.py +3 -0
  297. docvortex/render/_internal/common/context.py +43 -0
  298. docvortex/render/_internal/common/html_table.py +178 -0
  299. docvortex/render/_internal/common/index.py +33 -0
  300. docvortex/render/_internal/common/list_items.py +158 -0
  301. docvortex/render/_internal/common/planner.py +140 -0
  302. docvortex/render/_internal/docx/__init__.py +3 -0
  303. docvortex/render/_internal/docx/assets.py +202 -0
  304. docvortex/render/_internal/docx/inline.py +434 -0
  305. docvortex/render/_internal/docx/math.py +220 -0
  306. docvortex/render/_internal/docx/renderer.py +905 -0
  307. docvortex/render/_internal/docx/styles.py +195 -0
  308. docvortex/render/_internal/docx/table.py +442 -0
  309. docvortex/render/_internal/epub/__init__.py +5 -0
  310. docvortex/render/_internal/epub/assets.py +173 -0
  311. docvortex/render/_internal/epub/package.py +249 -0
  312. docvortex/render/_internal/epub/renderer.py +1156 -0
  313. docvortex/render/_internal/html/__init__.py +3 -0
  314. docvortex/render/_internal/html/inline.py +346 -0
  315. docvortex/render/_internal/html/renderer.py +1041 -0
  316. docvortex/render/_internal/html/sanitizer.py +478 -0
  317. docvortex/render/_internal/html/table.py +121 -0
  318. docvortex/render/_internal/latex/__init__.py +1 -0
  319. docvortex/render/_internal/latex/assets.py +85 -0
  320. docvortex/render/_internal/latex/inline.py +145 -0
  321. docvortex/render/_internal/latex/renderer.py +506 -0
  322. docvortex/render/_internal/latex/table.py +347 -0
  323. docvortex/render/_internal/markdown/__init__.py +3 -0
  324. docvortex/render/_internal/markdown/assets.py +78 -0
  325. docvortex/render/_internal/markdown/blocks.py +635 -0
  326. docvortex/render/_internal/markdown/escaping.py +51 -0
  327. docvortex/render/_internal/markdown/inline.py +260 -0
  328. docvortex/render/_internal/markdown/renderer.py +93 -0
  329. docvortex/render/_internal/markdown/table.py +281 -0
  330. docvortex/render/_internal/pdf/__init__.py +3 -0
  331. docvortex/render/_internal/pdf/assets.py +197 -0
  332. docvortex/render/_internal/pdf/formula.py +417 -0
  333. docvortex/render/_internal/pdf/inline.py +343 -0
  334. docvortex/render/_internal/pdf/renderer.py +734 -0
  335. docvortex/render/_internal/pdf/styles.py +206 -0
  336. docvortex/render/_internal/pdf/table.py +272 -0
  337. docvortex/render/_internal/structured_content/__init__.py +3 -0
  338. docvortex/render/_internal/structured_content/renderer.py +193 -0
  339. docvortex/render/api.py +199 -0
  340. docvortex/render/contracts.py +205 -0
  341. docvortex/render/docx.py +38 -0
  342. docvortex/render/epub.py +35 -0
  343. docvortex/render/fragments.py +70 -0
  344. docvortex/render/html.py +29 -0
  345. docvortex/render/latex.py +24 -0
  346. docvortex/render/markdown.py +50 -0
  347. docvortex/render/pdf.py +25 -0
  348. docvortex/render/structured_content.py +23 -0
  349. docvortex/resources/epub/docvortex.css +91 -0
  350. docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
  351. docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
  352. docvortex/resources/fonts/NOTICE +190 -0
  353. docvortex/resources/fonts/manifest.json +9 -0
  354. docvortex/resources/html/docvortex.css +601 -0
  355. docvortex/resources/html/docvortex.min.css +1 -0
  356. docvortex/result.py +91 -0
  357. docvortex/schema.py +1141 -0
  358. docvortex/version.py +3 -0
  359. docvortex-0.2.1.dist-info/METADATA +193 -0
  360. docvortex-0.2.1.dist-info/RECORD +364 -0
  361. docvortex-0.2.1.dist-info/WHEEL +5 -0
  362. docvortex-0.2.1.dist-info/entry_points.txt +2 -0
  363. docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
  364. docvortex-0.2.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,261 @@
1
+ """读取 Word 97–2003 WordDocument stream 中的变长 FIB。"""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import dataclass
6
+ import struct
7
+
8
+ from ..errors import LegacyOfficeMalformedError
9
+
10
+ FIB_IDENT = 0xA5EC
11
+ MIN_WORD97_NFIB = 0x00C1
12
+
13
+ FCLCB_STSHF = 1
14
+ FCLCB_FOOTNOTE_REF = 2
15
+ FCLCB_FOOTNOTE_TEXT = 3
16
+ FCLCB_SECTION = 6
17
+ FCLCB_HEADER = 11
18
+ FCLCB_BTE_CHPX = 12
19
+ FCLCB_BTE_PAPX = 13
20
+ FCLCB_FIELD_MAIN = 16
21
+ FCLCB_FIELD_HEADER = 17
22
+ FCLCB_FIELD_FOOTNOTE = 18
23
+ FCLCB_BOOKMARK_NAMES = 21
24
+ FCLCB_BOOKMARK_START = 22
25
+ FCLCB_BOOKMARK_END = 23
26
+ FCLCB_DOP = 31
27
+ FCLCB_CLX = 33
28
+ FCLCB_SHAPE_MAIN = 40
29
+ FCLCB_SHAPE_HEADER = 41
30
+ FCLCB_ENDNOTE_REF = 46
31
+ FCLCB_ENDNOTE_TEXT = 47
32
+ FCLCB_FIELD_ENDNOTE = 48
33
+ FCLCB_DGG_INFO = 50
34
+ FCLCB_TEXTBOX_TEXT = 56
35
+ FCLCB_FIELD_TEXTBOX = 57
36
+ FCLCB_HEADER_TEXTBOX_TEXT = 58
37
+ FCLCB_FIELD_HEADER_TEXTBOX = 59
38
+ FCLCB_LISTS = 73
39
+ FCLCB_LIST_OVERRIDES = 74
40
+ FCLCB_TEXTBOX_BREAK = 75
41
+ FCLCB_HEADER_TEXTBOX_BREAK = 76
42
+
43
+
44
+ @dataclass(frozen=True, slots=True)
45
+ class FcLcb:
46
+ """FIB 中一对 stream 偏移和字节长度。"""
47
+
48
+ fc: int = 0
49
+ lcb: int = 0
50
+
51
+
52
+ @dataclass(frozen=True, slots=True)
53
+ class FibBase:
54
+ """FIB 固定头中与解析相关的字段。"""
55
+
56
+ n_fib: int
57
+ lid: int
58
+ flags: int
59
+ fc_min: int
60
+ fc_mac: int
61
+
62
+ @property
63
+ def complex(self) -> bool:
64
+ """返回文档是否使用 complex/fast-save piece table。"""
65
+
66
+ return bool(self.flags & 0x0004)
67
+
68
+ @property
69
+ def encrypted(self) -> bool:
70
+ """返回文档是否设置加密标志。"""
71
+
72
+ return bool(self.flags & 0x0100)
73
+
74
+ @property
75
+ def uses_1table(self) -> bool:
76
+ """返回 FIB 指定的首选 Table stream。"""
77
+
78
+ return bool(self.flags & 0x0200)
79
+
80
+ @property
81
+ def far_east(self) -> bool:
82
+ """返回文档是否优先使用远东语言标识。"""
83
+
84
+ return bool(self.flags & 0x4000)
85
+
86
+ @property
87
+ def obfuscated(self) -> bool:
88
+ """返回文档是否设置 XOR 混淆标志。"""
89
+
90
+ return bool(self.flags & 0x8000)
91
+
92
+
93
+ @dataclass(frozen=True, slots=True)
94
+ class FileInformationBlock:
95
+ """完成边界校验的 Word 97+ FIB。"""
96
+
97
+ base: FibBase
98
+ rgw: tuple[int, ...]
99
+ rglw: tuple[int, ...]
100
+ pairs: tuple[FcLcb, ...]
101
+ csw_new: tuple[int, ...]
102
+ size: int
103
+
104
+ @property
105
+ def n_fib(self) -> int:
106
+ """返回版本扩展中的有效 nFib。"""
107
+
108
+ return self.csw_new[0] if self.csw_new else self.base.n_fib
109
+
110
+ def pair(self, index: int) -> FcLcb:
111
+ """读取可选 fc/lcb 对,不存在时返回零值。"""
112
+
113
+ return self.pairs[index] if 0 <= index < len(self.pairs) else FcLcb()
114
+
115
+ def story_count(self, index: int) -> int:
116
+ """读取 FibRgLw97 中一个 story 的 UTF-16 CP 数。"""
117
+
118
+ return int(self.rglw[index]) if 0 <= index < len(self.rglw) else 0
119
+
120
+ @property
121
+ def ccp_text(self) -> int:
122
+ """返回主文档 story 的 CP 数。"""
123
+
124
+ return self.story_count(3)
125
+
126
+ @property
127
+ def ccp_footnote(self) -> int:
128
+ """返回脚注 story 的 CP 数。"""
129
+
130
+ return self.story_count(4)
131
+
132
+ @property
133
+ def ccp_header(self) -> int:
134
+ """返回页眉页脚 story 的 CP 数。"""
135
+
136
+ return self.story_count(5)
137
+
138
+ @property
139
+ def ccp_macro(self) -> int:
140
+ """返回宏 story 的 CP 数。"""
141
+
142
+ return self.story_count(6)
143
+
144
+ @property
145
+ def ccp_annotation(self) -> int:
146
+ """返回批注 story 的 CP 数。"""
147
+
148
+ return self.story_count(7)
149
+
150
+ @property
151
+ def ccp_endnote(self) -> int:
152
+ """返回尾注 story 的 CP 数。"""
153
+
154
+ return self.story_count(8)
155
+
156
+ @property
157
+ def ccp_textbox(self) -> int:
158
+ """返回正文文本框 story 的 CP 数。"""
159
+
160
+ return self.story_count(9)
161
+
162
+ @property
163
+ def ccp_header_textbox(self) -> int:
164
+ """返回页眉文本框 story 的 CP 数。"""
165
+
166
+ return self.story_count(10)
167
+
168
+ @property
169
+ def total_story_cp(self) -> int:
170
+ """返回全部已知 story 的累计 CP 数。"""
171
+
172
+ return sum(self.story_count(index) for index in range(3, 11))
173
+
174
+ @property
175
+ def story_bases(self) -> dict[str, int]:
176
+ """返回各 story 在全局 CP 空间中的起点。"""
177
+
178
+ counts = [self.story_count(index) for index in range(3, 11)]
179
+ names = [
180
+ "main",
181
+ "footnote",
182
+ "header",
183
+ "macro",
184
+ "annotation",
185
+ "endnote",
186
+ "textbox",
187
+ "header_textbox",
188
+ ]
189
+ bases: dict[str, int] = {}
190
+ cursor = 0
191
+ for name, count in zip(names, counts, strict=True):
192
+ bases[name] = cursor
193
+ cursor += count
194
+ return bases
195
+
196
+
197
+ def _read_values(data: bytes, offset: int, count: int, width: int, label: str) -> tuple[tuple[int, ...], int]:
198
+ """按指定宽度读取一组无符号小端整数。"""
199
+
200
+ if count < 0 or width not in {2, 4}:
201
+ raise LegacyOfficeMalformedError(f"invalid {label} count")
202
+ size = count * width
203
+ end = offset + size
204
+ if offset < 0 or end < offset or end > len(data):
205
+ raise LegacyOfficeMalformedError(f"truncated FIB {label}")
206
+ if count == 0:
207
+ return (), end
208
+ code = "H" if width == 2 else "I"
209
+ return tuple(int(value) for value in struct.unpack_from(f"<{count}{code}", data, offset)), end
210
+
211
+
212
+ def parse_fib(word_document: bytes) -> FileInformationBlock:
213
+ """按 MS-DOC 变长布局解析 FIB,并拒绝 Word 95 及更早版本。"""
214
+
215
+ if len(word_document) < 34:
216
+ raise LegacyOfficeMalformedError("WordDocument FIB is truncated")
217
+ ident, n_fib = struct.unpack_from("<HH", word_document, 0)
218
+ if ident != FIB_IDENT:
219
+ raise LegacyOfficeMalformedError("WordDocument FIB magic is invalid")
220
+ if n_fib < MIN_WORD97_NFIB:
221
+ raise LegacyOfficeMalformedError(f"Word 95 or earlier nFib is unsupported: 0x{n_fib:04X}")
222
+ lid = int(struct.unpack_from("<H", word_document, 6)[0])
223
+ flags = int(struct.unpack_from("<H", word_document, 10)[0])
224
+ fc_min = int(struct.unpack_from("<I", word_document, 24)[0])
225
+ fc_mac = int(struct.unpack_from("<I", word_document, 28)[0])
226
+ base = FibBase(n_fib=n_fib, lid=lid, flags=flags, fc_min=fc_min, fc_mac=fc_mac)
227
+
228
+ cursor = 32
229
+ csw = int(struct.unpack_from("<H", word_document, cursor)[0])
230
+ cursor += 2
231
+ rgw, cursor = _read_values(word_document, cursor, csw, 2, "FibRgW")
232
+ if cursor + 2 > len(word_document):
233
+ raise LegacyOfficeMalformedError("truncated FIB cslw")
234
+ cslw = int(struct.unpack_from("<H", word_document, cursor)[0])
235
+ cursor += 2
236
+ rglw, cursor = _read_values(word_document, cursor, cslw, 4, "FibRgLw")
237
+ if cursor + 2 > len(word_document):
238
+ raise LegacyOfficeMalformedError("truncated FIB cbRgFcLcb")
239
+ pair_count = int(struct.unpack_from("<H", word_document, cursor)[0])
240
+ cursor += 2
241
+ raw_pairs, cursor = _read_values(word_document, cursor, pair_count * 2, 4, "FibRgFcLcb")
242
+ pairs = tuple(FcLcb(raw_pairs[index], raw_pairs[index + 1]) for index in range(0, len(raw_pairs), 2))
243
+
244
+ csw_new: tuple[int, ...] = ()
245
+ if cursor + 2 <= len(word_document):
246
+ count = int(struct.unpack_from("<H", word_document, cursor)[0])
247
+ cursor += 2
248
+ csw_new, cursor = _read_values(word_document, cursor, count, 2, "FibRgCswNew")
249
+ if len(rglw) <= 3:
250
+ # 确定性最小 fixture 可能省略变长计数,但仍保留 Word 97 固定槽位;
251
+ # 仅在标准布局不可用时按这些公开槽位做恢复读取。
252
+ if len(word_document) < 0x6C:
253
+ raise LegacyOfficeMalformedError("FIB does not contain ccpText")
254
+ rglw = tuple(int(struct.unpack_from("<I", word_document, 0x40 + index * 4)[0]) for index in range(11))
255
+ if not pairs and len(word_document) >= 0x382:
256
+ pairs = tuple(FcLcb(*struct.unpack_from("<II", word_document, 0x9A + index * 8)) for index in range(93))
257
+ cursor = max(cursor, 0x382)
258
+ fib = FileInformationBlock(base=base, rgw=rgw, rglw=rglw, pairs=pairs, csw_new=csw_new, size=cursor)
259
+ if fib.total_story_cp < fib.ccp_text:
260
+ raise LegacyOfficeMalformedError("FIB story CP count overflow")
261
+ return fib
@@ -0,0 +1,101 @@
1
+ """解析 DOC 字段指令并安全恢复超链接、目录和 caption 语义。"""
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+
7
+ from loguru import logger
8
+
9
+ from ..._shared.hyperlink import OFFICE_EXTERNAL_HYPERLINK_SCHEMES, sanitize_hyperlink_target
10
+ from .models import DocTextRun
11
+
12
+ _TOKEN_RE = re.compile(r'"(?:\\.|[^"\\])*"|\\\S|\S+')
13
+
14
+
15
+ def field_keyword(instruction: str) -> str:
16
+ """返回字段指令的首个关键字大写形式。"""
17
+
18
+ tokens = _TOKEN_RE.findall(instruction.strip())
19
+ return tokens[0].strip('"').upper() if tokens else ""
20
+
21
+
22
+ def is_toc_field(instruction: str) -> bool:
23
+ """判断字段是否为多段落 TOC。"""
24
+
25
+ return field_keyword(instruction) == "TOC"
26
+
27
+
28
+ def is_caption_field(instruction: str) -> bool:
29
+ """判断字段是否为 Word SEQ caption 编号。"""
30
+
31
+ return field_keyword(instruction) == "SEQ"
32
+
33
+
34
+ def is_chart_embed_field(instruction: str) -> bool:
35
+ """判断 EMBED 字段是否声明 Excel.Chart 或 MSGraph.Chart 对象。"""
36
+
37
+ tokens = _TOKEN_RE.findall(instruction.strip())
38
+ if len(tokens) < 2 or _unquote(tokens[0]).casefold() != "embed":
39
+ return False
40
+ prog_id = _unquote(tokens[1]).casefold()
41
+ return prog_id.startswith(("excel.chart", "msgraph.chart"))
42
+
43
+
44
+ def _unquote(token: str) -> str:
45
+ """解码字段引号内允许的反斜杠转义。"""
46
+
47
+ if len(token) >= 2 and token[0] == token[-1] == '"':
48
+ token = token[1:-1]
49
+ return token.replace(r"\"", '"').replace(r"\\", "\\")
50
+
51
+
52
+ def hyperlink_target(instruction: str) -> str | None:
53
+ """从 HYPERLINK 字段读取 URL 与可选内部书签。"""
54
+
55
+ tokens = _TOKEN_RE.findall(instruction.strip())
56
+ if not tokens or _unquote(tokens[0]).casefold() != "hyperlink":
57
+ return None
58
+ url: str | None = None
59
+ anchor: str | None = None
60
+ index = 1
61
+ while index < len(tokens):
62
+ token = tokens[index]
63
+ if token.startswith("\\"):
64
+ switch = token[1:].casefold()
65
+ argument: str | None = None
66
+ if switch in {"l", "o", "t"} and index + 1 < len(tokens) and not tokens[index + 1].startswith("\\"):
67
+ index += 1
68
+ argument = _unquote(tokens[index]).strip()
69
+ if switch == "l" and argument:
70
+ anchor = argument
71
+ elif url is None:
72
+ url = _unquote(token).strip()
73
+ index += 1
74
+ if url and anchor:
75
+ candidate = f"{url}#{anchor}"
76
+ elif url:
77
+ candidate = url
78
+ elif anchor:
79
+ candidate = f"#{anchor}"
80
+ else:
81
+ return None
82
+ safe = sanitize_hyperlink_target(
83
+ candidate,
84
+ allowed_schemes=OFFICE_EXTERNAL_HYPERLINK_SCHEMES,
85
+ allow_relative=True,
86
+ allow_fragment=True,
87
+ )
88
+ if safe is None:
89
+ logger.warning(f"DOC hyperlink target was rejected: {candidate!r}")
90
+ return safe
91
+
92
+
93
+ def apply_field_result(instruction: str, runs: list[DocTextRun]) -> list[DocTextRun]:
94
+ """把 HYPERLINK 目标绑定到字段结果,其他字段仅保留缓存结果。"""
95
+
96
+ if field_keyword(instruction) != "HYPERLINK":
97
+ return runs
98
+ target = hyperlink_target(instruction)
99
+ if target is None:
100
+ return runs
101
+ return [DocTextRun(run.text, run.style, target, run.formula) for run in runs]
@@ -0,0 +1,171 @@
1
+ """解析 Word CHPX/PAPX FKP 页面并提供按 FC 查询的格式 run。"""
2
+
3
+ from __future__ import annotations
4
+
5
+ from bisect import bisect_right
6
+ from dataclasses import dataclass
7
+
8
+ from loguru import logger
9
+
10
+ from ..legacy.binary import bounded_slice, get_u32
11
+ from .records import DocBudget
12
+ from .sprm import PapDelta, apply_paragraph_sprms
13
+
14
+
15
+ @dataclass(frozen=True, slots=True)
16
+ class CharacterRun:
17
+ """一个物理 FC 范围内的原始 CHPX。"""
18
+
19
+ fc_start: int
20
+ fc_end: int
21
+ grpprl: bytes
22
+
23
+
24
+ @dataclass(frozen=True, slots=True)
25
+ class ParagraphRun:
26
+ """一个物理 FC 范围内的段落样式和 PAPX。"""
27
+
28
+ fc_start: int
29
+ fc_end: int
30
+ style_id: int
31
+ delta: PapDelta
32
+
33
+
34
+ class FormattingRuns:
35
+ """按起始 FC 排序的 CHPX/PAPX 查询索引。"""
36
+
37
+ def __init__(self, characters: list[CharacterRun], paragraphs: list[ParagraphRun]) -> None:
38
+ """排序 run 并缓存二分查询键。"""
39
+
40
+ self.characters = sorted(characters, key=lambda run: run.fc_start)
41
+ self.paragraphs = sorted(paragraphs, key=lambda run: run.fc_start)
42
+ self._character_starts = [run.fc_start for run in self.characters]
43
+ self._paragraph_starts = [run.fc_start for run in self.paragraphs]
44
+
45
+ def character_at(self, fc: int) -> CharacterRun | None:
46
+ """返回覆盖指定 FC 的最后一个 CHPX run。"""
47
+
48
+ index = bisect_right(self._character_starts, fc) - 1
49
+ if index < 0:
50
+ return None
51
+ run = self.characters[index]
52
+ return run if fc < run.fc_end else None
53
+
54
+ def paragraph_at(self, fc: int) -> ParagraphRun | None:
55
+ """返回覆盖指定 FC 的最后一个 PAPX run。"""
56
+
57
+ index = bisect_right(self._paragraph_starts, fc) - 1
58
+ if index < 0:
59
+ return None
60
+ run = self.paragraphs[index]
61
+ return run if fc < run.fc_end else None
62
+
63
+
64
+ def _parse_bte_pages(table_stream: bytes, offset: int, size: int) -> list[int]:
65
+ """从 PlcBteChpx/PlcBtePapx 读取 FKP page number。"""
66
+
67
+ plc = bounded_slice(table_stream, offset, size)
68
+ if plc is None or len(plc) < 8 or (len(plc) - 4) % 8:
69
+ return []
70
+ count = (len(plc) - 4) // 8
71
+ page_offset = (count + 1) * 4
72
+ pages: list[int] = []
73
+ for index in range(count):
74
+ raw = get_u32(plc, page_offset + index * 4)
75
+ if raw is not None:
76
+ pages.append(raw & 0x003F_FFFF)
77
+ return pages
78
+
79
+
80
+ def _parse_chpx_page(page: bytes, budget: DocBudget) -> list[CharacterRun]:
81
+ """解析一个 512 字节 ChpxFkp。"""
82
+
83
+ count = page[511]
84
+ if count == 0 or (count + 1) * 4 + count > 511:
85
+ return []
86
+ result: list[CharacterRun] = []
87
+ offset_base = (count + 1) * 4
88
+ for index in range(count):
89
+ fc_start = get_u32(page, index * 4)
90
+ fc_end = get_u32(page, (index + 1) * 4)
91
+ if fc_start is None or fc_end is None or fc_end <= fc_start:
92
+ continue
93
+ byte_offset = page[offset_base + index]
94
+ grpprl = b""
95
+ if byte_offset:
96
+ payload_offset = byte_offset * 2
97
+ if payload_offset < 511:
98
+ length = page[payload_offset]
99
+ grpprl = page[payload_offset + 1 : payload_offset + 1 + length]
100
+ budget.charge()
101
+ result.append(CharacterRun(fc_start, fc_end, grpprl))
102
+ return result
103
+
104
+
105
+ def _parse_papx_page(page: bytes, data_stream: bytes, budget: DocBudget) -> list[ParagraphRun]:
106
+ """解析一个 512 字节 PapxFkp。"""
107
+
108
+ count = page[511]
109
+ header_end = (count + 1) * 4 + count * 13
110
+ if count == 0 or header_end > 511:
111
+ return []
112
+ result: list[ParagraphRun] = []
113
+ bx_base = (count + 1) * 4
114
+ for index in range(count):
115
+ fc_start = get_u32(page, index * 4)
116
+ fc_end = get_u32(page, (index + 1) * 4)
117
+ if fc_start is None or fc_end is None or fc_end <= fc_start:
118
+ continue
119
+ byte_offset = page[bx_base + index * 13]
120
+ style_id = 0
121
+ delta = PapDelta()
122
+ if byte_offset:
123
+ payload_offset = byte_offset * 2
124
+ if payload_offset < 511:
125
+ first_length = page[payload_offset]
126
+ if first_length:
127
+ content_offset = payload_offset + 1
128
+ content_length = first_length * 2 - 1
129
+ elif payload_offset + 1 < 511:
130
+ content_offset = payload_offset + 2
131
+ content_length = page[payload_offset + 1] * 2
132
+ else:
133
+ content_offset = 511
134
+ content_length = 0
135
+ content = page[content_offset : min(content_offset + content_length, 511)]
136
+ if len(content) >= 2:
137
+ style_id = int.from_bytes(content[:2], "little")
138
+ delta = apply_paragraph_sprms(content[2:], data_stream, budget=budget)
139
+ budget.charge()
140
+ result.append(ParagraphRun(fc_start, fc_end, style_id, delta))
141
+ return result
142
+
143
+
144
+ def parse_formatting_runs(
145
+ word_document: bytes,
146
+ table_stream: bytes,
147
+ data_stream: bytes,
148
+ *,
149
+ chpx_offset: int,
150
+ chpx_size: int,
151
+ papx_offset: int,
152
+ papx_size: int,
153
+ budget: DocBudget,
154
+ ) -> FormattingRuns:
155
+ """解析 FIB 指向的全部 CHPX/PAPX FKP;坏可选页仅告警跳过。"""
156
+
157
+ characters: list[CharacterRun] = []
158
+ paragraphs: list[ParagraphRun] = []
159
+ for page_number in _parse_bte_pages(table_stream, chpx_offset, chpx_size):
160
+ page = bounded_slice(word_document, page_number * 512, 512)
161
+ if page is None:
162
+ logger.warning(f"DOC ChpxFkp page is truncated: {page_number}")
163
+ continue
164
+ characters.extend(_parse_chpx_page(page, budget))
165
+ for page_number in _parse_bte_pages(table_stream, papx_offset, papx_size):
166
+ page = bounded_slice(word_document, page_number * 512, 512)
167
+ if page is None:
168
+ logger.warning(f"DOC PapxFkp page is truncated: {page_number}")
169
+ continue
170
+ paragraphs.extend(_parse_papx_page(page, data_stream, budget))
171
+ return FormattingRuns(characters, paragraphs)
@@ -0,0 +1,162 @@
1
+ """从 DOC Data/PICF 与 Word OfficeArt drawing 中恢复图片。"""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import dataclass, field
6
+ import hashlib
7
+
8
+ from loguru import logger
9
+
10
+ from ..errors import LegacyOfficeResourceLimitError
11
+ from ..legacy.binary import bounded_slice, get_u16, get_u32
12
+ from ..limits import MAX_ASSET_TOTAL_BYTES
13
+ from ..legacy.officeart import OfficeImagePayload, decode_bstore, extract_word_shapes, first_blip, record_at
14
+ from ..equation.image import OfficeImageEquationDecoder
15
+
16
+ from .models import DocImage, DocImagePayload
17
+ from .records import DocBudget, parse_plc
18
+
19
+ _PLACEABLE_WMF_MAGIC = b"\xd7\xcd\xc6\x9a"
20
+
21
+
22
+ @dataclass(slots=True)
23
+ class ImageStore:
24
+ """按内容去重并限制 DOC 图片累计字节数。"""
25
+
26
+ total: int = 0
27
+ cache: dict[tuple[bytes, tuple[int, int] | None], DocImagePayload] = field(default_factory=dict)
28
+ accounted_digests: set[bytes] = field(default_factory=set)
29
+ equation_cache: dict[bytes, str | None] = field(default_factory=dict)
30
+ equation_decoder: OfficeImageEquationDecoder = field(default_factory=OfficeImageEquationDecoder)
31
+
32
+ def add(self, payload: OfficeImagePayload) -> DocImagePayload:
33
+ """计入一张唯一图片并返回内部载荷。"""
34
+
35
+ digest = hashlib.sha256(payload.data).digest()
36
+ cache_key = digest, payload.render_size_emu
37
+ cached = self.cache.get(cache_key)
38
+ if cached is not None:
39
+ return cached
40
+ is_new_payload = digest not in self.accounted_digests
41
+ if is_new_payload and self.total + len(payload.data) > MAX_ASSET_TOTAL_BYTES:
42
+ raise LegacyOfficeResourceLimitError(f"embedded assets exceed max_asset_total_bytes={MAX_ASSET_TOTAL_BYTES}")
43
+ if digest in self.equation_cache:
44
+ equation_latex = self.equation_cache[digest]
45
+ else:
46
+ equation_latex = self.equation_decoder.decode(
47
+ payload.data,
48
+ part_name=f"image.{payload.extension}",
49
+ content_type=payload.content_type,
50
+ )
51
+ converted = DocImagePayload(
52
+ payload.data,
53
+ payload.extension,
54
+ payload.content_type,
55
+ equation_latex,
56
+ render_size_emu=payload.render_size_emu,
57
+ )
58
+ if is_new_payload:
59
+ self.total += len(payload.data)
60
+ self.accounted_digests.add(digest)
61
+ self.equation_cache[digest] = equation_latex
62
+ self.cache[cache_key] = converted
63
+ return converted
64
+
65
+
66
+ def inline_picture(
67
+ data_stream: bytes,
68
+ *,
69
+ offset: int,
70
+ store: ImageStore,
71
+ budget: DocBudget,
72
+ ) -> DocImagePayload | None:
73
+ """解析 sprmCPicLocation 指向的 PICFAndOfficeArtData。"""
74
+
75
+ total_length = get_u32(data_stream, offset)
76
+ header_length = get_u16(data_stream, offset + 4)
77
+ if total_length is None or header_length is None or total_length < header_length:
78
+ return None
79
+ picf = bounded_slice(data_stream, offset, total_length)
80
+ if picf is None:
81
+ logger.warning(f"DOC picture at Data offset {offset} is out of bounds")
82
+ return None
83
+ art = picf[min(header_length, len(picf)) :]
84
+ decoded = first_blip(art, charge=budget.charge)
85
+ if decoded is None:
86
+ # 少量旧文件把原始位图直接放在 PICF 尾部,按 magic 尽力保留。
87
+ signatures = (
88
+ (_PLACEABLE_WMF_MAGIC, "wmf", "image/wmf"),
89
+ (b"\x89PNG\r\n\x1a\n", "png", "image/png"),
90
+ (b"\xff\xd8\xff", "jpg", "image/jpeg"),
91
+ (b"GIF8", "gif", "image/gif"),
92
+ (b"BM", "bmp", "image/bmp"),
93
+ (b"II*\x00", "tiff", "image/tiff"),
94
+ (b"MM\x00*", "tiff", "image/tiff"),
95
+ )
96
+ for signature, extension, content_type in signatures:
97
+ position = art.find(signature)
98
+ if position >= 0:
99
+ decoded = OfficeImagePayload(art[position:], extension, content_type)
100
+ break
101
+ return store.add(decoded) if decoded is not None else None
102
+
103
+
104
+ def floating_pictures(
105
+ table_stream: bytes,
106
+ *,
107
+ word_document: bytes,
108
+ shape_plc_offset: int,
109
+ shape_plc_size: int,
110
+ drawing_offset: int,
111
+ drawing_size: int,
112
+ store: ImageStore,
113
+ budget: DocBudget,
114
+ ) -> list[DocImage]:
115
+ """按 PlcfSpaMom anchor CP 将 floating shape 的 BStore 图片绑定到正文。"""
116
+
117
+ plc_payload = bounded_slice(table_stream, shape_plc_offset, shape_plc_size)
118
+ drawing = bounded_slice(table_stream, drawing_offset, drawing_size)
119
+ if plc_payload is None or drawing is None:
120
+ return []
121
+ cps, items = parse_plc(plc_payload, item_size=26, budget=budget)
122
+ if not items:
123
+ return []
124
+ assets = decode_bstore(
125
+ drawing,
126
+ charge=budget.charge,
127
+ delay_stream=word_document,
128
+ )
129
+ shapes = {}
130
+ first = record_at(drawing, 0, charge=budget.charge)
131
+ cursor = (8 + len(first.payload)) if first is not None else len(drawing)
132
+ while cursor < len(drawing):
133
+ # OfficeArtWordDrawing 在 DgContainer 前有一个 main/header 标签字节。
134
+ cursor += 1
135
+ container = record_at(drawing, cursor, charge=budget.charge)
136
+ if container is None:
137
+ break
138
+ for shape in extract_word_shapes(container.payload, charge=budget.charge):
139
+ if shape.shape_id is not None:
140
+ shapes.setdefault(shape.shape_id, shape)
141
+ cursor += 8 + len(container.payload)
142
+ anchors = [
143
+ (shape_id, cps[index])
144
+ for index, item in enumerate(items)
145
+ if index < len(cps) and (shape_id := get_u32(item, 0)) is not None
146
+ ]
147
+ images: list[DocImage] = []
148
+ for shape_id, shape in sorted(shapes.items()):
149
+ if shape.hidden or shape.pib is None:
150
+ continue
151
+ payload = assets.get(shape.pib)
152
+ if payload is None:
153
+ continue
154
+ anchor_cp = next((cp for anchor_id, cp in anchors if anchor_id == shape_id), None)
155
+ if anchor_cp is None:
156
+ candidates = [(anchor_id, cp) for anchor_id, cp in anchors if anchor_id <= shape_id]
157
+ if candidates:
158
+ anchor_cp = max(candidates)[1]
159
+ if anchor_cp is None:
160
+ continue
161
+ images.append(DocImage(cp=anchor_cp, payload=store.add(payload)))
162
+ return images