extract-python 0.8.5__py3-none-any.whl → 0.9.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
extract_python/miner_u.py CHANGED
@@ -1,15 +1,14 @@
1
- import json
2
1
  import os
3
- import shutil
4
2
  from collections.abc import AsyncIterable, Callable, Iterable
5
- from functools import partial
3
+ from functools import cache, partial
6
4
  from pathlib import Path
7
- from tempfile import TemporaryDirectory
8
5
 
6
+ from docvortex.export import materialize_middle, validate_materialized_assets
7
+ from docvortex.render import RenderMode, render_markdown
8
+ from docvortex.schema import DocumentMetadata, PageInfo, Producer
9
9
  from extract_core import (
10
10
  ConversionOutput,
11
11
  InputDoc,
12
- MinerUBackend,
13
12
  MinerUPipelineConfig,
14
13
  OutputFormat,
15
14
  Pipeline,
@@ -17,6 +16,7 @@ from extract_core import (
17
16
  Result,
18
17
  Status,
19
18
  )
19
+ from mineru import MiddleJson, ParseResult
20
20
 
21
21
  from .constants import ARTIFACTS, DEFAULT_MD_PAGE_SEP
22
22
  from .utils import path_to_artifacts_dirname, reset_env, write_pages
@@ -30,185 +30,85 @@ class MinerUPipeline(Pipeline):
30
30
  def __init__(self, config: MinerUPipelineConfig):
31
31
  super().__init__(config)
32
32
  self._language = self._config.language
33
- self._md_make_fn = _parse_md_make_fn(self._config.config.backend)
34
33
 
35
34
  async def extract_content(
36
35
  self, docs: Iterable[InputDoc], output_format: OutputFormat, output_path: Path
37
36
  ) -> AsyncIterable[Result]:
38
- from mineru.cli.common import aio_do_parse # noqa: PLC0415
37
+ from mineru import MinerUParser # noqa: PLC0415
39
38
 
40
39
  with reset_env():
41
40
  os.environ["MINERU_DEVICE_MODE"] = self._device
42
- docs = list(docs)
43
41
  # TODO: exclude files which are not pdf and return an error
44
- pdfs_bytes = [d.path.read_bytes() for d in docs]
45
- pdfs_names = [d.path.name for d in docs]
46
- p_lang_list = [self._language for _ in pdfs_names]
47
42
  # TODO: we should only process valid PDFs
48
- with TemporaryDirectory(prefix="mineru-") as workdir:
49
- workdir = Path(workdir) # noqa: PLW2901
50
- await aio_do_parse(
51
- output_dir=workdir,
52
- pdf_file_names=pdfs_names,
53
- pdf_bytes_list=pdfs_bytes,
54
- p_lang_list=p_lang_list,
55
- **self._config.config.as_parse_kwargs(),
56
- )
57
- res_paths = [
58
- _revert_mineru_output(workdir, pdf_filename=p) for p in pdfs_names
59
- ]
60
- for doc, res_path in zip(docs, res_paths, strict=True):
61
- yield _process_doc(
62
- doc,
63
- md_make_fn=self._md_make_fn,
64
- res_path=res_path,
65
- output_format=output_format,
66
- output_path=output_path,
67
- )
68
-
69
-
70
- def _revert_mineru_output(output_dir: Path, *, pdf_filename: str) -> Path:
71
- output_path = output_dir / pdf_filename
72
- if not output_path.exists():
73
- msg = f"couldn't find result for {pdf_filename}"
74
- raise FileNotFoundError(msg)
75
- dirs = [p for p in output_path.iterdir() if p.is_dir()]
76
- if len(dirs) != 1:
77
- msg = f"expected exactly one result directory, found: {dirs}"
78
- raise ValueError(msg)
79
- return output_dir / dirs[0]
80
-
81
-
82
- def _parse_md_make_fn(backend: MinerUBackend) -> MDMakeFunction:
83
-
84
- match backend:
85
- case MinerUBackend.PIPELINE:
86
- from mineru.backend.pipeline.pipeline_middle_json_mkcontent import ( # noqa: PLC0415
87
- union_make,
43
+ config = self._config.config
44
+ parser = MinerUParser(
45
+ tier=config.tier,
46
+ parse_mode=config.parse_mode,
47
+ image_analysis=config.image_analysis,
48
+ vlm_config=config.vlm_config,
88
49
  )
89
-
90
- return union_make
91
- case MinerUBackend.VLM:
92
- from mineru.backend.vlm.vlm_middle_json_mkcontent import ( # noqa: PLC0415
93
- union_make,
94
- )
95
-
96
- return union_make
97
- case _:
98
- raise ValueError(f"Unsupported backend: {backend}")
50
+ for doc in docs:
51
+ res = await parser.parse_async(doc.path)
52
+ yield _process_doc(
53
+ doc, res, output_format=output_format, output_path=output_path
54
+ )
99
55
 
100
56
 
101
57
  def _process_doc(
102
58
  doc: InputDoc,
59
+ result: ParseResult,
103
60
  *,
104
- md_make_fn: MDMakeFunction,
105
- res_path: Path,
106
61
  output_format: OutputFormat,
107
62
  output_path: Path,
108
63
  ) -> Result:
109
- md_dir_name = path_to_artifacts_dirname(doc.path)
110
- md_dir = Path(output_path) / md_dir_name
111
- md_dir.mkdir(parents=True, exist_ok=False)
112
- artifacts_dir = md_dir / ARTIFACTS
113
- md_path = (md_dir / md_dir_name).with_suffix(OutputFormat.MARKDOWN.value)
64
+ output_path_dir_name = path_to_artifacts_dirname(doc.path)
65
+ output_dir = Path(output_path) / output_path_dir_name
66
+ output_dir.mkdir(parents=True)
114
67
  # Fail early
115
68
  match output_format:
116
69
  case OutputFormat.MARKDOWN:
117
- im_rel_dir = artifacts_dir.relative_to(md_dir)
118
70
  dump_content_fn = partial(
119
- _dump_md_content,
120
- md_make_fn=md_make_fn,
121
- output_path=output_path,
122
- md_path=md_path,
123
- im_dir=im_rel_dir,
71
+ _dump_md_content, output_path=output_path, md_path=output_dir
124
72
  )
125
73
  case _:
126
74
  raise NotImplementedError(f"unsupported output format {output_format}")
127
- middle_json_path = res_path / f"{doc.path.name}_middle.json"
128
- middle_json = json.loads(middle_json_path.read_text())
129
- shutil.move(res_path / "images", artifacts_dir)
130
- output = dump_content_fn(middle_json)
75
+ output = dump_content_fn(result.middle_json)
131
76
  return Result(input=doc, status=Status.SUCCESS, output=output)
132
77
 
133
78
 
79
+ @cache
80
+ def _mineru_md_page_sep() -> str:
81
+ dummy_json = MiddleJson(
82
+ pages=[PageInfo(page_idx=0), PageInfo(page_idx=1)],
83
+ is_full_document=True,
84
+ metadata=DocumentMetadata(
85
+ file_suffix="pdf", producer=Producer(name="dummy", version="1.0")
86
+ ),
87
+ )
88
+ md = render_markdown(dummy_json, mode=RenderMode.FULL)
89
+ return md.replace(" ", "")
90
+
91
+
134
92
  def _dump_md_content(
135
- middle_json: dict,
93
+ middle_json: MiddleJson,
136
94
  *,
137
- md_make_fn: MDMakeFunction,
138
95
  page_sep: str = DEFAULT_MD_PAGE_SEP,
139
96
  output_path: Path,
140
97
  md_path: Path,
141
- im_dir: Path,
142
- md_make_mode: str | None = None,
143
98
  ) -> ConversionOutput:
144
- from mineru.utils.enum_class import MakeMode # noqa: PLC0415
145
-
146
- pdf_info = middle_json["pdf_info"]
147
- if md_make_mode is None:
148
- md_make_mode = MakeMode.MM_MD
149
- pages = (md_make_fn([p], md_make_mode, str(im_dir)) for p in pdf_info)
150
- with md_path.open("wb") as f:
99
+ middle_json, assets = materialize_middle(middle_json)
100
+ exported = ParseResult(middle_json=middle_json)
101
+ validate_materialized_assets(middle_json, assets)
102
+ md = exported.markdown(mode=RenderMode.FULL)
103
+ for path, im_bytes in assets.items():
104
+ im_path = md_path / ARTIFACTS / path
105
+ im_path.parent.mkdir(parents=True, exist_ok=True)
106
+ im_path.write_bytes(im_bytes)
107
+ md = md.replace(path, str(im_path.relative_to(md_path)))
108
+ pages = md.split(_mineru_md_page_sep())
109
+ md_content_path = (md_path / md_path.name).with_suffix(OutputFormat.MARKDOWN)
110
+ with open(md_content_path, "wb") as f:
151
111
  pages = write_pages(pages, page_sep, f)
152
- output_path = md_path.parent.relative_to(output_path)
153
- confidence = _mineru_confidence(pdf_info)
154
- output = ConversionOutput(path=output_path, pages=pages, confidence=confidence)
112
+ path = md_path.relative_to(output_path)
113
+ output = ConversionOutput(path=path, pages=pages, confidence=None)
155
114
  return output
156
-
157
-
158
- def _mineru_confidence(pdf_info: list[dict]) -> float:
159
- if not pdf_info:
160
- return 1.0
161
- block_conf = _mineru_block_confidence(pdf_info)
162
- line_config = _mineru_line_confidence(pdf_info)
163
- return (block_conf + line_config) / 2.0
164
-
165
-
166
- def _mineru_block_confidence(pdf_info: list[dict]) -> float:
167
- import numpy as np # noqa: PLC0415
168
-
169
- scores = []
170
- for info in pdf_info:
171
- for block in info["para_blocks"]:
172
- score = block.get("score")
173
- if score is not None:
174
- scores.append(score)
175
- if scores:
176
- return np.average(scores)
177
- return 1.0
178
-
179
-
180
- def _mineru_line_confidence(pdf_info: list[dict]) -> float:
181
- import numpy as np # noqa: PLC0415
182
-
183
- scores = []
184
- lengths = []
185
- for info in pdf_info:
186
- for block in info["para_blocks"]:
187
- for line in block.get("lines", []):
188
- for span in line["spans"]:
189
- score = span.get("score")
190
- if score is not None:
191
- scores.append(score)
192
- lengths.append(len(span["content"]))
193
- if scores:
194
- return np.average(scores, weights=lengths)
195
- return 1.0
196
-
197
-
198
- def _parse_block(block: dict) -> tuple[list[float], list[float]]:
199
- if "lines" in block:
200
- scores = []
201
- lengths = []
202
- for line in block.get("lines", []):
203
- for span in line["spans"]:
204
- score = span.get("score")
205
- if score is not None:
206
- scores.append(score)
207
- lengths.append(len(span["content"]))
208
- return scores, lengths
209
- if "blocs" in block:
210
- scores, lengths = (_parse_block(b) for b in block["blocs"])
211
- scores = sum(*scores, start=[])
212
- lengths = sum(*lengths, start=[])
213
- return scores, lengths
214
- raise NotImplementedError(f"unsupported block: {block}")
@@ -1,13 +1,13 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: extract-python
3
- Version: 0.8.5
3
+ Version: 0.9.0
4
4
  Summary: Structured content extraction
5
5
  Project-URL: Homepage, https://github.com/ICIJ/extract-python
6
6
  Project-URL: Repository, https://github.com/ICIJ/extract-python
7
7
  Project-URL: Issues, https://github.com/ICIJ/extract-python/issues
8
8
  Author-email: Clément Doumouro <cdoumouro@icij.org>
9
9
  Requires-Python: <3.15,>=3.13
10
- Requires-Dist: extract-core~=0.8.0
10
+ Requires-Dist: extract-core~=0.9.0
11
11
  Requires-Dist: icij-common~=0.8.2
12
12
  Requires-Dist: pympler~=1.1
13
13
  Provides-Extra: benches
@@ -20,8 +20,7 @@ Requires-Dist: docling-slim[feat-ocr-easyocr,feat-ocr-mac,feat-ocr-tesserocr,sta
20
20
  Provides-Extra: marker
21
21
  Requires-Dist: marker-pdf~=1.10; extra == 'marker'
22
22
  Provides-Extra: mineru
23
- Requires-Dist: mineru[mlx]~=3.2; (sys_platform == 'darwin') and extra == 'mineru'
24
- Requires-Dist: mineru[pipeline,vlm]~=3.2; extra == 'mineru'
23
+ Requires-Dist: mineru[torch,vlm]~=4.0; extra == 'mineru'
25
24
  Requires-Dist: pydantic-extra-types[pycountry]~=2.11; extra == 'mineru'
26
25
  Requires-Dist: python-pptx~=1.0; extra == 'mineru'
27
26
  Requires-Dist: six~=1.17; extra == 'mineru'
@@ -2,8 +2,8 @@ extract_python/__init__.py,sha256=DA2LUro6vMjfS8fb2MsqO95FbJEZHyZ7kFyn42q02Wk,75
2
2
  extract_python/constants.py,sha256=659V40LcTWJhX3IbuJLSSvI5AsGJh9ciMrGCfzJn2zA,98
3
3
  extract_python/docling_.py,sha256=H_A-6pMObQ0Q7XA62GCBv0nbGtW5rKjfOVW3giU4NCk,10146
4
4
  extract_python/marker_.py,sha256=bb0DJXrSWOznIuPnw4-adR5_R9QiwtFGGBh5KlXFxjM,3323
5
- extract_python/miner_u.py,sha256=OCIleNu-VvaJlx_zG0VuagJvQvElXV7e-aPXqHbp31A,7302
5
+ extract_python/miner_u.py,sha256=sEU1PhPpfPVo4jP7flNlbGO9PfsLFLz973-ufZEd0ws,3966
6
6
  extract_python/utils.py,sha256=LrR1tlOgpWEri6TLDlYYPRBmlka-AjOepWnbySf2Haw,9974
7
- extract_python-0.8.5.dist-info/METADATA,sha256=bdSC_ZpYSwJaPx_th2q2VWgTJwfi7pzPGHhQJPJgSeE,1290
8
- extract_python-0.8.5.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
9
- extract_python-0.8.5.dist-info/RECORD,,
7
+ extract_python-0.9.0.dist-info/METADATA,sha256=ChieXoOAjl4y-ycPKQCKPvlVqx7ukeogABYdChcCuSs,1205
8
+ extract_python-0.9.0.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
9
+ extract_python-0.9.0.dist-info/RECORD,,
@@ -1,4 +1,4 @@
1
1
  Wheel-Version: 1.0
2
- Generator: hatchling 1.32.0
2
+ Generator: hatchling 1.32.4
3
3
  Root-Is-Purelib: true
4
4
  Tag: py3-none-any