extract-python 0.8.4__tar.gz → 0.9.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,13 +1,13 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: extract-python
3
- Version: 0.8.4
3
+ Version: 0.9.0
4
4
  Summary: Structured content extraction
5
5
  Project-URL: Homepage, https://github.com/ICIJ/extract-python
6
6
  Project-URL: Repository, https://github.com/ICIJ/extract-python
7
7
  Project-URL: Issues, https://github.com/ICIJ/extract-python/issues
8
8
  Author-email: Clément Doumouro <cdoumouro@icij.org>
9
9
  Requires-Python: <3.15,>=3.13
10
- Requires-Dist: extract-core~=0.8.0
10
+ Requires-Dist: extract-core~=0.9.0
11
11
  Requires-Dist: icij-common~=0.8.2
12
12
  Requires-Dist: pympler~=1.1
13
13
  Provides-Extra: benches
@@ -20,8 +20,7 @@ Requires-Dist: docling-slim[feat-ocr-easyocr,feat-ocr-mac,feat-ocr-tesserocr,sta
20
20
  Provides-Extra: marker
21
21
  Requires-Dist: marker-pdf~=1.10; extra == 'marker'
22
22
  Provides-Extra: mineru
23
- Requires-Dist: mineru[mlx]~=3.2; (sys_platform == 'darwin') and extra == 'mineru'
24
- Requires-Dist: mineru[pipeline,vlm]~=3.2; extra == 'mineru'
23
+ Requires-Dist: mineru[torch,vlm]~=4.0; extra == 'mineru'
25
24
  Requires-Dist: pydantic-extra-types[pycountry]~=2.11; extra == 'mineru'
26
25
  Requires-Dist: python-pptx~=1.0; extra == 'mineru'
27
26
  Requires-Dist: six~=1.17; extra == 'mineru'
@@ -0,0 +1,114 @@
1
+ import os
2
+ from collections.abc import AsyncIterable, Callable, Iterable
3
+ from functools import cache, partial
4
+ from pathlib import Path
5
+
6
+ from docvortex.export import materialize_middle, validate_materialized_assets
7
+ from docvortex.render import RenderMode, render_markdown
8
+ from docvortex.schema import DocumentMetadata, PageInfo, Producer
9
+ from extract_core import (
10
+ ConversionOutput,
11
+ InputDoc,
12
+ MinerUPipelineConfig,
13
+ OutputFormat,
14
+ Pipeline,
15
+ PipelineType,
16
+ Result,
17
+ Status,
18
+ )
19
+ from mineru import MiddleJson, ParseResult
20
+
21
+ from .constants import ARTIFACTS, DEFAULT_MD_PAGE_SEP
22
+ from .utils import path_to_artifacts_dirname, reset_env, write_pages
23
+
24
+ _MINER_U_CONVERSION_ERRORS = tuple()
25
+ MDMakeFunction = Callable[[list, str, str], str | None]
26
+
27
+
28
+ @Pipeline.register(PipelineType.MINER_U)
29
+ class MinerUPipeline(Pipeline):
30
+ def __init__(self, config: MinerUPipelineConfig):
31
+ super().__init__(config)
32
+ self._language = self._config.language
33
+
34
+ async def extract_content(
35
+ self, docs: Iterable[InputDoc], output_format: OutputFormat, output_path: Path
36
+ ) -> AsyncIterable[Result]:
37
+ from mineru import MinerUParser # noqa: PLC0415
38
+
39
+ with reset_env():
40
+ os.environ["MINERU_DEVICE_MODE"] = self._device
41
+ # TODO: exclude files which are not pdf and return an error
42
+ # TODO: we should only process valid PDFs
43
+ config = self._config.config
44
+ parser = MinerUParser(
45
+ tier=config.tier,
46
+ parse_mode=config.parse_mode,
47
+ image_analysis=config.image_analysis,
48
+ vlm_config=config.vlm_config,
49
+ )
50
+ for doc in docs:
51
+ res = await parser.parse_async(doc.path)
52
+ yield _process_doc(
53
+ doc, res, output_format=output_format, output_path=output_path
54
+ )
55
+
56
+
57
+ def _process_doc(
58
+ doc: InputDoc,
59
+ result: ParseResult,
60
+ *,
61
+ output_format: OutputFormat,
62
+ output_path: Path,
63
+ ) -> Result:
64
+ output_path_dir_name = path_to_artifacts_dirname(doc.path)
65
+ output_dir = Path(output_path) / output_path_dir_name
66
+ output_dir.mkdir(parents=True)
67
+ # Fail early
68
+ match output_format:
69
+ case OutputFormat.MARKDOWN:
70
+ dump_content_fn = partial(
71
+ _dump_md_content, output_path=output_path, md_path=output_dir
72
+ )
73
+ case _:
74
+ raise NotImplementedError(f"unsupported output format {output_format}")
75
+ output = dump_content_fn(result.middle_json)
76
+ return Result(input=doc, status=Status.SUCCESS, output=output)
77
+
78
+
79
+ @cache
80
+ def _mineru_md_page_sep() -> str:
81
+ dummy_json = MiddleJson(
82
+ pages=[PageInfo(page_idx=0), PageInfo(page_idx=1)],
83
+ is_full_document=True,
84
+ metadata=DocumentMetadata(
85
+ file_suffix="pdf", producer=Producer(name="dummy", version="1.0")
86
+ ),
87
+ )
88
+ md = render_markdown(dummy_json, mode=RenderMode.FULL)
89
+ return md.replace(" ", "")
90
+
91
+
92
+ def _dump_md_content(
93
+ middle_json: MiddleJson,
94
+ *,
95
+ page_sep: str = DEFAULT_MD_PAGE_SEP,
96
+ output_path: Path,
97
+ md_path: Path,
98
+ ) -> ConversionOutput:
99
+ middle_json, assets = materialize_middle(middle_json)
100
+ exported = ParseResult(middle_json=middle_json)
101
+ validate_materialized_assets(middle_json, assets)
102
+ md = exported.markdown(mode=RenderMode.FULL)
103
+ for path, im_bytes in assets.items():
104
+ im_path = md_path / ARTIFACTS / path
105
+ im_path.parent.mkdir(parents=True, exist_ok=True)
106
+ im_path.write_bytes(im_bytes)
107
+ md = md.replace(path, str(im_path.relative_to(md_path)))
108
+ pages = md.split(_mineru_md_page_sep())
109
+ md_content_path = (md_path / md_path.name).with_suffix(OutputFormat.MARKDOWN)
110
+ with open(md_content_path, "wb") as f:
111
+ pages = write_pages(pages, page_sep, f)
112
+ path = md_path.relative_to(output_path)
113
+ output = ConversionOutput(path=path, pages=pages, confidence=None)
114
+ return output
@@ -8,7 +8,7 @@ authors = [
8
8
  readme = "README.md"
9
9
  requires-python = ">=3.13,<3.15"
10
10
  dependencies = [
11
- "extract-core~=0.8.0",
11
+ "extract-core~=0.9.0",
12
12
  "icij-common~=0.8.2",
13
13
  "pympler~=1.1",
14
14
  ]
@@ -30,8 +30,7 @@ marker = [
30
30
  ]
31
31
 
32
32
  mineru = [
33
- "mineru[pipeline,vlm]~=3.2",
34
- "mineru[mlx]~=3.2; sys_platform == 'darwin'",
33
+ "mineru[vlm,torch]~=4.0",
35
34
  "pydantic-extra-types[pycountry]~=2.11",
36
35
  "python-pptx~=1.0",
37
36
  "six~=1.17",