extract-python 0.7.3__tar.gz → 0.8.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,14 +1,15 @@
1
- Metadata-Version: 2.4
1
+ Metadata-Version: 2.5
2
2
  Name: extract-python
3
- Version: 0.7.3
3
+ Version: 0.8.1
4
4
  Summary: Structured content extraction
5
5
  Project-URL: Homepage, https://github.com/ICIJ/extract-python
6
6
  Project-URL: Repository, https://github.com/ICIJ/extract-python
7
7
  Project-URL: Issues, https://github.com/ICIJ/extract-python/issues
8
8
  Author-email: Clément Doumouro <cdoumouro@icij.org>
9
9
  Requires-Python: <3.15,>=3.13
10
- Requires-Dist: extract-core~=0.7.0
10
+ Requires-Dist: extract-core~=0.8.0
11
11
  Requires-Dist: icij-common~=0.8.2
12
+ Requires-Dist: pympler~=1.1
12
13
  Provides-Extra: benches
13
14
  Requires-Dist: html2image~=2.0.7; extra == 'benches'
14
15
  Requires-Dist: markdown2>=2.5.4; extra == 'benches'
@@ -24,3 +25,6 @@ Requires-Dist: mineru[pipeline,vlm]~=3.2; extra == 'mineru'
24
25
  Requires-Dist: pydantic-extra-types[pycountry]~=2.11; extra == 'mineru'
25
26
  Requires-Dist: python-pptx~=1.0; extra == 'mineru'
26
27
  Requires-Dist: six~=1.17; extra == 'mineru'
28
+ Description-Content-Type: text/markdown
29
+
30
+ ô
@@ -0,0 +1 @@
1
+ ô
@@ -1,20 +1,27 @@
1
1
  import asyncio
2
2
  import json
3
3
  import logging
4
+ import operator
4
5
  import shutil
5
6
  import tempfile
6
- from collections.abc import AsyncGenerator, Iterable, Iterator
7
- from functools import partial
7
+ from collections.abc import AsyncIterable, Iterable
8
+ from contextlib import AbstractContextManager
9
+ from functools import partial, reduce
8
10
  from pathlib import Path
9
11
  from typing import Any, Self
10
12
 
11
- from docling.datamodel.document import ConversionResult
13
+ from docling.datamodel.document import ConversionAssets
12
14
  from docling.datamodel.pipeline_options import PipelineOptions
15
+ from docling.datamodel.settings import (
16
+ DEFAULT_PAGE_RANGE,
17
+ AppSettings,
18
+ scoped,
19
+ )
13
20
  from docling.document_converter import DocumentConverter, FormatOption
21
+ from docling_core.types import DoclingDocument
14
22
 
15
23
  # TODO: this is long to load improve it
16
24
  from docling_core.types.doc import ImageRefMode
17
- from docling_core.types.io import DocumentStream
18
25
  from extract_core import (
19
26
  BaseModel,
20
27
  DoclingFormatOption,
@@ -33,7 +40,14 @@ from pydantic import ConfigDict, field_serializer
33
40
  from pydantic_core.core_schema import SerializerFunctionWrapHandler
34
41
 
35
42
  from .constants import ARTIFACTS, DEFAULT_MD_PAGE_SEP
36
- from .utils import chdir, map_and_preserve, path_to_artifacts_dirname, write_pages
43
+ from .utils import (
44
+ Range,
45
+ ResultBuffer,
46
+ batch_per_pages,
47
+ chdir,
48
+ path_to_artifacts_dirname,
49
+ write_pages,
50
+ )
37
51
 
38
52
  logger = logging.getLogger(__name__)
39
53
 
@@ -61,54 +75,112 @@ class DoclingPipeline(Pipeline):
61
75
 
62
76
  async def extract_content(
63
77
  self, docs: Iterable[InputDoc], output_format: OutputFormat, output_path: Path
64
- ) -> AsyncGenerator[Result, None]:
65
- docs, path_or_streams = map_and_preserve(_to_docling, docs)
66
- outputs = self._converter.convert_all(path_or_streams, raises_on_error=False)
78
+ ) -> AsyncIterable[Result]:
79
+ settings = self._config.settings
80
+ logger.info("starting extraction with settings: %s", settings)
81
+ with self._scoped_settings, self._result_buffer as buffer:
82
+ max_page_batches = settings.perf.max_page_batches
83
+ page_batch_size = settings.perf.page_batch_size
84
+ batches = batch_per_pages(
85
+ docs, page_batch_size, max_page_batches=max_page_batches
86
+ )
87
+ for batch in batches:
88
+ page_range = list({pages.page_range for pages in batch})
89
+ if len(page_range) > 1:
90
+ msg = "convert_all only accept 1 page range for all docs"
91
+ raise ValueError(msg)
92
+ page_range = page_range[0]
93
+ page_range = _docling_range(page_range)
94
+ docling_docs = (pages.doc.to_docling() for pages in batch)
95
+ outputs = self._converter.convert_all(
96
+ docling_docs, raises_on_error=False, page_range=page_range
97
+ )
98
+ processed = iter(batch)
99
+ sentinel = object()
100
+ while True:
101
+ res = await asyncio.to_thread(next, outputs, sentinel)
102
+ if res is sentinel:
103
+ break
104
+ pages = next(processed)
105
+ buffer.add(pages, res)
106
+ if buffer.is_complete(pages.doc_idx):
107
+ doc_pages = buffer.pop_complete(pages.doc_idx)
108
+ yield _to_result(
109
+ doc_pages, pages.doc, output_format, output_path=output_path
110
+ )
111
+
112
+ @property
113
+ def _scoped_settings(self) -> AbstractContextManager[AppSettings]:
114
+ settings = self._config.settings
115
+ docling_settings = scoped(
116
+ perf=settings.perf, debug=settings.debug, inference=settings.inference
117
+ )
118
+ return docling_settings
119
+
120
+ @property
121
+ def _result_buffer(self) -> ResultBuffer:
122
+ buffer = ResultBuffer(
123
+ max_size_bytes=self._config.result_buffer.max_size,
124
+ root=self._config.result_buffer.root,
125
+ save_fn=_save_conversion_result,
126
+ load_fn=_load_conversion_result,
127
+ )
128
+ return buffer
129
+
67
130
 
68
- sentinel = object()
69
- while True:
70
- res = await asyncio.to_thread(next, outputs, sentinel)
71
- if res is sentinel:
72
- return
73
- doc = next(docs)
74
- yield _to_result(res, doc, output_format, output_path=output_path)
131
+ def _save_conversion_result(res: ConversionAssets, path: Path) -> None:
132
+ return res.save(filename=path)
75
133
 
76
134
 
77
- def _to_docling(docs: Iterable[InputDoc]) -> Iterator["Path | DocumentStream"]:
78
- for d in docs:
79
- yield d.to_docling()
135
+ def _load_conversion_result(path: Path) -> ConversionAssets:
136
+ return ConversionAssets.load(path)
80
137
 
81
138
 
82
139
  def _to_result(
83
- res: ConversionResult,
84
- input_document: InputDoc,
140
+ buffer: list[ConversionAssets],
141
+ input_doc: InputDoc,
85
142
  output_format: OutputFormat,
86
143
  output_path: Path,
87
144
  **kwargs,
88
145
  ) -> Result:
146
+ import numpy as np # noqa: PLC0415
147
+
148
+ if not buffer:
149
+ raise ValueError("empty buffer")
150
+ merged = DoclingDocument.concatenate([res.document for res in buffer])
89
151
  output_path.mkdir(parents=True, exist_ok=True)
152
+ status = reduce(operator.iadd, (Status.from_docling(d.status) for d in buffer))
153
+ # TODO: implement confidence weight
154
+ confidence = np.mean([res.confidence.mean_score for res in buffer])
90
155
  output = None
91
- status = Status.from_docling(res.status)
92
156
  if status.allows_conversion:
93
157
  match output_format:
94
158
  case OutputFormat.MARKDOWN:
95
- output = _to_markdown_doc(res, output_path, **kwargs)
159
+ output = _to_markdown_doc(
160
+ merged,
161
+ input_path=input_doc.path,
162
+ output_path=output_path,
163
+ confidence=confidence,
164
+ **kwargs,
165
+ )
96
166
  case _:
97
167
  raise NotImplementedError(f"unsupported output format {output_format}")
98
- errors = [Error.from_docling(e) for e in res.errors]
99
- input_doc = input_document.without_content()
168
+ errors = [Error.from_docling(e) for res in buffer for e in res.errors]
100
169
  return Result(input=input_doc, status=status, errors=errors, output=output)
101
170
 
102
171
 
103
172
  def _to_markdown_doc(
104
- res: ConversionResult,
173
+ doc: DoclingDocument,
174
+ input_path: Path,
175
+ *,
105
176
  output_path: Path,
106
177
  page_sep: str = DEFAULT_MD_PAGE_SEP,
178
+ confidence: float,
107
179
  **kwargs,
108
180
  ) -> MarkdownDoc:
109
181
  # TODO: Should we add a hash to avoid collision between files with same names
110
182
  # nested in the tree structured
111
- md_dir_name = path_to_artifacts_dirname(res.input.file)
183
+ md_dir_name = path_to_artifacts_dirname(input_path)
112
184
  md_dir = output_path / md_dir_name
113
185
  if md_dir.exists():
114
186
  raise FileExistsError(f"directory {md_dir} already exists")
@@ -121,21 +193,21 @@ def _to_markdown_doc(
121
193
  with chdir(tmp_dir):
122
194
  # We do a chdir to bypass a Docling bug which only allows to maintain
123
195
  # relative image ref when saving the markdown to a relative path
124
- pages = _docling_pages_it(res, current_page_path, **kwargs)
196
+ pages = _docling_pages_it(doc, current_page_path, **kwargs)
125
197
  with md_path.open("wb") as f:
126
198
  pages = write_pages(pages, page_sep, f)
127
199
  # Clean up the tmp page file before move everything to the end destination
128
200
  current_page_path.unlink(missing_ok=True)
129
201
  shutil.move(tmp_dir, md_dir)
130
- return MarkdownDoc(path=Path(md_dir_name), pages=pages)
202
+ return MarkdownDoc(path=Path(md_dir_name), pages=pages, confidence=confidence)
131
203
 
132
204
 
133
205
  def _docling_pages_it(
134
- res: ConversionResult, output_path: Path, **kwargs
206
+ doc: DoclingDocument, output_path: Path, **kwargs
135
207
  ) -> Iterable[str]:
136
- n_pages = len(res.pages)
208
+ n_pages = len(doc.pages)
137
209
  for page_i in range(n_pages):
138
- res.document.save_as_markdown(
210
+ doc.save_as_markdown(
139
211
  output_path,
140
212
  page_no=page_i + 1,
141
213
  image_mode=ImageRefMode.REFERENCED,
@@ -193,3 +265,9 @@ class SerializableFormatOptions(DoclingFormatOption):
193
265
  serialized["table_structure_options"] = dict()
194
266
  serialized["table_structure_options"]["kind"] = table_structure_opts.kind
195
267
  return serialized
268
+
269
+
270
+ def _docling_range(rng: Range | None) -> tuple[int, int]:
271
+ if rng is None:
272
+ return DEFAULT_PAGE_RANGE
273
+ return (rng[0] + 1, rng[1] + 1)
@@ -1,6 +1,6 @@
1
1
  import asyncio
2
2
  import gc
3
- from collections.abc import AsyncGenerator, Iterable
3
+ from collections.abc import AsyncIterable, Iterable
4
4
  from copy import deepcopy
5
5
  from pathlib import Path
6
6
  from typing import TYPE_CHECKING
@@ -30,7 +30,7 @@ _MARKER_CONVERSION_ERRORS = tuple()
30
30
  class MarkerPipeline(Pipeline):
31
31
  async def extract_content(
32
32
  self, docs: Iterable[InputDoc], output_format: OutputFormat, output_path: Path
33
- ) -> AsyncGenerator[Result, None]:
33
+ ) -> AsyncIterable[Result]:
34
34
  from marker.config.parser import ConfigParser # noqa: PLC0415
35
35
  from marker.converters.pdf import PdfConverter # noqa: PLC0415
36
36
  from marker.models import create_model_dict # noqa: PLC0415
@@ -67,8 +67,7 @@ async def _process_doc(
67
67
  )
68
68
  case _:
69
69
  raise NotImplementedError(f"unsupported output format {output_format}")
70
- input_doc = doc.without_content()
71
- return Result(input=input_doc, status=Status.SUCCESS, output=output)
70
+ return Result(input=doc, status=Status.SUCCESS, output=output)
72
71
 
73
72
 
74
73
  def _to_markdown_doc(
@@ -96,4 +95,4 @@ def _to_markdown_doc(
96
95
  md_path = md_path.with_suffix(OutputFormat.MARKDOWN.value)
97
96
  with md_path.open("wb") as f:
98
97
  pages = write_pages(pages, page_sep, f)
99
- return MarkdownDoc(path=Path(md_dir_name), pages=pages)
98
+ return MarkdownDoc(path=Path(md_dir_name), pages=pages, confidence=None)
@@ -1,7 +1,7 @@
1
1
  import json
2
2
  import os
3
3
  import shutil
4
- from collections.abc import AsyncGenerator, Callable, Iterable
4
+ from collections.abc import AsyncIterable, Callable, Iterable
5
5
  from functools import partial
6
6
  from pathlib import Path
7
7
  from tempfile import TemporaryDirectory
@@ -34,7 +34,7 @@ class MinerUPipeline(Pipeline):
34
34
 
35
35
  async def extract_content(
36
36
  self, docs: Iterable[InputDoc], output_format: OutputFormat, output_path: Path
37
- ) -> AsyncGenerator[Result, None]:
37
+ ) -> AsyncIterable[Result]:
38
38
  from mineru.cli.common import aio_do_parse # noqa: PLC0415
39
39
 
40
40
  with reset_env():
@@ -126,15 +126,13 @@ def _process_doc(
126
126
  raise NotImplementedError(f"unsupported output format {output_format}")
127
127
  middle_json_path = res_path / f"{doc.path.name}_middle.json"
128
128
  middle_json = json.loads(middle_json_path.read_text())
129
- pdf_info = middle_json["pdf_info"]
130
129
  shutil.move(res_path / "images", artifacts_dir)
131
- output = dump_content_fn(pdf_info)
132
- input_doc = doc.without_content()
133
- return Result(input=input_doc, status=Status.SUCCESS, output=output)
130
+ output = dump_content_fn(middle_json)
131
+ return Result(input=doc, status=Status.SUCCESS, output=output)
134
132
 
135
133
 
136
134
  def _dump_md_content(
137
- pdf_info: list[dict],
135
+ middle_json: dict,
138
136
  *,
139
137
  md_make_fn: MDMakeFunction,
140
138
  page_sep: str = DEFAULT_MD_PAGE_SEP,
@@ -145,11 +143,72 @@ def _dump_md_content(
145
143
  ) -> ConversionOutput:
146
144
  from mineru.utils.enum_class import MakeMode # noqa: PLC0415
147
145
 
146
+ pdf_info = middle_json["pdf_info"]
148
147
  if md_make_mode is None:
149
148
  md_make_mode = MakeMode.MM_MD
150
149
  pages = (md_make_fn([p], md_make_mode, str(im_dir)) for p in pdf_info)
151
150
  with md_path.open("wb") as f:
152
151
  pages = write_pages(pages, page_sep, f)
153
152
  output_path = md_path.parent.relative_to(output_path)
154
- output = ConversionOutput(path=output_path, pages=pages)
153
+ confidence = _mineru_confidence(pdf_info)
154
+ output = ConversionOutput(path=output_path, pages=pages, confidence=confidence)
155
155
  return output
156
+
157
+
158
+ def _mineru_confidence(pdf_info: list[dict]) -> float:
159
+ if not pdf_info:
160
+ return 1.0
161
+ block_conf = _mineru_block_confidence(pdf_info)
162
+ line_config = _mineru_line_confidence(pdf_info)
163
+ return (block_conf + line_config) / 2.0
164
+
165
+
166
+ def _mineru_block_confidence(pdf_info: list[dict]) -> float:
167
+ import numpy as np # noqa: PLC0415
168
+
169
+ scores = []
170
+ for info in pdf_info:
171
+ for block in info["para_blocks"]:
172
+ score = block.get("score")
173
+ if score is not None:
174
+ scores.append(score)
175
+ if scores:
176
+ return np.average(scores)
177
+ return 1.0
178
+
179
+
180
+ def _mineru_line_confidence(pdf_info: list[dict]) -> float:
181
+ import numpy as np # noqa: PLC0415
182
+
183
+ scores = []
184
+ lengths = []
185
+ for info in pdf_info:
186
+ for block in info["para_blocks"]:
187
+ for line in block.get("lines", []):
188
+ for span in line["spans"]:
189
+ score = span.get("score")
190
+ if score is not None:
191
+ scores.append(score)
192
+ lengths.append(len(span["content"]))
193
+ if scores:
194
+ return np.average(scores, weights=lengths)
195
+ return 1.0
196
+
197
+
198
+ def _parse_block(block: dict) -> tuple[list[float], list[float]]:
199
+ if "lines" in block:
200
+ scores = []
201
+ lengths = []
202
+ for line in block.get("lines", []):
203
+ for span in line["spans"]:
204
+ score = span.get("score")
205
+ if score is not None:
206
+ scores.append(score)
207
+ lengths.append(len(span["content"]))
208
+ return scores, lengths
209
+ if "blocs" in block:
210
+ scores, lengths = (_parse_block(b) for b in block["blocs"])
211
+ scores = sum(*scores, start=[])
212
+ lengths = sum(*lengths, start=[])
213
+ return scores, lengths
214
+ raise NotImplementedError(f"unsupported block: {block}")
@@ -0,0 +1,299 @@
1
+ import gc
2
+ import itertools
3
+ import logging
4
+ import os
5
+ import shutil
6
+ import uuid
7
+ from collections import defaultdict, deque
8
+ from collections.abc import Callable, Generator, Iterable, Iterator
9
+ from contextlib import contextmanager
10
+ from copy import copy
11
+ from dataclasses import dataclass
12
+ from functools import wraps
13
+ from itertools import tee
14
+ from pathlib import Path, PurePath
15
+ from tempfile import TemporaryDirectory
16
+ from types import TracebackType
17
+ from typing import BinaryIO, Protocol, Self
18
+
19
+ from extract_core import Error, InputDoc, Pages, Result, Status
20
+ from pympler import asizeof
21
+
22
+ logger = logging.getLogger(__name__)
23
+
24
+
25
+ def map_and_preserve[I, R](
26
+ fn: Callable[[Iterable[I]], Iterator[R]], inputs: Iterable[I]
27
+ ) -> tuple[Iterable[I], Iterator[R]]:
28
+ save_inputs, function_inputs = tee(inputs)
29
+ outputs = iter(fn(function_inputs))
30
+ return save_inputs, outputs
31
+
32
+
33
+ def path_to_artifacts_dirname(path: PurePath, sep: str = "_") -> str:
34
+ dirname = f"{path.name[: -len(path.suffix)]}"
35
+ ext = path.suffix
36
+ if ext:
37
+ dirname += sep + ext[1:]
38
+ return dirname
39
+
40
+
41
+ class DocProcessingFn(Protocol):
42
+ def __call__(self, doc: InputDoc, *arg, **kwargs) -> Result: ...
43
+
44
+
45
+ def report_recoverable_errors(
46
+ recoverable_errors: tuple[type[Exception], ...] = tuple(),
47
+ ) -> Callable[[DocProcessingFn], DocProcessingFn]:
48
+ def make_decorator(f: DocProcessingFn) -> DocProcessingFn:
49
+ @wraps(f)
50
+ def wrapped(doc: InputDoc, *args, **kwargs) -> Result:
51
+ try:
52
+ return f(doc, *args, **kwargs)
53
+ except recoverable_errors as e:
54
+ error = Error.from_exception(e)
55
+ return Result(
56
+ input=doc, status=Status.FAILURE, errors=[error], output=None
57
+ )
58
+
59
+ return wrapped
60
+
61
+ return make_decorator
62
+
63
+
64
+ @contextmanager
65
+ def chdir(path: Path) -> Generator[None]:
66
+ cwd = Path.cwd()
67
+ try:
68
+ os.chdir(path)
69
+ yield
70
+ finally:
71
+ os.chdir(cwd)
72
+
73
+
74
+ @contextmanager
75
+ def reset_env() -> Generator[None]:
76
+ old_env = copy(dict(os.environ))
77
+ try:
78
+ yield
79
+ finally:
80
+ os.environ.clear()
81
+ os.environ.update(old_env)
82
+
83
+
84
+ def write_pages(pages: Iterable[str], page_sep: str, out: BinaryIO) -> Pages:
85
+ pages_byte_sizes = []
86
+ pages = iter(pages)
87
+ content = None
88
+ for p in pages:
89
+ if content:
90
+ pages_byte_sizes.append(out.write((content + page_sep).encode()))
91
+ content = p
92
+ if content:
93
+ pages_byte_sizes.append(out.write(content.encode()))
94
+ return Pages.from_pages_bytes_sizes(pages_byte_sizes)
95
+
96
+
97
+ Range = tuple[int, int]
98
+
99
+
100
+ @dataclass(frozen=True)
101
+ class ProcessedPages:
102
+ doc: InputDoc
103
+ doc_idx: int
104
+ page_range: Range | None = None
105
+
106
+ @property
107
+ def page_length(self) -> int:
108
+ if self.page_range is None:
109
+ return self.doc.n_pages
110
+ return self.page_range[1] - self.page_range[0]
111
+
112
+
113
+ class ResultBuffer[R]:
114
+ def __init__(
115
+ self,
116
+ max_size_bytes: int,
117
+ save_fn: Callable[[R, Path], None],
118
+ *,
119
+ load_fn: Callable[[Path], R],
120
+ root: Path | None = None,
121
+ ):
122
+ self._max_bytes = max_size_bytes
123
+ self._save_fn = save_fn
124
+ self._load_fn = load_fn
125
+ self._tmp_dir = None
126
+ if root is None:
127
+ self._tmp_dir = TemporaryDirectory()
128
+ root = Path(self._tmp_dir.name)
129
+ self._root = root
130
+ self.__fs_buffer_path = None
131
+ self._mem_buffer: dict[int, list[R | Path]] = defaultdict(list)
132
+ self._missing_pages: dict[int, int] = dict()
133
+ self._current_size: int = 0
134
+
135
+ def __enter__(self) -> Self:
136
+ if self._tmp_dir is not None:
137
+ self._tmp_dir.__enter__()
138
+ self.__fs_buffer_path = self._root / uuid.uuid4().hex
139
+ self.__fs_buffer_path.mkdir()
140
+ return self
141
+
142
+ def __exit__(
143
+ self,
144
+ exc_type: type[BaseException] | None,
145
+ exc_val: BaseException | None,
146
+ exc_tb: TracebackType | None,
147
+ ) -> None:
148
+ if self._mem_buffer or self._missing_pages:
149
+ logger.warning("closing an non empty buffer")
150
+ if self._tmp_dir is not None:
151
+ self._tmp_dir.__exit__(exc_type, exc_val, exc_tb)
152
+ if self._fs_buffer_path.exists():
153
+ shutil.rmtree(self._fs_buffer_path)
154
+ self._mem_buffer = dict()
155
+ self._missing_pages = dict()
156
+
157
+ @property
158
+ def _fs_buffer_path(self) -> Path:
159
+ if not self.__fs_buffer_path:
160
+ msg = (
161
+ f"inconsistent state, {ResultBuffer.__class__.__name__} is a context"
162
+ f" manager, call __enter__ before using it"
163
+ )
164
+ raise ValueError(msg)
165
+ return self.__fs_buffer_path
166
+
167
+ def add(self, processed: ProcessedPages, result: R) -> None:
168
+ size = asizeof.asizeof(result)
169
+ if self._current_size + size > self._max_bytes:
170
+ path = self._page_path(processed.doc_idx)
171
+ self._save_fn(result, path)
172
+ result = path
173
+ else:
174
+ self._current_size += size
175
+ self._mem_buffer[processed.doc_idx].append(result)
176
+ if processed.doc_idx not in self._missing_pages:
177
+ self._missing_pages[processed.doc_idx] = processed.doc.n_pages
178
+ self._missing_pages[processed.doc_idx] -= processed.page_length
179
+
180
+ def is_complete(self, doc: int) -> bool:
181
+ return self._missing_pages[doc] == 0
182
+
183
+ def pop_complete(self, doc: int) -> list[R]:
184
+ if not self.is_complete(doc):
185
+ raise ValueError(f"{doc} is incomplete")
186
+ pages = self._mem_buffer.pop(doc)
187
+ self._missing_pages.pop(doc)
188
+ for i, page in enumerate(pages):
189
+ if isinstance(page, Path):
190
+ page = self._load_fn(page) # noqa: PLW2901
191
+ else:
192
+ self._current_size -= asizeof.asizeof(page)
193
+ pages[i] = page
194
+ return pages
195
+
196
+ def _page_path(self, doc_id: int) -> Path:
197
+ pages = self._mem_buffer[doc_id]
198
+ return self._fs_buffer_path / f"doc-{doc_id}-pages-{len(pages)}"
199
+
200
+ def __len__(self) -> int:
201
+ return len(self._mem_buffer)
202
+
203
+
204
+ # The converter process page_batch_size in parallel (GPU sees page_batch_size batches).
205
+ #
206
+ # The batching tradeoff is: avoid calling convert_all to many times vs. releasing the
207
+ # GIL often enough.
208
+ #
209
+ # Calling convert_all to many times on the same doc results in overhead. Each time we
210
+ # call the function, we create a doc processing backend + reload the doc.
211
+ #
212
+ # On the other hand we have to use a reasonable max_page_batches otherwise we process
213
+ # all the stream in a single call and take the risk to lock the GIL for too long. Some
214
+ # docling ops are sadly not async (numpy or torch inference are, but document loading
215
+ # and conversion aren't, so the asyncio.to_thread is not helping)
216
+ def batch_per_pages(
217
+ docs: Iterable[InputDoc],
218
+ page_batch_size: int,
219
+ *,
220
+ max_page_batches: int,
221
+ chunk_size: int = 1000,
222
+ ) -> Iterable[tuple[ProcessedPages]]:
223
+ # convert_all only accept to process docs on the exact same page_range
224
+ #
225
+ # We collect by chunk to avoid collecting too many inputs, input docs are
226
+ # lightweight anyway so memory impact should stay limited
227
+ #
228
+ # Additionally, results can be output unordered and partial results are buffered
229
+ # it's OK to process doc pages unordered.
230
+ # TODO: if it's not OK to sort because inputs is l
231
+ max_pages = page_batch_size * max_page_batches
232
+ docs = itertools.batched(docs, chunk_size, strict=False)
233
+ for chunk in docs:
234
+ short_docs = [d for d in chunk if d.n_pages <= max_pages]
235
+ long_docs = [d for d in chunk if d.n_pages > max_pages]
236
+ del chunk
237
+ gc.collect()
238
+ # Bin fill for docs smaller than max_pages
239
+ offset = yield from _bin_fill(short_docs, max_pages=max_pages)
240
+ # otherwise we just yield chunks of max_pages except the last chunk which is
241
+ # grouped by page_range
242
+ yield from _by_page_ranges(long_docs, max_pages=max_pages, offset=offset)
243
+
244
+
245
+ def _bin_fill(
246
+ docs: Iterable[InputDoc], max_pages: int, offset: int = 0
247
+ ) -> Generator[tuple[ProcessedPages], None, int]:
248
+ bins = defaultdict(deque)
249
+ doc_idx = offset
250
+ for doc in docs:
251
+ if doc.n_pages > max_pages:
252
+ msg = f"expected docs to have <= {max_pages} pages"
253
+ raise ValueError(msg)
254
+ pages = ProcessedPages(doc=doc, doc_idx=doc_idx)
255
+ doc_idx += 1
256
+ available_space = (s for s in sorted(bins.keys()) if doc.n_pages <= s)
257
+ available_space = next(available_space, max_pages)
258
+ selected = bins[available_space]
259
+ selected = selected.pop() if selected else []
260
+ selected.append(pages)
261
+ available_space -= doc.n_pages
262
+ if available_space == 0:
263
+ yield tuple(selected)
264
+ continue
265
+ bins[available_space].append(selected)
266
+ for range_bins in bins.values():
267
+ for b in range_bins:
268
+ yield tuple(b)
269
+ return doc_idx
270
+
271
+
272
+ def _by_page_ranges(
273
+ docs: Iterable[InputDoc], max_pages: int, offset: int = 0
274
+ ) -> Generator[tuple[ProcessedPages], None, int]:
275
+ by_range = defaultdict(list)
276
+ doc_idx = offset
277
+ for doc in docs:
278
+ if doc.n_pages < max_pages:
279
+ msg = f"expected docs to have >= {max_pages} pages"
280
+ raise ValueError(msg)
281
+
282
+ for i in range(0, doc.n_pages, max_pages):
283
+ start = i
284
+ end = min(start + max_pages, doc.n_pages)
285
+ rng = (start, end)
286
+ rng_size = end - start
287
+ alone_in_batch = rng_size == max_pages
288
+ pages = ProcessedPages(doc=doc, doc_idx=doc_idx, page_range=rng)
289
+ if alone_in_batch:
290
+ yield (pages,)
291
+ continue
292
+ by_range[rng].append(pages)
293
+ is_complete = len(by_range[rng]) == (max_pages // rng_size)
294
+ if is_complete:
295
+ yield tuple(by_range.pop(rng))
296
+ doc_idx += 1
297
+ for v in by_range.values():
298
+ yield tuple(v)
299
+ return doc_idx
@@ -8,8 +8,9 @@ authors = [
8
8
  readme = "README.md"
9
9
  requires-python = ">=3.13,<3.15"
10
10
  dependencies = [
11
+ "extract-core~=0.8.0",
11
12
  "icij-common~=0.8.2",
12
- "extract-core~=0.7.0",
13
+ "pympler~=1.1",
13
14
  ]
14
15
 
15
16
  [project.optional-dependencies]
@@ -919,6 +919,7 @@ source = { editable = "." }
919
919
  dependencies = [
920
920
  { name = "extract-core" },
921
921
  { name = "icij-common" },
922
+ { name = "pympler" },
922
923
  ]
923
924
 
924
925
  [package.optional-dependencies]
@@ -965,6 +966,7 @@ requires-dist = [
965
966
  { name = "mineru", extras = ["pipeline", "vlm"], marker = "extra == 'mineru'", specifier = "~=3.2" },
966
967
  { name = "notebook", marker = "extra == 'benches'", specifier = ">=7.4.5" },
967
968
  { name = "pydantic-extra-types", extras = ["pycountry"], marker = "extra == 'mineru'", specifier = "~=2.11" },
969
+ { name = "pympler", specifier = "~=1.1" },
968
970
  { name = "pypdfium2", marker = "extra == 'benches'", specifier = ">=4.30.0" },
969
971
  { name = "python-pptx", marker = "extra == 'mineru'", specifier = "~=1.0" },
970
972
  { name = "six", marker = "extra == 'mineru'", specifier = "~=1.17" },
@@ -2600,9 +2602,9 @@ name = "ocrmac"
2600
2602
  version = "1.0.1"
2601
2603
  source = { registry = "https://pypi.org/simple" }
2602
2604
  dependencies = [
2603
- { name = "click", marker = "sys_platform == 'darwin' or extra == 'extra-14-extract-python-marker' or extra != 'extra-14-extract-python-mineru'" },
2604
- { name = "pillow", marker = "sys_platform == 'darwin' or extra == 'extra-14-extract-python-marker' or extra != 'extra-14-extract-python-mineru'" },
2605
- { name = "pyobjc-framework-vision", marker = "sys_platform == 'darwin' or extra == 'extra-14-extract-python-marker' or extra != 'extra-14-extract-python-mineru'" },
2605
+ { name = "click", marker = "sys_platform == 'darwin' or (extra == 'extra-14-extract-python-marker' and extra == 'extra-14-extract-python-mineru')" },
2606
+ { name = "pillow", marker = "sys_platform == 'darwin' or (extra == 'extra-14-extract-python-marker' and extra == 'extra-14-extract-python-mineru')" },
2607
+ { name = "pyobjc-framework-vision", marker = "sys_platform == 'darwin' or (extra == 'extra-14-extract-python-marker' and extra == 'extra-14-extract-python-mineru')" },
2606
2608
  ]
2607
2609
  sdist = { url = "https://files.pythonhosted.org/packages/5e/07/3e15ab404f75875c5e48c47163300eb90b7409044d8711fc3aaf52503f2e/ocrmac-1.0.1.tar.gz", hash = "sha256:507fe5e4cbd67b2d03f6729a52bbc11f9d0b58241134eb958a5daafd4b9d93d9", size = 1454317, upload-time = "2026-01-08T16:44:26.412Z" }
2608
2610
  wheels = [
@@ -3291,6 +3293,18 @@ version = "2.10"
3291
3293
  source = { registry = "https://pypi.org/simple" }
3292
3294
  sdist = { url = "https://files.pythonhosted.org/packages/5d/ab/34ec41718af73c00119d0351b7a2531d2ebddb51833a36448fc7b862be60/pylatexenc-2.10.tar.gz", hash = "sha256:3dd8fd84eb46dc30bee1e23eaab8d8fb5a7f507347b23e5f38ad9675c84f40d3", size = 162597, upload-time = "2021-04-06T07:56:07.854Z" }
3293
3295
 
3296
+ [[package]]
3297
+ name = "pympler"
3298
+ version = "1.1"
3299
+ source = { registry = "https://pypi.org/simple" }
3300
+ dependencies = [
3301
+ { name = "pywin32", marker = "sys_platform == 'win32' or (extra == 'extra-14-extract-python-marker' and extra == 'extra-14-extract-python-mineru')" },
3302
+ ]
3303
+ sdist = { url = "https://files.pythonhosted.org/packages/dd/37/c384631908029676d8e7213dd956bb686af303a80db7afbc9be36bc49495/pympler-1.1.tar.gz", hash = "sha256:1eaa867cb8992c218430f1708fdaccda53df064144d1c5656b1e6f1ee6000424", size = 179954, upload-time = "2024-06-28T19:56:06.563Z" }
3304
+ wheels = [
3305
+ { url = "https://files.pythonhosted.org/packages/79/4f/a6a2e2b202d7fd97eadfe90979845b8706676b41cbd3b42ba75adf329d1f/Pympler-1.1-py3-none-any.whl", hash = "sha256:5b223d6027d0619584116a0cbc28e8d2e378f7a79c1e5e024f9ff3b673c58506", size = 165766, upload-time = "2024-06-28T19:56:05.087Z" },
3306
+ ]
3307
+
3294
3308
  [[package]]
3295
3309
  name = "pyobjc-core"
3296
3310
  version = "12.2.1"
@@ -3308,7 +3322,7 @@ name = "pyobjc-framework-cocoa"
3308
3322
  version = "12.2.1"
3309
3323
  source = { registry = "https://pypi.org/simple" }
3310
3324
  dependencies = [
3311
- { name = "pyobjc-core", marker = "sys_platform == 'darwin' or extra == 'extra-14-extract-python-marker' or extra != 'extra-14-extract-python-mineru'" },
3325
+ { name = "pyobjc-core", marker = "sys_platform == 'darwin' or (extra == 'extra-14-extract-python-marker' and extra == 'extra-14-extract-python-mineru')" },
3312
3326
  ]
3313
3327
  sdist = { url = "https://files.pythonhosted.org/packages/51/34/fbe38a204643aa4e1b91391cdce07a34da565a69171ebcad08de7438a556/pyobjc_framework_cocoa-12.2.1.tar.gz", hash = "sha256:b94b37fe5730e5ae1fb0052912cd174e6ec329b0bfba4a012ae5db1014b5864b", size = 3125751, upload-time = "2026-06-19T16:20:05.159Z" }
3314
3328
  wheels = [
@@ -3323,8 +3337,8 @@ name = "pyobjc-framework-coreml"
3323
3337
  version = "12.2.1"
3324
3338
  source = { registry = "https://pypi.org/simple" }
3325
3339
  dependencies = [
3326
- { name = "pyobjc-core", marker = "sys_platform == 'darwin' or extra == 'extra-14-extract-python-marker' or extra != 'extra-14-extract-python-mineru'" },
3327
- { name = "pyobjc-framework-cocoa", marker = "sys_platform == 'darwin' or extra == 'extra-14-extract-python-marker' or extra != 'extra-14-extract-python-mineru'" },
3340
+ { name = "pyobjc-core", marker = "sys_platform == 'darwin' or (extra == 'extra-14-extract-python-marker' and extra == 'extra-14-extract-python-mineru')" },
3341
+ { name = "pyobjc-framework-cocoa", marker = "sys_platform == 'darwin' or (extra == 'extra-14-extract-python-marker' and extra == 'extra-14-extract-python-mineru')" },
3328
3342
  ]
3329
3343
  sdist = { url = "https://files.pythonhosted.org/packages/98/1e/7d2db3e4468eb04cc92264be83113d86eea4f96302742437de695a445d6d/pyobjc_framework_coreml-12.2.1.tar.gz", hash = "sha256:ef3c2b6a160891b44173235603d10174929656b9c206d6f2f443fe2aa903c2cb", size = 49272, upload-time = "2026-06-19T16:20:18.459Z" }
3330
3344
  wheels = [
@@ -3339,8 +3353,8 @@ name = "pyobjc-framework-quartz"
3339
3353
  version = "12.2.1"
3340
3354
  source = { registry = "https://pypi.org/simple" }
3341
3355
  dependencies = [
3342
- { name = "pyobjc-core", marker = "sys_platform == 'darwin' or extra == 'extra-14-extract-python-marker' or extra != 'extra-14-extract-python-mineru'" },
3343
- { name = "pyobjc-framework-cocoa", marker = "sys_platform == 'darwin' or extra == 'extra-14-extract-python-marker' or extra != 'extra-14-extract-python-mineru'" },
3356
+ { name = "pyobjc-core", marker = "sys_platform == 'darwin' or (extra == 'extra-14-extract-python-marker' and extra == 'extra-14-extract-python-mineru')" },
3357
+ { name = "pyobjc-framework-cocoa", marker = "sys_platform == 'darwin' or (extra == 'extra-14-extract-python-marker' and extra == 'extra-14-extract-python-mineru')" },
3344
3358
  ]
3345
3359
  sdist = { url = "https://files.pythonhosted.org/packages/3b/f6/2a8b84dbf1fe7c04dd96ea73d991678d4e09a909f51971ecc51629bb2ab4/pyobjc_framework_quartz-12.2.1.tar.gz", hash = "sha256:b3b8b6f71e66147f8ff9e6213864cc8527e3a0b1ee90835b93ce221f4802d9b0", size = 3215521, upload-time = "2026-06-19T16:21:30.199Z" }
3346
3360
  wheels = [
@@ -3355,10 +3369,10 @@ name = "pyobjc-framework-vision"
3355
3369
  version = "12.2.1"
3356
3370
  source = { registry = "https://pypi.org/simple" }
3357
3371
  dependencies = [
3358
- { name = "pyobjc-core", marker = "sys_platform == 'darwin' or extra == 'extra-14-extract-python-marker' or extra != 'extra-14-extract-python-mineru'" },
3359
- { name = "pyobjc-framework-cocoa", marker = "sys_platform == 'darwin' or extra == 'extra-14-extract-python-marker' or extra != 'extra-14-extract-python-mineru'" },
3360
- { name = "pyobjc-framework-coreml", marker = "sys_platform == 'darwin' or extra == 'extra-14-extract-python-marker' or extra != 'extra-14-extract-python-mineru'" },
3361
- { name = "pyobjc-framework-quartz", marker = "sys_platform == 'darwin' or extra == 'extra-14-extract-python-marker' or extra != 'extra-14-extract-python-mineru'" },
3372
+ { name = "pyobjc-core", marker = "sys_platform == 'darwin' or (extra == 'extra-14-extract-python-marker' and extra == 'extra-14-extract-python-mineru')" },
3373
+ { name = "pyobjc-framework-cocoa", marker = "sys_platform == 'darwin' or (extra == 'extra-14-extract-python-marker' and extra == 'extra-14-extract-python-mineru')" },
3374
+ { name = "pyobjc-framework-coreml", marker = "sys_platform == 'darwin' or (extra == 'extra-14-extract-python-marker' and extra == 'extra-14-extract-python-mineru')" },
3375
+ { name = "pyobjc-framework-quartz", marker = "sys_platform == 'darwin' or (extra == 'extra-14-extract-python-marker' and extra == 'extra-14-extract-python-mineru')" },
3362
3376
  ]
3363
3377
  sdist = { url = "https://files.pythonhosted.org/packages/0e/7a/1fdffff1b6bf124b260a2169869f4b71a08b9f6603698f7dec990d5ae5f3/pyobjc_framework_vision-12.2.1.tar.gz", hash = "sha256:debfd59dd7d962a6053bf733370148c11a9ec44091b517a0966f48d81c305879", size = 72683, upload-time = "2026-06-19T16:22:01.102Z" }
3364
3378
  wheels = [
File without changes
@@ -1,88 +0,0 @@
1
- import os
2
- from collections.abc import Callable, Generator, Iterable, Iterator
3
- from contextlib import contextmanager
4
- from copy import copy
5
- from functools import wraps
6
- from itertools import tee
7
- from pathlib import Path, PurePath
8
- from typing import BinaryIO, Protocol, TypeVar
9
-
10
- from extract_core import Error, InputDoc, Pages, Result, Status
11
-
12
- R = TypeVar("R")
13
- In = TypeVar("In")
14
-
15
-
16
- def map_and_preserve(
17
- fn: Callable[[Iterable[In]], Iterator[R]], inputs: Iterable[In]
18
- ) -> tuple[Iterable[In], Iterator[R]]:
19
- save_inputs, function_inputs = tee(inputs)
20
- outputs = iter(fn(function_inputs))
21
- return save_inputs, outputs
22
-
23
-
24
- def path_to_artifacts_dirname(path: PurePath, sep: str = "_") -> str:
25
- dirname = f"{path.name[: -len(path.suffix)]}"
26
- ext = path.suffix
27
- if ext:
28
- dirname += sep + ext[1:]
29
- return dirname
30
-
31
-
32
- class DocProcessingFn(Protocol):
33
- def __call__(self, doc: InputDoc, *arg, **kwargs) -> Result: ...
34
-
35
-
36
- def report_recoverable_errors(
37
- recoverable_errors: tuple[type[Exception], ...] = tuple(),
38
- ) -> Callable[[DocProcessingFn], DocProcessingFn]:
39
- def make_decorator(f: DocProcessingFn) -> DocProcessingFn:
40
- @wraps(f)
41
- def wrapped(doc: InputDoc, *args, **kwargs) -> Result:
42
- try:
43
- return f(doc, *args, **kwargs)
44
- except recoverable_errors as e:
45
- error = Error.from_exception(e)
46
- return Result(
47
- input=doc.without_content(),
48
- status=Status.FAILURE,
49
- errors=[error],
50
- output=None,
51
- )
52
-
53
- return wrapped
54
-
55
- return make_decorator
56
-
57
-
58
- @contextmanager
59
- def chdir(path: Path) -> Generator[None, None, None]:
60
- cwd = Path.cwd()
61
- try:
62
- os.chdir(path)
63
- yield
64
- finally:
65
- os.chdir(cwd)
66
-
67
-
68
- @contextmanager
69
- def reset_env() -> Generator[None, None, None]:
70
- old_env = copy(dict(os.environ))
71
- try:
72
- yield
73
- finally:
74
- os.environ.clear()
75
- os.environ.update(old_env)
76
-
77
-
78
- def write_pages(pages: Iterable[str], page_sep: str, out: BinaryIO) -> Pages:
79
- pages_byte_sizes = []
80
- pages = iter(pages)
81
- content = None
82
- for p in pages:
83
- if content:
84
- pages_byte_sizes.append(out.write((content + page_sep).encode()))
85
- content = p
86
- if content:
87
- pages_byte_sizes.append(out.write(content.encode()))
88
- return Pages.from_pages_bytes_sizes(pages_byte_sizes)