extract-python 0.7.2__py3-none-any.whl → 0.8.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,20 +1,27 @@
1
1
  import asyncio
2
2
  import json
3
3
  import logging
4
+ import operator
4
5
  import shutil
5
6
  import tempfile
6
- from collections.abc import AsyncGenerator, Iterable, Iterator
7
- from functools import partial
7
+ from collections.abc import AsyncIterable, Iterable
8
+ from contextlib import AbstractContextManager
9
+ from functools import partial, reduce
8
10
  from pathlib import Path
9
11
  from typing import Any, Self
10
12
 
11
- from docling.datamodel.document import ConversionResult
13
+ from docling.datamodel.document import ConversionAssets
12
14
  from docling.datamodel.pipeline_options import PipelineOptions
15
+ from docling.datamodel.settings import (
16
+ DEFAULT_PAGE_RANGE,
17
+ AppSettings,
18
+ scoped,
19
+ )
13
20
  from docling.document_converter import DocumentConverter, FormatOption
21
+ from docling_core.types import DoclingDocument
14
22
 
15
23
  # TODO: this is long to load improve it
16
24
  from docling_core.types.doc import ImageRefMode
17
- from docling_core.types.io import DocumentStream
18
25
  from extract_core import (
19
26
  BaseModel,
20
27
  DoclingFormatOption,
@@ -33,7 +40,14 @@ from pydantic import ConfigDict, field_serializer
33
40
  from pydantic_core.core_schema import SerializerFunctionWrapHandler
34
41
 
35
42
  from .constants import ARTIFACTS, DEFAULT_MD_PAGE_SEP
36
- from .utils import chdir, map_and_preserve, path_to_artifacts_dirname, write_pages
43
+ from .utils import (
44
+ Range,
45
+ ResultBuffer,
46
+ batch_per_pages,
47
+ chdir,
48
+ path_to_artifacts_dirname,
49
+ write_pages,
50
+ )
37
51
 
38
52
  logger = logging.getLogger(__name__)
39
53
 
@@ -61,54 +75,112 @@ class DoclingPipeline(Pipeline):
61
75
 
62
76
  async def extract_content(
63
77
  self, docs: Iterable[InputDoc], output_format: OutputFormat, output_path: Path
64
- ) -> AsyncGenerator[Result, None]:
65
- docs, path_or_streams = map_and_preserve(_to_docling, docs)
66
- outputs = self._converter.convert_all(path_or_streams, raises_on_error=False)
78
+ ) -> AsyncIterable[Result]:
79
+ settings = self._config.settings
80
+ logger.info("starting extraction with settings: %s", settings)
81
+ with self._scoped_settings, self._result_buffer as buffer:
82
+ max_page_batches = settings.perf.max_page_batches
83
+ page_batch_size = settings.perf.page_batch_size
84
+ batches = batch_per_pages(
85
+ docs, page_batch_size, max_page_batches=max_page_batches
86
+ )
87
+ for batch in batches:
88
+ page_range = list({pages.page_range for pages in batch})
89
+ if len(page_range) > 1:
90
+ msg = "convert_all only accept 1 page range for all docs"
91
+ raise ValueError(msg)
92
+ page_range = page_range[0]
93
+ page_range = _docling_range(page_range)
94
+ docling_docs = (pages.doc.to_docling() for pages in batch)
95
+ outputs = self._converter.convert_all(
96
+ docling_docs, raises_on_error=False, page_range=page_range
97
+ )
98
+ processed = iter(batch)
99
+ sentinel = object()
100
+ while True:
101
+ res = await asyncio.to_thread(next, outputs, sentinel)
102
+ if res is sentinel:
103
+ break
104
+ pages = next(processed)
105
+ buffer.add(pages, res)
106
+ if buffer.is_complete(pages.doc_idx):
107
+ doc_pages = buffer.pop_complete(pages.doc_idx)
108
+ yield _to_result(
109
+ doc_pages, pages.doc, output_format, output_path=output_path
110
+ )
111
+
112
+ @property
113
+ def _scoped_settings(self) -> AbstractContextManager[AppSettings]:
114
+ settings = self._config.settings
115
+ docling_settings = scoped(
116
+ perf=settings.perf, debug=settings.debug, inference=settings.inference
117
+ )
118
+ return docling_settings
119
+
120
+ @property
121
+ def _result_buffer(self) -> ResultBuffer:
122
+ buffer = ResultBuffer(
123
+ max_size_bytes=self._config.result_buffer.max_size,
124
+ root=self._config.result_buffer.root,
125
+ save_fn=_save_conversion_result,
126
+ load_fn=_load_conversion_result,
127
+ )
128
+ return buffer
129
+
67
130
 
68
- sentinel = object()
69
- while True:
70
- res = await asyncio.to_thread(next, outputs, sentinel)
71
- if res is sentinel:
72
- return
73
- doc = next(docs)
74
- yield _to_result(res, doc, output_format, output_path=output_path)
131
+ def _save_conversion_result(res: ConversionAssets, path: Path) -> None:
132
+ return res.save(filename=path)
75
133
 
76
134
 
77
- def _to_docling(docs: Iterable[InputDoc]) -> Iterator["Path | DocumentStream"]:
78
- for d in docs:
79
- yield d.to_docling()
135
+ def _load_conversion_result(path: Path) -> ConversionAssets:
136
+ return ConversionAssets.load(path)
80
137
 
81
138
 
82
139
  def _to_result(
83
- res: ConversionResult,
84
- input_document: InputDoc,
140
+ buffer: list[ConversionAssets],
141
+ input_doc: InputDoc,
85
142
  output_format: OutputFormat,
86
143
  output_path: Path,
87
144
  **kwargs,
88
145
  ) -> Result:
146
+ import numpy as np # noqa: PLC0415
147
+
148
+ if not buffer:
149
+ raise ValueError("empty buffer")
150
+ merged = DoclingDocument.concatenate([res.document for res in buffer])
89
151
  output_path.mkdir(parents=True, exist_ok=True)
152
+ status = reduce(operator.iadd, (Status.from_docling(d.status) for d in buffer))
153
+ # TODO: implement confidence weight
154
+ confidence = np.mean([res.confidence.mean_score for res in buffer])
90
155
  output = None
91
- status = Status.from_docling(res.status)
92
156
  if status.allows_conversion:
93
157
  match output_format:
94
158
  case OutputFormat.MARKDOWN:
95
- output = _to_markdown_doc(res, output_path, **kwargs)
159
+ output = _to_markdown_doc(
160
+ merged,
161
+ input_path=input_doc.path,
162
+ output_path=output_path,
163
+ confidence=confidence,
164
+ **kwargs,
165
+ )
96
166
  case _:
97
167
  raise NotImplementedError(f"unsupported output format {output_format}")
98
- errors = [Error.from_docling(e) for e in res.errors]
99
- input_doc = input_document.without_content()
168
+ errors = [Error.from_docling(e) for res in buffer for e in res.errors]
100
169
  return Result(input=input_doc, status=status, errors=errors, output=output)
101
170
 
102
171
 
103
172
  def _to_markdown_doc(
104
- res: ConversionResult,
173
+ doc: DoclingDocument,
174
+ input_path: Path,
175
+ *,
105
176
  output_path: Path,
106
177
  page_sep: str = DEFAULT_MD_PAGE_SEP,
178
+ confidence: float,
107
179
  **kwargs,
108
180
  ) -> MarkdownDoc:
109
181
  # TODO: Should we add a hash to avoid collision between files with same names
110
182
  # nested in the tree structured
111
- md_dir_name = path_to_artifacts_dirname(res.input.file)
183
+ md_dir_name = path_to_artifacts_dirname(input_path)
112
184
  md_dir = output_path / md_dir_name
113
185
  if md_dir.exists():
114
186
  raise FileExistsError(f"directory {md_dir} already exists")
@@ -121,21 +193,21 @@ def _to_markdown_doc(
121
193
  with chdir(tmp_dir):
122
194
  # We do a chdir to bypass a Docling bug which only allows to maintain
123
195
  # relative image ref when saving the markdown to a relative path
124
- pages = _docling_pages_it(res, current_page_path, **kwargs)
196
+ pages = _docling_pages_it(doc, current_page_path, **kwargs)
125
197
  with md_path.open("wb") as f:
126
198
  pages = write_pages(pages, page_sep, f)
127
199
  # Clean up the tmp page file before move everything to the end destination
128
200
  current_page_path.unlink(missing_ok=True)
129
201
  shutil.move(tmp_dir, md_dir)
130
- return MarkdownDoc(path=Path(md_dir_name), pages=pages)
202
+ return MarkdownDoc(path=Path(md_dir_name), pages=pages, confidence=confidence)
131
203
 
132
204
 
133
205
  def _docling_pages_it(
134
- res: ConversionResult, output_path: Path, **kwargs
206
+ doc: DoclingDocument, output_path: Path, **kwargs
135
207
  ) -> Iterable[str]:
136
- n_pages = len(res.pages)
208
+ n_pages = len(doc.pages)
137
209
  for page_i in range(n_pages):
138
- res.document.save_as_markdown(
210
+ doc.save_as_markdown(
139
211
  output_path,
140
212
  page_no=page_i + 1,
141
213
  image_mode=ImageRefMode.REFERENCED,
@@ -193,3 +265,9 @@ class SerializableFormatOptions(DoclingFormatOption):
193
265
  serialized["table_structure_options"] = dict()
194
266
  serialized["table_structure_options"]["kind"] = table_structure_opts.kind
195
267
  return serialized
268
+
269
+
270
+ def _docling_range(rng: Range | None) -> tuple[int, int]:
271
+ if rng is None:
272
+ return DEFAULT_PAGE_RANGE
273
+ return (rng[0] + 1, rng[1] + 1)
extract_python/marker_.py CHANGED
@@ -1,6 +1,6 @@
1
1
  import asyncio
2
2
  import gc
3
- from collections.abc import AsyncGenerator, Iterable
3
+ from collections.abc import AsyncIterable, Iterable
4
4
  from copy import deepcopy
5
5
  from pathlib import Path
6
6
  from typing import TYPE_CHECKING
@@ -30,7 +30,7 @@ _MARKER_CONVERSION_ERRORS = tuple()
30
30
  class MarkerPipeline(Pipeline):
31
31
  async def extract_content(
32
32
  self, docs: Iterable[InputDoc], output_format: OutputFormat, output_path: Path
33
- ) -> AsyncGenerator[Result, None]:
33
+ ) -> AsyncIterable[Result]:
34
34
  from marker.config.parser import ConfigParser # noqa: PLC0415
35
35
  from marker.converters.pdf import PdfConverter # noqa: PLC0415
36
36
  from marker.models import create_model_dict # noqa: PLC0415
@@ -67,8 +67,7 @@ async def _process_doc(
67
67
  )
68
68
  case _:
69
69
  raise NotImplementedError(f"unsupported output format {output_format}")
70
- input_doc = doc.without_content()
71
- return Result(input=input_doc, status=Status.SUCCESS, output=output)
70
+ return Result(input=doc, status=Status.SUCCESS, output=output)
72
71
 
73
72
 
74
73
  def _to_markdown_doc(
@@ -96,4 +95,4 @@ def _to_markdown_doc(
96
95
  md_path = md_path.with_suffix(OutputFormat.MARKDOWN.value)
97
96
  with md_path.open("wb") as f:
98
97
  pages = write_pages(pages, page_sep, f)
99
- return MarkdownDoc(path=Path(md_dir_name), pages=pages)
98
+ return MarkdownDoc(path=Path(md_dir_name), pages=pages, confidence=None)
extract_python/miner_u.py CHANGED
@@ -1,7 +1,7 @@
1
1
  import json
2
2
  import os
3
3
  import shutil
4
- from collections.abc import AsyncGenerator, Callable, Iterable
4
+ from collections.abc import AsyncIterable, Callable, Iterable
5
5
  from functools import partial
6
6
  from pathlib import Path
7
7
  from tempfile import TemporaryDirectory
@@ -34,7 +34,7 @@ class MinerUPipeline(Pipeline):
34
34
 
35
35
  async def extract_content(
36
36
  self, docs: Iterable[InputDoc], output_format: OutputFormat, output_path: Path
37
- ) -> AsyncGenerator[Result, None]:
37
+ ) -> AsyncIterable[Result]:
38
38
  from mineru.cli.common import aio_do_parse # noqa: PLC0415
39
39
 
40
40
  with reset_env():
@@ -126,15 +126,13 @@ def _process_doc(
126
126
  raise NotImplementedError(f"unsupported output format {output_format}")
127
127
  middle_json_path = res_path / f"{doc.path.name}_middle.json"
128
128
  middle_json = json.loads(middle_json_path.read_text())
129
- pdf_info = middle_json["pdf_info"]
130
129
  shutil.move(res_path / "images", artifacts_dir)
131
- output = dump_content_fn(pdf_info)
132
- input_doc = doc.without_content()
133
- return Result(input=input_doc, status=Status.SUCCESS, output=output)
130
+ output = dump_content_fn(middle_json)
131
+ return Result(input=doc, status=Status.SUCCESS, output=output)
134
132
 
135
133
 
136
134
  def _dump_md_content(
137
- pdf_info: list[dict],
135
+ middle_json: dict,
138
136
  *,
139
137
  md_make_fn: MDMakeFunction,
140
138
  page_sep: str = DEFAULT_MD_PAGE_SEP,
@@ -145,11 +143,72 @@ def _dump_md_content(
145
143
  ) -> ConversionOutput:
146
144
  from mineru.utils.enum_class import MakeMode # noqa: PLC0415
147
145
 
146
+ pdf_info = middle_json["pdf_info"]
148
147
  if md_make_mode is None:
149
148
  md_make_mode = MakeMode.MM_MD
150
149
  pages = (md_make_fn([p], md_make_mode, str(im_dir)) for p in pdf_info)
151
150
  with md_path.open("wb") as f:
152
151
  pages = write_pages(pages, page_sep, f)
153
152
  output_path = md_path.parent.relative_to(output_path)
154
- output = ConversionOutput(path=output_path, pages=pages)
153
+ confidence = _mineru_confidence(pdf_info)
154
+ output = ConversionOutput(path=output_path, pages=pages, confidence=confidence)
155
155
  return output
156
+
157
+
158
+ def _mineru_confidence(pdf_info: list[dict]) -> float:
159
+ if not pdf_info:
160
+ return 1.0
161
+ block_conf = _mineru_block_confidence(pdf_info)
162
+ line_config = _mineru_line_confidence(pdf_info)
163
+ return (block_conf + line_config) / 2.0
164
+
165
+
166
+ def _mineru_block_confidence(pdf_info: list[dict]) -> float:
167
+ import numpy as np # noqa: PLC0415
168
+
169
+ scores = []
170
+ for info in pdf_info:
171
+ for block in info["para_blocks"]:
172
+ score = block.get("score")
173
+ if score is not None:
174
+ scores.append(score)
175
+ if scores:
176
+ return np.average(scores)
177
+ return 1.0
178
+
179
+
180
+ def _mineru_line_confidence(pdf_info: list[dict]) -> float:
181
+ import numpy as np # noqa: PLC0415
182
+
183
+ scores = []
184
+ lengths = []
185
+ for info in pdf_info:
186
+ for block in info["para_blocks"]:
187
+ for line in block.get("lines", []):
188
+ for span in line["spans"]:
189
+ score = span.get("score")
190
+ if score is not None:
191
+ scores.append(score)
192
+ lengths.append(len(span["content"]))
193
+ if scores:
194
+ return np.average(scores, weights=lengths)
195
+ return 1.0
196
+
197
+
198
+ def _parse_block(block: dict) -> tuple[list[float], list[float]]:
199
+ if "lines" in block:
200
+ scores = []
201
+ lengths = []
202
+ for line in block.get("lines", []):
203
+ for span in line["spans"]:
204
+ score = span.get("score")
205
+ if score is not None:
206
+ scores.append(score)
207
+ lengths.append(len(span["content"]))
208
+ return scores, lengths
209
+ if "blocs" in block:
210
+ scores, lengths = (_parse_block(b) for b in block["blocs"])
211
+ scores = sum(*scores, start=[])
212
+ lengths = sum(*lengths, start=[])
213
+ return scores, lengths
214
+ raise NotImplementedError(f"unsupported block: {block}")
extract_python/utils.py CHANGED
@@ -1,21 +1,30 @@
1
+ import gc
2
+ import itertools
3
+ import logging
1
4
  import os
5
+ import shutil
6
+ import uuid
7
+ from collections import defaultdict, deque
2
8
  from collections.abc import Callable, Generator, Iterable, Iterator
3
9
  from contextlib import contextmanager
4
10
  from copy import copy
11
+ from dataclasses import dataclass
5
12
  from functools import wraps
6
13
  from itertools import tee
7
14
  from pathlib import Path, PurePath
8
- from typing import BinaryIO, Protocol, TypeVar
15
+ from tempfile import TemporaryDirectory
16
+ from types import TracebackType
17
+ from typing import BinaryIO, Protocol, Self
9
18
 
10
19
  from extract_core import Error, InputDoc, Pages, Result, Status
20
+ from pympler import asizeof
11
21
 
12
- R = TypeVar("R")
13
- In = TypeVar("In")
22
+ logger = logging.getLogger(__name__)
14
23
 
15
24
 
16
- def map_and_preserve(
17
- fn: Callable[[Iterable[In]], Iterator[R]], inputs: Iterable[In]
18
- ) -> tuple[Iterable[In], Iterator[R]]:
25
+ def map_and_preserve[I, R](
26
+ fn: Callable[[Iterable[I]], Iterator[R]], inputs: Iterable[I]
27
+ ) -> tuple[Iterable[I], Iterator[R]]:
19
28
  save_inputs, function_inputs = tee(inputs)
20
29
  outputs = iter(fn(function_inputs))
21
30
  return save_inputs, outputs
@@ -44,10 +53,7 @@ def report_recoverable_errors(
44
53
  except recoverable_errors as e:
45
54
  error = Error.from_exception(e)
46
55
  return Result(
47
- input=doc.without_content(),
48
- status=Status.FAILURE,
49
- errors=[error],
50
- output=None,
56
+ input=doc, status=Status.FAILURE, errors=[error], output=None
51
57
  )
52
58
 
53
59
  return wrapped
@@ -56,7 +62,7 @@ def report_recoverable_errors(
56
62
 
57
63
 
58
64
  @contextmanager
59
- def chdir(path: Path) -> Generator[None, None, None]:
65
+ def chdir(path: Path) -> Generator[None]:
60
66
  cwd = Path.cwd()
61
67
  try:
62
68
  os.chdir(path)
@@ -66,7 +72,7 @@ def chdir(path: Path) -> Generator[None, None, None]:
66
72
 
67
73
 
68
74
  @contextmanager
69
- def reset_env() -> Generator[None, None, None]:
75
+ def reset_env() -> Generator[None]:
70
76
  old_env = copy(dict(os.environ))
71
77
  try:
72
78
  yield
@@ -86,3 +92,208 @@ def write_pages(pages: Iterable[str], page_sep: str, out: BinaryIO) -> Pages:
86
92
  if content:
87
93
  pages_byte_sizes.append(out.write(content.encode()))
88
94
  return Pages.from_pages_bytes_sizes(pages_byte_sizes)
95
+
96
+
97
+ Range = tuple[int, int]
98
+
99
+
100
+ @dataclass(frozen=True)
101
+ class ProcessedPages:
102
+ doc: InputDoc
103
+ doc_idx: int
104
+ page_range: Range | None = None
105
+
106
+ @property
107
+ def page_length(self) -> int:
108
+ if self.page_range is None:
109
+ return self.doc.n_pages
110
+ return self.page_range[1] - self.page_range[0]
111
+
112
+
113
+ class ResultBuffer[R]:
114
+ def __init__(
115
+ self,
116
+ max_size_bytes: int,
117
+ save_fn: Callable[[R, Path], None],
118
+ *,
119
+ load_fn: Callable[[Path], R],
120
+ root: Path | None = None,
121
+ ):
122
+ self._max_bytes = max_size_bytes
123
+ self._save_fn = save_fn
124
+ self._load_fn = load_fn
125
+ self._tmp_dir = None
126
+ if root is None:
127
+ self._tmp_dir = TemporaryDirectory()
128
+ root = Path(self._tmp_dir.name)
129
+ self._root = root
130
+ self.__fs_buffer_path = None
131
+ self._mem_buffer: dict[int, list[R | Path]] = defaultdict(list)
132
+ self._missing_pages: dict[int, int] = dict()
133
+ self._current_size: int = 0
134
+
135
+ def __enter__(self) -> Self:
136
+ if self._tmp_dir is not None:
137
+ self._tmp_dir.__enter__()
138
+ self.__fs_buffer_path = self._root / uuid.uuid4().hex
139
+ self.__fs_buffer_path.mkdir()
140
+ return self
141
+
142
+ def __exit__(
143
+ self,
144
+ exc_type: type[BaseException] | None,
145
+ exc_val: BaseException | None,
146
+ exc_tb: TracebackType | None,
147
+ ) -> None:
148
+ if self._mem_buffer or self._missing_pages:
149
+ logger.warning("closing an non empty buffer")
150
+ if self._tmp_dir is not None:
151
+ self._tmp_dir.__exit__(exc_type, exc_val, exc_tb)
152
+ if self._fs_buffer_path.exists():
153
+ shutil.rmtree(self._fs_buffer_path)
154
+ self._mem_buffer = dict()
155
+ self._missing_pages = dict()
156
+
157
+ @property
158
+ def _fs_buffer_path(self) -> Path:
159
+ if not self.__fs_buffer_path:
160
+ msg = (
161
+ f"inconsistent state, {ResultBuffer.__class__.__name__} is a context"
162
+ f" manager, call __enter__ before using it"
163
+ )
164
+ raise ValueError(msg)
165
+ return self.__fs_buffer_path
166
+
167
+ def add(self, processed: ProcessedPages, result: R) -> None:
168
+ size = asizeof.asizeof(result)
169
+ if self._current_size + size > self._max_bytes:
170
+ path = self._page_path(processed.doc_idx)
171
+ self._save_fn(result, path)
172
+ result = path
173
+ else:
174
+ self._current_size += size
175
+ self._mem_buffer[processed.doc_idx].append(result)
176
+ if processed.doc_idx not in self._missing_pages:
177
+ self._missing_pages[processed.doc_idx] = processed.doc.n_pages
178
+ self._missing_pages[processed.doc_idx] -= processed.page_length
179
+
180
+ def is_complete(self, doc: int) -> bool:
181
+ return self._missing_pages[doc] == 0
182
+
183
+ def pop_complete(self, doc: int) -> list[R]:
184
+ if not self.is_complete(doc):
185
+ raise ValueError(f"{doc} is incomplete")
186
+ pages = self._mem_buffer.pop(doc)
187
+ self._missing_pages.pop(doc)
188
+ for i, page in enumerate(pages):
189
+ if isinstance(page, Path):
190
+ page = self._load_fn(page) # noqa: PLW2901
191
+ else:
192
+ self._current_size -= asizeof.asizeof(page)
193
+ pages[i] = page
194
+ return pages
195
+
196
+ def _page_path(self, doc_id: int) -> Path:
197
+ pages = self._mem_buffer[doc_id]
198
+ return self._fs_buffer_path / f"doc-{doc_id}-pages-{len(pages)}"
199
+
200
+ def __len__(self) -> int:
201
+ return len(self._mem_buffer)
202
+
203
+
204
+ # The converter process page_batch_size in parallel (GPU sees page_batch_size batches).
205
+ #
206
+ # The batching tradeoff is: avoid calling convert_all to many times vs. releasing the
207
+ # GIL often enough.
208
+ #
209
+ # Calling convert_all to many times on the same doc results in overhead. Each time we
210
+ # call the function, we create a doc processing backend + reload the doc.
211
+ #
212
+ # On the other hand we have to use a reasonable max_page_batches otherwise we process
213
+ # all the stream in a single call and take the risk to lock the GIL for too long. Some
214
+ # docling ops are sadly not async (numpy or torch inference are, but document loading
215
+ # and conversion aren't, so the asyncio.to_thread is not helping)
216
+ def batch_per_pages(
217
+ docs: Iterable[InputDoc],
218
+ page_batch_size: int,
219
+ *,
220
+ max_page_batches: int,
221
+ chunk_size: int = 1000,
222
+ ) -> Iterable[tuple[ProcessedPages]]:
223
+ # convert_all only accept to process docs on the exact same page_range
224
+ #
225
+ # We collect by chunk to avoid collecting too many inputs, input docs are
226
+ # lightweight anyway so memory impact should stay limited
227
+ #
228
+ # Additionally, results can be output unordered and partial results are buffered
229
+ # it's OK to process doc pages unordered.
230
+ # TODO: if it's not OK to sort because inputs is l
231
+ max_pages = page_batch_size * max_page_batches
232
+ docs = itertools.batched(docs, chunk_size, strict=False)
233
+ for chunk in docs:
234
+ short_docs = [d for d in chunk if d.n_pages <= max_pages]
235
+ long_docs = [d for d in chunk if d.n_pages > max_pages]
236
+ del chunk
237
+ gc.collect()
238
+ # Bin fill for docs smaller than max_pages
239
+ offset = yield from _bin_fill(short_docs, max_pages=max_pages)
240
+ # otherwise we just yield chunks of max_pages except the last chunk which is
241
+ # grouped by page_range
242
+ yield from _by_page_ranges(long_docs, max_pages=max_pages, offset=offset)
243
+
244
+
245
+ def _bin_fill(
246
+ docs: Iterable[InputDoc], max_pages: int, offset: int = 0
247
+ ) -> Generator[tuple[ProcessedPages], None, int]:
248
+ bins = defaultdict(deque)
249
+ doc_idx = offset
250
+ for doc in docs:
251
+ if doc.n_pages > max_pages:
252
+ msg = f"expected docs to have <= {max_pages} pages"
253
+ raise ValueError(msg)
254
+ pages = ProcessedPages(doc=doc, doc_idx=doc_idx)
255
+ doc_idx += 1
256
+ available_space = (s for s in sorted(bins.keys()) if doc.n_pages <= s)
257
+ available_space = next(available_space, max_pages)
258
+ selected = bins[available_space]
259
+ selected = selected.pop() if selected else []
260
+ selected.append(pages)
261
+ available_space -= doc.n_pages
262
+ if available_space == 0:
263
+ yield tuple(selected)
264
+ continue
265
+ bins[available_space].append(selected)
266
+ for range_bins in bins.values():
267
+ for b in range_bins:
268
+ yield tuple(b)
269
+ return doc_idx
270
+
271
+
272
+ def _by_page_ranges(
273
+ docs: Iterable[InputDoc], max_pages: int, offset: int = 0
274
+ ) -> Generator[tuple[ProcessedPages], None, int]:
275
+ by_range = defaultdict(list)
276
+ doc_idx = offset
277
+ for doc in docs:
278
+ if doc.n_pages < max_pages:
279
+ msg = f"expected docs to have >= {max_pages} pages"
280
+ raise ValueError(msg)
281
+
282
+ for i in range(0, doc.n_pages, max_pages):
283
+ start = i
284
+ end = min(start + max_pages, doc.n_pages)
285
+ rng = (start, end)
286
+ rng_size = end - start
287
+ alone_in_batch = rng_size == max_pages
288
+ pages = ProcessedPages(doc=doc, doc_idx=doc_idx, page_range=rng)
289
+ if alone_in_batch:
290
+ yield (pages,)
291
+ continue
292
+ by_range[rng].append(pages)
293
+ is_complete = len(by_range[rng]) == (max_pages // rng_size)
294
+ if is_complete:
295
+ yield tuple(by_range.pop(rng))
296
+ doc_idx += 1
297
+ for v in by_range.values():
298
+ yield tuple(v)
299
+ return doc_idx
@@ -1,6 +1,6 @@
1
- Metadata-Version: 2.4
1
+ Metadata-Version: 2.5
2
2
  Name: extract-python
3
- Version: 0.7.2
3
+ Version: 0.8.0
4
4
  Summary: Structured content extraction
5
5
  Project-URL: Homepage, https://github.com/ICIJ/extract-python
6
6
  Project-URL: Repository, https://github.com/ICIJ/extract-python
@@ -9,6 +9,7 @@ Author-email: Clément Doumouro <cdoumouro@icij.org>
9
9
  Requires-Python: <3.15,>=3.13
10
10
  Requires-Dist: extract-core~=0.7.0
11
11
  Requires-Dist: icij-common~=0.8.2
12
+ Requires-Dist: pympler~=1.1
12
13
  Provides-Extra: benches
13
14
  Requires-Dist: html2image~=2.0.7; extra == 'benches'
14
15
  Requires-Dist: markdown2>=2.5.4; extra == 'benches'
@@ -24,3 +25,6 @@ Requires-Dist: mineru[pipeline,vlm]~=3.2; extra == 'mineru'
24
25
  Requires-Dist: pydantic-extra-types[pycountry]~=2.11; extra == 'mineru'
25
26
  Requires-Dist: python-pptx~=1.0; extra == 'mineru'
26
27
  Requires-Dist: six~=1.17; extra == 'mineru'
28
+ Description-Content-Type: text/markdown
29
+
30
+ ô
@@ -0,0 +1,9 @@
1
+ extract_python/__init__.py,sha256=DA2LUro6vMjfS8fb2MsqO95FbJEZHyZ7kFyn42q02Wk,759
2
+ extract_python/constants.py,sha256=659V40LcTWJhX3IbuJLSSvI5AsGJh9ciMrGCfzJn2zA,98
3
+ extract_python/docling_.py,sha256=H_A-6pMObQ0Q7XA62GCBv0nbGtW5rKjfOVW3giU4NCk,10146
4
+ extract_python/marker_.py,sha256=bb0DJXrSWOznIuPnw4-adR5_R9QiwtFGGBh5KlXFxjM,3323
5
+ extract_python/miner_u.py,sha256=OCIleNu-VvaJlx_zG0VuagJvQvElXV7e-aPXqHbp31A,7302
6
+ extract_python/utils.py,sha256=d_Cp3Db5mHR64Z2V_y6ZMhRfSFZ72skWoHv3fUPBXe0,10023
7
+ extract_python-0.8.0.dist-info/METADATA,sha256=BosDjnRUbQZvv5r5-CwwYLOD4XunmnSmdtc7O5c_KAA,1289
8
+ extract_python-0.8.0.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
9
+ extract_python-0.8.0.dist-info/RECORD,,
@@ -1,4 +1,4 @@
1
1
  Wheel-Version: 1.0
2
- Generator: hatchling 1.30.1
2
+ Generator: hatchling 1.32.0
3
3
  Root-Is-Purelib: true
4
4
  Tag: py3-none-any
@@ -1,9 +0,0 @@
1
- extract_python/__init__.py,sha256=DA2LUro6vMjfS8fb2MsqO95FbJEZHyZ7kFyn42q02Wk,759
2
- extract_python/constants.py,sha256=659V40LcTWJhX3IbuJLSSvI5AsGJh9ciMrGCfzJn2zA,98
3
- extract_python/docling_.py,sha256=j1rVhKG7m1ef43VDsS6XGP0INPRY1Rcovzf1mjZ57tU,7352
4
- extract_python/marker_.py,sha256=R_SXhqk5GmEWqJrYgg3tRdXKHms7n0FueNr-aOCDvLc,3358
5
- extract_python/miner_u.py,sha256=MtXmnG-dFIGa3dXVrixfUU32yc88US0dhu7E3x6wQIM,5415
6
- extract_python/utils.py,sha256=9IWW9_VVdUPHOHhdDgkXx16R1X1FPz8-nTBNYsLCFfA,2443
7
- extract_python-0.7.2.dist-info/METADATA,sha256=3YNy7IE2YZw0cfMxKqTNYbfl-cUjeAUFJMMq2702-lA,1218
8
- extract_python-0.7.2.dist-info/WHEEL,sha256=mffPy8wBnZQn2VnJUU5jE99KsxaSfiyMHV9Yt0aLVxs,87
9
- extract_python-0.7.2.dist-info/RECORD,,