extract-python 0.7.2__tar.gz → 0.8.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- extract_python-0.8.0/.python-version +1 -0
- {extract_python-0.7.2 → extract_python-0.8.0}/PKG-INFO +6 -2
- extract_python-0.8.0/README.md +1 -0
- {extract_python-0.7.2 → extract_python-0.8.0}/extract_python/docling_.py +109 -31
- {extract_python-0.7.2 → extract_python-0.8.0}/extract_python/marker_.py +4 -5
- {extract_python-0.7.2 → extract_python-0.8.0}/extract_python/miner_u.py +67 -8
- extract_python-0.8.0/extract_python/utils.py +299 -0
- {extract_python-0.7.2 → extract_python-0.8.0}/pyproject.toml +2 -1
- {extract_python-0.7.2 → extract_python-0.8.0}/uv.lock +891 -1291
- extract_python-0.7.2/.python-version +0 -1
- extract_python-0.7.2/README.md +0 -0
- extract_python-0.7.2/extract_python/utils.py +0 -88
- {extract_python-0.7.2 → extract_python-0.8.0}/.gitignore +0 -0
- {extract_python-0.7.2 → extract_python-0.8.0}/benches/__init__.py +0 -0
- {extract_python-0.7.2 → extract_python-0.8.0}/benches/compare.ipynb +0 -0
- {extract_python-0.7.2 → extract_python-0.8.0}/benches/compare.py +0 -0
- {extract_python-0.7.2 → extract_python-0.8.0}/benches/constants.py +0 -0
- {extract_python-0.7.2 → extract_python-0.8.0}/data/.gitignore +0 -0
- {extract_python-0.7.2 → extract_python-0.8.0}/extract_python/__init__.py +0 -0
- {extract_python-0.7.2 → extract_python-0.8.0}/extract_python/constants.py +0 -0
|
@@ -0,0 +1 @@
|
|
|
1
|
+
3.14
|
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
2
|
Name: extract-python
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.8.0
|
|
4
4
|
Summary: Structured content extraction
|
|
5
5
|
Project-URL: Homepage, https://github.com/ICIJ/extract-python
|
|
6
6
|
Project-URL: Repository, https://github.com/ICIJ/extract-python
|
|
@@ -9,6 +9,7 @@ Author-email: Clément Doumouro <cdoumouro@icij.org>
|
|
|
9
9
|
Requires-Python: <3.15,>=3.13
|
|
10
10
|
Requires-Dist: extract-core~=0.7.0
|
|
11
11
|
Requires-Dist: icij-common~=0.8.2
|
|
12
|
+
Requires-Dist: pympler~=1.1
|
|
12
13
|
Provides-Extra: benches
|
|
13
14
|
Requires-Dist: html2image~=2.0.7; extra == 'benches'
|
|
14
15
|
Requires-Dist: markdown2>=2.5.4; extra == 'benches'
|
|
@@ -24,3 +25,6 @@ Requires-Dist: mineru[pipeline,vlm]~=3.2; extra == 'mineru'
|
|
|
24
25
|
Requires-Dist: pydantic-extra-types[pycountry]~=2.11; extra == 'mineru'
|
|
25
26
|
Requires-Dist: python-pptx~=1.0; extra == 'mineru'
|
|
26
27
|
Requires-Dist: six~=1.17; extra == 'mineru'
|
|
28
|
+
Description-Content-Type: text/markdown
|
|
29
|
+
|
|
30
|
+
ô
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
ô
|
|
@@ -1,20 +1,27 @@
|
|
|
1
1
|
import asyncio
|
|
2
2
|
import json
|
|
3
3
|
import logging
|
|
4
|
+
import operator
|
|
4
5
|
import shutil
|
|
5
6
|
import tempfile
|
|
6
|
-
from collections.abc import
|
|
7
|
-
from
|
|
7
|
+
from collections.abc import AsyncIterable, Iterable
|
|
8
|
+
from contextlib import AbstractContextManager
|
|
9
|
+
from functools import partial, reduce
|
|
8
10
|
from pathlib import Path
|
|
9
11
|
from typing import Any, Self
|
|
10
12
|
|
|
11
|
-
from docling.datamodel.document import
|
|
13
|
+
from docling.datamodel.document import ConversionAssets
|
|
12
14
|
from docling.datamodel.pipeline_options import PipelineOptions
|
|
15
|
+
from docling.datamodel.settings import (
|
|
16
|
+
DEFAULT_PAGE_RANGE,
|
|
17
|
+
AppSettings,
|
|
18
|
+
scoped,
|
|
19
|
+
)
|
|
13
20
|
from docling.document_converter import DocumentConverter, FormatOption
|
|
21
|
+
from docling_core.types import DoclingDocument
|
|
14
22
|
|
|
15
23
|
# TODO: this is long to load improve it
|
|
16
24
|
from docling_core.types.doc import ImageRefMode
|
|
17
|
-
from docling_core.types.io import DocumentStream
|
|
18
25
|
from extract_core import (
|
|
19
26
|
BaseModel,
|
|
20
27
|
DoclingFormatOption,
|
|
@@ -33,7 +40,14 @@ from pydantic import ConfigDict, field_serializer
|
|
|
33
40
|
from pydantic_core.core_schema import SerializerFunctionWrapHandler
|
|
34
41
|
|
|
35
42
|
from .constants import ARTIFACTS, DEFAULT_MD_PAGE_SEP
|
|
36
|
-
from .utils import
|
|
43
|
+
from .utils import (
|
|
44
|
+
Range,
|
|
45
|
+
ResultBuffer,
|
|
46
|
+
batch_per_pages,
|
|
47
|
+
chdir,
|
|
48
|
+
path_to_artifacts_dirname,
|
|
49
|
+
write_pages,
|
|
50
|
+
)
|
|
37
51
|
|
|
38
52
|
logger = logging.getLogger(__name__)
|
|
39
53
|
|
|
@@ -61,54 +75,112 @@ class DoclingPipeline(Pipeline):
|
|
|
61
75
|
|
|
62
76
|
async def extract_content(
|
|
63
77
|
self, docs: Iterable[InputDoc], output_format: OutputFormat, output_path: Path
|
|
64
|
-
) ->
|
|
65
|
-
|
|
66
|
-
|
|
78
|
+
) -> AsyncIterable[Result]:
|
|
79
|
+
settings = self._config.settings
|
|
80
|
+
logger.info("starting extraction with settings: %s", settings)
|
|
81
|
+
with self._scoped_settings, self._result_buffer as buffer:
|
|
82
|
+
max_page_batches = settings.perf.max_page_batches
|
|
83
|
+
page_batch_size = settings.perf.page_batch_size
|
|
84
|
+
batches = batch_per_pages(
|
|
85
|
+
docs, page_batch_size, max_page_batches=max_page_batches
|
|
86
|
+
)
|
|
87
|
+
for batch in batches:
|
|
88
|
+
page_range = list({pages.page_range for pages in batch})
|
|
89
|
+
if len(page_range) > 1:
|
|
90
|
+
msg = "convert_all only accept 1 page range for all docs"
|
|
91
|
+
raise ValueError(msg)
|
|
92
|
+
page_range = page_range[0]
|
|
93
|
+
page_range = _docling_range(page_range)
|
|
94
|
+
docling_docs = (pages.doc.to_docling() for pages in batch)
|
|
95
|
+
outputs = self._converter.convert_all(
|
|
96
|
+
docling_docs, raises_on_error=False, page_range=page_range
|
|
97
|
+
)
|
|
98
|
+
processed = iter(batch)
|
|
99
|
+
sentinel = object()
|
|
100
|
+
while True:
|
|
101
|
+
res = await asyncio.to_thread(next, outputs, sentinel)
|
|
102
|
+
if res is sentinel:
|
|
103
|
+
break
|
|
104
|
+
pages = next(processed)
|
|
105
|
+
buffer.add(pages, res)
|
|
106
|
+
if buffer.is_complete(pages.doc_idx):
|
|
107
|
+
doc_pages = buffer.pop_complete(pages.doc_idx)
|
|
108
|
+
yield _to_result(
|
|
109
|
+
doc_pages, pages.doc, output_format, output_path=output_path
|
|
110
|
+
)
|
|
111
|
+
|
|
112
|
+
@property
|
|
113
|
+
def _scoped_settings(self) -> AbstractContextManager[AppSettings]:
|
|
114
|
+
settings = self._config.settings
|
|
115
|
+
docling_settings = scoped(
|
|
116
|
+
perf=settings.perf, debug=settings.debug, inference=settings.inference
|
|
117
|
+
)
|
|
118
|
+
return docling_settings
|
|
119
|
+
|
|
120
|
+
@property
|
|
121
|
+
def _result_buffer(self) -> ResultBuffer:
|
|
122
|
+
buffer = ResultBuffer(
|
|
123
|
+
max_size_bytes=self._config.result_buffer.max_size,
|
|
124
|
+
root=self._config.result_buffer.root,
|
|
125
|
+
save_fn=_save_conversion_result,
|
|
126
|
+
load_fn=_load_conversion_result,
|
|
127
|
+
)
|
|
128
|
+
return buffer
|
|
129
|
+
|
|
67
130
|
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
res = await asyncio.to_thread(next, outputs, sentinel)
|
|
71
|
-
if res is sentinel:
|
|
72
|
-
return
|
|
73
|
-
doc = next(docs)
|
|
74
|
-
yield _to_result(res, doc, output_format, output_path=output_path)
|
|
131
|
+
def _save_conversion_result(res: ConversionAssets, path: Path) -> None:
|
|
132
|
+
return res.save(filename=path)
|
|
75
133
|
|
|
76
134
|
|
|
77
|
-
def
|
|
78
|
-
|
|
79
|
-
yield d.to_docling()
|
|
135
|
+
def _load_conversion_result(path: Path) -> ConversionAssets:
|
|
136
|
+
return ConversionAssets.load(path)
|
|
80
137
|
|
|
81
138
|
|
|
82
139
|
def _to_result(
|
|
83
|
-
|
|
84
|
-
|
|
140
|
+
buffer: list[ConversionAssets],
|
|
141
|
+
input_doc: InputDoc,
|
|
85
142
|
output_format: OutputFormat,
|
|
86
143
|
output_path: Path,
|
|
87
144
|
**kwargs,
|
|
88
145
|
) -> Result:
|
|
146
|
+
import numpy as np # noqa: PLC0415
|
|
147
|
+
|
|
148
|
+
if not buffer:
|
|
149
|
+
raise ValueError("empty buffer")
|
|
150
|
+
merged = DoclingDocument.concatenate([res.document for res in buffer])
|
|
89
151
|
output_path.mkdir(parents=True, exist_ok=True)
|
|
152
|
+
status = reduce(operator.iadd, (Status.from_docling(d.status) for d in buffer))
|
|
153
|
+
# TODO: implement confidence weight
|
|
154
|
+
confidence = np.mean([res.confidence.mean_score for res in buffer])
|
|
90
155
|
output = None
|
|
91
|
-
status = Status.from_docling(res.status)
|
|
92
156
|
if status.allows_conversion:
|
|
93
157
|
match output_format:
|
|
94
158
|
case OutputFormat.MARKDOWN:
|
|
95
|
-
output = _to_markdown_doc(
|
|
159
|
+
output = _to_markdown_doc(
|
|
160
|
+
merged,
|
|
161
|
+
input_path=input_doc.path,
|
|
162
|
+
output_path=output_path,
|
|
163
|
+
confidence=confidence,
|
|
164
|
+
**kwargs,
|
|
165
|
+
)
|
|
96
166
|
case _:
|
|
97
167
|
raise NotImplementedError(f"unsupported output format {output_format}")
|
|
98
|
-
errors = [Error.from_docling(e) for e in res.errors]
|
|
99
|
-
input_doc = input_document.without_content()
|
|
168
|
+
errors = [Error.from_docling(e) for res in buffer for e in res.errors]
|
|
100
169
|
return Result(input=input_doc, status=status, errors=errors, output=output)
|
|
101
170
|
|
|
102
171
|
|
|
103
172
|
def _to_markdown_doc(
|
|
104
|
-
|
|
173
|
+
doc: DoclingDocument,
|
|
174
|
+
input_path: Path,
|
|
175
|
+
*,
|
|
105
176
|
output_path: Path,
|
|
106
177
|
page_sep: str = DEFAULT_MD_PAGE_SEP,
|
|
178
|
+
confidence: float,
|
|
107
179
|
**kwargs,
|
|
108
180
|
) -> MarkdownDoc:
|
|
109
181
|
# TODO: Should we add a hash to avoid collision between files with same names
|
|
110
182
|
# nested in the tree structured
|
|
111
|
-
md_dir_name = path_to_artifacts_dirname(
|
|
183
|
+
md_dir_name = path_to_artifacts_dirname(input_path)
|
|
112
184
|
md_dir = output_path / md_dir_name
|
|
113
185
|
if md_dir.exists():
|
|
114
186
|
raise FileExistsError(f"directory {md_dir} already exists")
|
|
@@ -121,21 +193,21 @@ def _to_markdown_doc(
|
|
|
121
193
|
with chdir(tmp_dir):
|
|
122
194
|
# We do a chdir to bypass a Docling bug which only allows to maintain
|
|
123
195
|
# relative image ref when saving the markdown to a relative path
|
|
124
|
-
pages = _docling_pages_it(
|
|
196
|
+
pages = _docling_pages_it(doc, current_page_path, **kwargs)
|
|
125
197
|
with md_path.open("wb") as f:
|
|
126
198
|
pages = write_pages(pages, page_sep, f)
|
|
127
199
|
# Clean up the tmp page file before move everything to the end destination
|
|
128
200
|
current_page_path.unlink(missing_ok=True)
|
|
129
201
|
shutil.move(tmp_dir, md_dir)
|
|
130
|
-
return MarkdownDoc(path=Path(md_dir_name), pages=pages)
|
|
202
|
+
return MarkdownDoc(path=Path(md_dir_name), pages=pages, confidence=confidence)
|
|
131
203
|
|
|
132
204
|
|
|
133
205
|
def _docling_pages_it(
|
|
134
|
-
|
|
206
|
+
doc: DoclingDocument, output_path: Path, **kwargs
|
|
135
207
|
) -> Iterable[str]:
|
|
136
|
-
n_pages = len(
|
|
208
|
+
n_pages = len(doc.pages)
|
|
137
209
|
for page_i in range(n_pages):
|
|
138
|
-
|
|
210
|
+
doc.save_as_markdown(
|
|
139
211
|
output_path,
|
|
140
212
|
page_no=page_i + 1,
|
|
141
213
|
image_mode=ImageRefMode.REFERENCED,
|
|
@@ -193,3 +265,9 @@ class SerializableFormatOptions(DoclingFormatOption):
|
|
|
193
265
|
serialized["table_structure_options"] = dict()
|
|
194
266
|
serialized["table_structure_options"]["kind"] = table_structure_opts.kind
|
|
195
267
|
return serialized
|
|
268
|
+
|
|
269
|
+
|
|
270
|
+
def _docling_range(rng: Range | None) -> tuple[int, int]:
|
|
271
|
+
if rng is None:
|
|
272
|
+
return DEFAULT_PAGE_RANGE
|
|
273
|
+
return (rng[0] + 1, rng[1] + 1)
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import asyncio
|
|
2
2
|
import gc
|
|
3
|
-
from collections.abc import
|
|
3
|
+
from collections.abc import AsyncIterable, Iterable
|
|
4
4
|
from copy import deepcopy
|
|
5
5
|
from pathlib import Path
|
|
6
6
|
from typing import TYPE_CHECKING
|
|
@@ -30,7 +30,7 @@ _MARKER_CONVERSION_ERRORS = tuple()
|
|
|
30
30
|
class MarkerPipeline(Pipeline):
|
|
31
31
|
async def extract_content(
|
|
32
32
|
self, docs: Iterable[InputDoc], output_format: OutputFormat, output_path: Path
|
|
33
|
-
) ->
|
|
33
|
+
) -> AsyncIterable[Result]:
|
|
34
34
|
from marker.config.parser import ConfigParser # noqa: PLC0415
|
|
35
35
|
from marker.converters.pdf import PdfConverter # noqa: PLC0415
|
|
36
36
|
from marker.models import create_model_dict # noqa: PLC0415
|
|
@@ -67,8 +67,7 @@ async def _process_doc(
|
|
|
67
67
|
)
|
|
68
68
|
case _:
|
|
69
69
|
raise NotImplementedError(f"unsupported output format {output_format}")
|
|
70
|
-
|
|
71
|
-
return Result(input=input_doc, status=Status.SUCCESS, output=output)
|
|
70
|
+
return Result(input=doc, status=Status.SUCCESS, output=output)
|
|
72
71
|
|
|
73
72
|
|
|
74
73
|
def _to_markdown_doc(
|
|
@@ -96,4 +95,4 @@ def _to_markdown_doc(
|
|
|
96
95
|
md_path = md_path.with_suffix(OutputFormat.MARKDOWN.value)
|
|
97
96
|
with md_path.open("wb") as f:
|
|
98
97
|
pages = write_pages(pages, page_sep, f)
|
|
99
|
-
return MarkdownDoc(path=Path(md_dir_name), pages=pages)
|
|
98
|
+
return MarkdownDoc(path=Path(md_dir_name), pages=pages, confidence=None)
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import json
|
|
2
2
|
import os
|
|
3
3
|
import shutil
|
|
4
|
-
from collections.abc import
|
|
4
|
+
from collections.abc import AsyncIterable, Callable, Iterable
|
|
5
5
|
from functools import partial
|
|
6
6
|
from pathlib import Path
|
|
7
7
|
from tempfile import TemporaryDirectory
|
|
@@ -34,7 +34,7 @@ class MinerUPipeline(Pipeline):
|
|
|
34
34
|
|
|
35
35
|
async def extract_content(
|
|
36
36
|
self, docs: Iterable[InputDoc], output_format: OutputFormat, output_path: Path
|
|
37
|
-
) ->
|
|
37
|
+
) -> AsyncIterable[Result]:
|
|
38
38
|
from mineru.cli.common import aio_do_parse # noqa: PLC0415
|
|
39
39
|
|
|
40
40
|
with reset_env():
|
|
@@ -126,15 +126,13 @@ def _process_doc(
|
|
|
126
126
|
raise NotImplementedError(f"unsupported output format {output_format}")
|
|
127
127
|
middle_json_path = res_path / f"{doc.path.name}_middle.json"
|
|
128
128
|
middle_json = json.loads(middle_json_path.read_text())
|
|
129
|
-
pdf_info = middle_json["pdf_info"]
|
|
130
129
|
shutil.move(res_path / "images", artifacts_dir)
|
|
131
|
-
output = dump_content_fn(
|
|
132
|
-
|
|
133
|
-
return Result(input=input_doc, status=Status.SUCCESS, output=output)
|
|
130
|
+
output = dump_content_fn(middle_json)
|
|
131
|
+
return Result(input=doc, status=Status.SUCCESS, output=output)
|
|
134
132
|
|
|
135
133
|
|
|
136
134
|
def _dump_md_content(
|
|
137
|
-
|
|
135
|
+
middle_json: dict,
|
|
138
136
|
*,
|
|
139
137
|
md_make_fn: MDMakeFunction,
|
|
140
138
|
page_sep: str = DEFAULT_MD_PAGE_SEP,
|
|
@@ -145,11 +143,72 @@ def _dump_md_content(
|
|
|
145
143
|
) -> ConversionOutput:
|
|
146
144
|
from mineru.utils.enum_class import MakeMode # noqa: PLC0415
|
|
147
145
|
|
|
146
|
+
pdf_info = middle_json["pdf_info"]
|
|
148
147
|
if md_make_mode is None:
|
|
149
148
|
md_make_mode = MakeMode.MM_MD
|
|
150
149
|
pages = (md_make_fn([p], md_make_mode, str(im_dir)) for p in pdf_info)
|
|
151
150
|
with md_path.open("wb") as f:
|
|
152
151
|
pages = write_pages(pages, page_sep, f)
|
|
153
152
|
output_path = md_path.parent.relative_to(output_path)
|
|
154
|
-
|
|
153
|
+
confidence = _mineru_confidence(pdf_info)
|
|
154
|
+
output = ConversionOutput(path=output_path, pages=pages, confidence=confidence)
|
|
155
155
|
return output
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def _mineru_confidence(pdf_info: list[dict]) -> float:
|
|
159
|
+
if not pdf_info:
|
|
160
|
+
return 1.0
|
|
161
|
+
block_conf = _mineru_block_confidence(pdf_info)
|
|
162
|
+
line_config = _mineru_line_confidence(pdf_info)
|
|
163
|
+
return (block_conf + line_config) / 2.0
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def _mineru_block_confidence(pdf_info: list[dict]) -> float:
|
|
167
|
+
import numpy as np # noqa: PLC0415
|
|
168
|
+
|
|
169
|
+
scores = []
|
|
170
|
+
for info in pdf_info:
|
|
171
|
+
for block in info["para_blocks"]:
|
|
172
|
+
score = block.get("score")
|
|
173
|
+
if score is not None:
|
|
174
|
+
scores.append(score)
|
|
175
|
+
if scores:
|
|
176
|
+
return np.average(scores)
|
|
177
|
+
return 1.0
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
def _mineru_line_confidence(pdf_info: list[dict]) -> float:
|
|
181
|
+
import numpy as np # noqa: PLC0415
|
|
182
|
+
|
|
183
|
+
scores = []
|
|
184
|
+
lengths = []
|
|
185
|
+
for info in pdf_info:
|
|
186
|
+
for block in info["para_blocks"]:
|
|
187
|
+
for line in block.get("lines", []):
|
|
188
|
+
for span in line["spans"]:
|
|
189
|
+
score = span.get("score")
|
|
190
|
+
if score is not None:
|
|
191
|
+
scores.append(score)
|
|
192
|
+
lengths.append(len(span["content"]))
|
|
193
|
+
if scores:
|
|
194
|
+
return np.average(scores, weights=lengths)
|
|
195
|
+
return 1.0
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
def _parse_block(block: dict) -> tuple[list[float], list[float]]:
|
|
199
|
+
if "lines" in block:
|
|
200
|
+
scores = []
|
|
201
|
+
lengths = []
|
|
202
|
+
for line in block.get("lines", []):
|
|
203
|
+
for span in line["spans"]:
|
|
204
|
+
score = span.get("score")
|
|
205
|
+
if score is not None:
|
|
206
|
+
scores.append(score)
|
|
207
|
+
lengths.append(len(span["content"]))
|
|
208
|
+
return scores, lengths
|
|
209
|
+
if "blocs" in block:
|
|
210
|
+
scores, lengths = (_parse_block(b) for b in block["blocs"])
|
|
211
|
+
scores = sum(*scores, start=[])
|
|
212
|
+
lengths = sum(*lengths, start=[])
|
|
213
|
+
return scores, lengths
|
|
214
|
+
raise NotImplementedError(f"unsupported block: {block}")
|
|
@@ -0,0 +1,299 @@
|
|
|
1
|
+
import gc
|
|
2
|
+
import itertools
|
|
3
|
+
import logging
|
|
4
|
+
import os
|
|
5
|
+
import shutil
|
|
6
|
+
import uuid
|
|
7
|
+
from collections import defaultdict, deque
|
|
8
|
+
from collections.abc import Callable, Generator, Iterable, Iterator
|
|
9
|
+
from contextlib import contextmanager
|
|
10
|
+
from copy import copy
|
|
11
|
+
from dataclasses import dataclass
|
|
12
|
+
from functools import wraps
|
|
13
|
+
from itertools import tee
|
|
14
|
+
from pathlib import Path, PurePath
|
|
15
|
+
from tempfile import TemporaryDirectory
|
|
16
|
+
from types import TracebackType
|
|
17
|
+
from typing import BinaryIO, Protocol, Self
|
|
18
|
+
|
|
19
|
+
from extract_core import Error, InputDoc, Pages, Result, Status
|
|
20
|
+
from pympler import asizeof
|
|
21
|
+
|
|
22
|
+
logger = logging.getLogger(__name__)
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def map_and_preserve[I, R](
|
|
26
|
+
fn: Callable[[Iterable[I]], Iterator[R]], inputs: Iterable[I]
|
|
27
|
+
) -> tuple[Iterable[I], Iterator[R]]:
|
|
28
|
+
save_inputs, function_inputs = tee(inputs)
|
|
29
|
+
outputs = iter(fn(function_inputs))
|
|
30
|
+
return save_inputs, outputs
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def path_to_artifacts_dirname(path: PurePath, sep: str = "_") -> str:
|
|
34
|
+
dirname = f"{path.name[: -len(path.suffix)]}"
|
|
35
|
+
ext = path.suffix
|
|
36
|
+
if ext:
|
|
37
|
+
dirname += sep + ext[1:]
|
|
38
|
+
return dirname
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
class DocProcessingFn(Protocol):
|
|
42
|
+
def __call__(self, doc: InputDoc, *arg, **kwargs) -> Result: ...
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def report_recoverable_errors(
|
|
46
|
+
recoverable_errors: tuple[type[Exception], ...] = tuple(),
|
|
47
|
+
) -> Callable[[DocProcessingFn], DocProcessingFn]:
|
|
48
|
+
def make_decorator(f: DocProcessingFn) -> DocProcessingFn:
|
|
49
|
+
@wraps(f)
|
|
50
|
+
def wrapped(doc: InputDoc, *args, **kwargs) -> Result:
|
|
51
|
+
try:
|
|
52
|
+
return f(doc, *args, **kwargs)
|
|
53
|
+
except recoverable_errors as e:
|
|
54
|
+
error = Error.from_exception(e)
|
|
55
|
+
return Result(
|
|
56
|
+
input=doc, status=Status.FAILURE, errors=[error], output=None
|
|
57
|
+
)
|
|
58
|
+
|
|
59
|
+
return wrapped
|
|
60
|
+
|
|
61
|
+
return make_decorator
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
@contextmanager
|
|
65
|
+
def chdir(path: Path) -> Generator[None]:
|
|
66
|
+
cwd = Path.cwd()
|
|
67
|
+
try:
|
|
68
|
+
os.chdir(path)
|
|
69
|
+
yield
|
|
70
|
+
finally:
|
|
71
|
+
os.chdir(cwd)
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
@contextmanager
|
|
75
|
+
def reset_env() -> Generator[None]:
|
|
76
|
+
old_env = copy(dict(os.environ))
|
|
77
|
+
try:
|
|
78
|
+
yield
|
|
79
|
+
finally:
|
|
80
|
+
os.environ.clear()
|
|
81
|
+
os.environ.update(old_env)
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def write_pages(pages: Iterable[str], page_sep: str, out: BinaryIO) -> Pages:
|
|
85
|
+
pages_byte_sizes = []
|
|
86
|
+
pages = iter(pages)
|
|
87
|
+
content = None
|
|
88
|
+
for p in pages:
|
|
89
|
+
if content:
|
|
90
|
+
pages_byte_sizes.append(out.write((content + page_sep).encode()))
|
|
91
|
+
content = p
|
|
92
|
+
if content:
|
|
93
|
+
pages_byte_sizes.append(out.write(content.encode()))
|
|
94
|
+
return Pages.from_pages_bytes_sizes(pages_byte_sizes)
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
Range = tuple[int, int]
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
@dataclass(frozen=True)
|
|
101
|
+
class ProcessedPages:
|
|
102
|
+
doc: InputDoc
|
|
103
|
+
doc_idx: int
|
|
104
|
+
page_range: Range | None = None
|
|
105
|
+
|
|
106
|
+
@property
|
|
107
|
+
def page_length(self) -> int:
|
|
108
|
+
if self.page_range is None:
|
|
109
|
+
return self.doc.n_pages
|
|
110
|
+
return self.page_range[1] - self.page_range[0]
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
class ResultBuffer[R]:
|
|
114
|
+
def __init__(
|
|
115
|
+
self,
|
|
116
|
+
max_size_bytes: int,
|
|
117
|
+
save_fn: Callable[[R, Path], None],
|
|
118
|
+
*,
|
|
119
|
+
load_fn: Callable[[Path], R],
|
|
120
|
+
root: Path | None = None,
|
|
121
|
+
):
|
|
122
|
+
self._max_bytes = max_size_bytes
|
|
123
|
+
self._save_fn = save_fn
|
|
124
|
+
self._load_fn = load_fn
|
|
125
|
+
self._tmp_dir = None
|
|
126
|
+
if root is None:
|
|
127
|
+
self._tmp_dir = TemporaryDirectory()
|
|
128
|
+
root = Path(self._tmp_dir.name)
|
|
129
|
+
self._root = root
|
|
130
|
+
self.__fs_buffer_path = None
|
|
131
|
+
self._mem_buffer: dict[int, list[R | Path]] = defaultdict(list)
|
|
132
|
+
self._missing_pages: dict[int, int] = dict()
|
|
133
|
+
self._current_size: int = 0
|
|
134
|
+
|
|
135
|
+
def __enter__(self) -> Self:
|
|
136
|
+
if self._tmp_dir is not None:
|
|
137
|
+
self._tmp_dir.__enter__()
|
|
138
|
+
self.__fs_buffer_path = self._root / uuid.uuid4().hex
|
|
139
|
+
self.__fs_buffer_path.mkdir()
|
|
140
|
+
return self
|
|
141
|
+
|
|
142
|
+
def __exit__(
|
|
143
|
+
self,
|
|
144
|
+
exc_type: type[BaseException] | None,
|
|
145
|
+
exc_val: BaseException | None,
|
|
146
|
+
exc_tb: TracebackType | None,
|
|
147
|
+
) -> None:
|
|
148
|
+
if self._mem_buffer or self._missing_pages:
|
|
149
|
+
logger.warning("closing an non empty buffer")
|
|
150
|
+
if self._tmp_dir is not None:
|
|
151
|
+
self._tmp_dir.__exit__(exc_type, exc_val, exc_tb)
|
|
152
|
+
if self._fs_buffer_path.exists():
|
|
153
|
+
shutil.rmtree(self._fs_buffer_path)
|
|
154
|
+
self._mem_buffer = dict()
|
|
155
|
+
self._missing_pages = dict()
|
|
156
|
+
|
|
157
|
+
@property
|
|
158
|
+
def _fs_buffer_path(self) -> Path:
|
|
159
|
+
if not self.__fs_buffer_path:
|
|
160
|
+
msg = (
|
|
161
|
+
f"inconsistent state, {ResultBuffer.__class__.__name__} is a context"
|
|
162
|
+
f" manager, call __enter__ before using it"
|
|
163
|
+
)
|
|
164
|
+
raise ValueError(msg)
|
|
165
|
+
return self.__fs_buffer_path
|
|
166
|
+
|
|
167
|
+
def add(self, processed: ProcessedPages, result: R) -> None:
|
|
168
|
+
size = asizeof.asizeof(result)
|
|
169
|
+
if self._current_size + size > self._max_bytes:
|
|
170
|
+
path = self._page_path(processed.doc_idx)
|
|
171
|
+
self._save_fn(result, path)
|
|
172
|
+
result = path
|
|
173
|
+
else:
|
|
174
|
+
self._current_size += size
|
|
175
|
+
self._mem_buffer[processed.doc_idx].append(result)
|
|
176
|
+
if processed.doc_idx not in self._missing_pages:
|
|
177
|
+
self._missing_pages[processed.doc_idx] = processed.doc.n_pages
|
|
178
|
+
self._missing_pages[processed.doc_idx] -= processed.page_length
|
|
179
|
+
|
|
180
|
+
def is_complete(self, doc: int) -> bool:
|
|
181
|
+
return self._missing_pages[doc] == 0
|
|
182
|
+
|
|
183
|
+
def pop_complete(self, doc: int) -> list[R]:
|
|
184
|
+
if not self.is_complete(doc):
|
|
185
|
+
raise ValueError(f"{doc} is incomplete")
|
|
186
|
+
pages = self._mem_buffer.pop(doc)
|
|
187
|
+
self._missing_pages.pop(doc)
|
|
188
|
+
for i, page in enumerate(pages):
|
|
189
|
+
if isinstance(page, Path):
|
|
190
|
+
page = self._load_fn(page) # noqa: PLW2901
|
|
191
|
+
else:
|
|
192
|
+
self._current_size -= asizeof.asizeof(page)
|
|
193
|
+
pages[i] = page
|
|
194
|
+
return pages
|
|
195
|
+
|
|
196
|
+
def _page_path(self, doc_id: int) -> Path:
|
|
197
|
+
pages = self._mem_buffer[doc_id]
|
|
198
|
+
return self._fs_buffer_path / f"doc-{doc_id}-pages-{len(pages)}"
|
|
199
|
+
|
|
200
|
+
def __len__(self) -> int:
|
|
201
|
+
return len(self._mem_buffer)
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
# The converter process page_batch_size in parallel (GPU sees page_batch_size batches).
|
|
205
|
+
#
|
|
206
|
+
# The batching tradeoff is: avoid calling convert_all to many times vs. releasing the
|
|
207
|
+
# GIL often enough.
|
|
208
|
+
#
|
|
209
|
+
# Calling convert_all to many times on the same doc results in overhead. Each time we
|
|
210
|
+
# call the function, we create a doc processing backend + reload the doc.
|
|
211
|
+
#
|
|
212
|
+
# On the other hand we have to use a reasonable max_page_batches otherwise we process
|
|
213
|
+
# all the stream in a single call and take the risk to lock the GIL for too long. Some
|
|
214
|
+
# docling ops are sadly not async (numpy or torch inference are, but document loading
|
|
215
|
+
# and conversion aren't, so the asyncio.to_thread is not helping)
|
|
216
|
+
def batch_per_pages(
|
|
217
|
+
docs: Iterable[InputDoc],
|
|
218
|
+
page_batch_size: int,
|
|
219
|
+
*,
|
|
220
|
+
max_page_batches: int,
|
|
221
|
+
chunk_size: int = 1000,
|
|
222
|
+
) -> Iterable[tuple[ProcessedPages]]:
|
|
223
|
+
# convert_all only accept to process docs on the exact same page_range
|
|
224
|
+
#
|
|
225
|
+
# We collect by chunk to avoid collecting too many inputs, input docs are
|
|
226
|
+
# lightweight anyway so memory impact should stay limited
|
|
227
|
+
#
|
|
228
|
+
# Additionally, results can be output unordered and partial results are buffered
|
|
229
|
+
# it's OK to process doc pages unordered.
|
|
230
|
+
# TODO: if it's not OK to sort because inputs is l
|
|
231
|
+
max_pages = page_batch_size * max_page_batches
|
|
232
|
+
docs = itertools.batched(docs, chunk_size, strict=False)
|
|
233
|
+
for chunk in docs:
|
|
234
|
+
short_docs = [d for d in chunk if d.n_pages <= max_pages]
|
|
235
|
+
long_docs = [d for d in chunk if d.n_pages > max_pages]
|
|
236
|
+
del chunk
|
|
237
|
+
gc.collect()
|
|
238
|
+
# Bin fill for docs smaller than max_pages
|
|
239
|
+
offset = yield from _bin_fill(short_docs, max_pages=max_pages)
|
|
240
|
+
# otherwise we just yield chunks of max_pages except the last chunk which is
|
|
241
|
+
# grouped by page_range
|
|
242
|
+
yield from _by_page_ranges(long_docs, max_pages=max_pages, offset=offset)
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
def _bin_fill(
|
|
246
|
+
docs: Iterable[InputDoc], max_pages: int, offset: int = 0
|
|
247
|
+
) -> Generator[tuple[ProcessedPages], None, int]:
|
|
248
|
+
bins = defaultdict(deque)
|
|
249
|
+
doc_idx = offset
|
|
250
|
+
for doc in docs:
|
|
251
|
+
if doc.n_pages > max_pages:
|
|
252
|
+
msg = f"expected docs to have <= {max_pages} pages"
|
|
253
|
+
raise ValueError(msg)
|
|
254
|
+
pages = ProcessedPages(doc=doc, doc_idx=doc_idx)
|
|
255
|
+
doc_idx += 1
|
|
256
|
+
available_space = (s for s in sorted(bins.keys()) if doc.n_pages <= s)
|
|
257
|
+
available_space = next(available_space, max_pages)
|
|
258
|
+
selected = bins[available_space]
|
|
259
|
+
selected = selected.pop() if selected else []
|
|
260
|
+
selected.append(pages)
|
|
261
|
+
available_space -= doc.n_pages
|
|
262
|
+
if available_space == 0:
|
|
263
|
+
yield tuple(selected)
|
|
264
|
+
continue
|
|
265
|
+
bins[available_space].append(selected)
|
|
266
|
+
for range_bins in bins.values():
|
|
267
|
+
for b in range_bins:
|
|
268
|
+
yield tuple(b)
|
|
269
|
+
return doc_idx
|
|
270
|
+
|
|
271
|
+
|
|
272
|
+
def _by_page_ranges(
|
|
273
|
+
docs: Iterable[InputDoc], max_pages: int, offset: int = 0
|
|
274
|
+
) -> Generator[tuple[ProcessedPages], None, int]:
|
|
275
|
+
by_range = defaultdict(list)
|
|
276
|
+
doc_idx = offset
|
|
277
|
+
for doc in docs:
|
|
278
|
+
if doc.n_pages < max_pages:
|
|
279
|
+
msg = f"expected docs to have >= {max_pages} pages"
|
|
280
|
+
raise ValueError(msg)
|
|
281
|
+
|
|
282
|
+
for i in range(0, doc.n_pages, max_pages):
|
|
283
|
+
start = i
|
|
284
|
+
end = min(start + max_pages, doc.n_pages)
|
|
285
|
+
rng = (start, end)
|
|
286
|
+
rng_size = end - start
|
|
287
|
+
alone_in_batch = rng_size == max_pages
|
|
288
|
+
pages = ProcessedPages(doc=doc, doc_idx=doc_idx, page_range=rng)
|
|
289
|
+
if alone_in_batch:
|
|
290
|
+
yield (pages,)
|
|
291
|
+
continue
|
|
292
|
+
by_range[rng].append(pages)
|
|
293
|
+
is_complete = len(by_range[rng]) == (max_pages // rng_size)
|
|
294
|
+
if is_complete:
|
|
295
|
+
yield tuple(by_range.pop(rng))
|
|
296
|
+
doc_idx += 1
|
|
297
|
+
for v in by_range.values():
|
|
298
|
+
yield tuple(v)
|
|
299
|
+
return doc_idx
|