pdf-figure-table-extractor 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,20 @@
1
+ name: Publish to PyPI
2
+
3
+ on:
4
+ release:
5
+ types: [published]
6
+
7
+ permissions:
8
+ contents: read
9
+
10
+ jobs:
11
+ publish:
12
+ runs-on: ubuntu-latest
13
+ environment: pypi
14
+ permissions:
15
+ id-token: write
16
+ steps:
17
+ - uses: actions/checkout@v4
18
+ - uses: astral-sh/setup-uv@v6
19
+ - run: uv build
20
+ - run: uv publish --trusted-publishing always
@@ -0,0 +1,22 @@
1
+ name: Tests
2
+
3
+ on:
4
+ push:
5
+ pull_request:
6
+
7
+ jobs:
8
+ test:
9
+ runs-on: ubuntu-latest
10
+ strategy:
11
+ matrix:
12
+ python-version: ["3.11", "3.12", "3.13", "3.14"]
13
+ steps:
14
+ - uses: actions/checkout@v4
15
+ - uses: astral-sh/setup-uv@v6
16
+ with:
17
+ python-version: ${{ matrix.python-version }}
18
+ enable-cache: true
19
+ - run: uv sync --locked --all-groups
20
+ - run: uv run ruff check .
21
+ - run: uv run pytest
22
+ - run: uv build
@@ -0,0 +1,8 @@
1
+ .venv/
2
+ .pytest_cache/
3
+ .ruff_cache/
4
+ __pycache__/
5
+ dist/
6
+ *.egg-info/
7
+ output/
8
+ .DS_Store
@@ -0,0 +1 @@
1
+ 3.11
@@ -0,0 +1,109 @@
1
+ Metadata-Version: 2.5
2
+ Name: pdf-figure-table-extractor
3
+ Version: 0.1.0
4
+ Summary: Extract tightly cropped, high-resolution figures and tables from PDF files.
5
+ Keywords: academic-papers,figure-extraction,pdf,table-extraction
6
+ Classifier: Development Status :: 4 - Beta
7
+ Classifier: Intended Audience :: Developers
8
+ Classifier: Intended Audience :: Science/Research
9
+ Classifier: Operating System :: OS Independent
10
+ Classifier: Programming Language :: Python :: 3
11
+ Classifier: Programming Language :: Python :: 3 :: Only
12
+ Classifier: Programming Language :: Python :: 3.11
13
+ Classifier: Programming Language :: Python :: 3.12
14
+ Classifier: Programming Language :: Python :: 3.13
15
+ Classifier: Programming Language :: Python :: 3.14
16
+ Classifier: Topic :: Scientific/Engineering :: Information Analysis
17
+ Classifier: Topic :: Utilities
18
+ Requires-Python: >=3.11
19
+ Requires-Dist: loguru<1,>=0.7
20
+ Requires-Dist: pymupdf<2,>=1.26
21
+ Description-Content-Type: text/markdown
22
+
23
+ # PDF Figure and Table Extractor
24
+
25
+ Extract labeled figures and tables from PDF files into tightly cropped,
26
+ high-resolution PNG images. The extractor uses PDF-native captions, vector
27
+ paths, embedded images, and text positions; it does not require OCR or a
28
+ vision model for text-based PDFs.
29
+
30
+ ## Features
31
+
32
+ - Extracts labeled `Figure`, `Fig.`, and `Table` regions.
33
+ - Handles figure captions below content and table titles above or below content.
34
+ - Supports vector figures, raster figures, ruled tables, and unruled tables.
35
+ - Includes complete captions by default.
36
+ - Produces a JSON manifest with page numbers, captions, crop boxes, and filenames.
37
+ - Re-running into the same directory removes only files recorded by the previous manifest.
38
+
39
+ ## Installation
40
+
41
+ Install the command-line tool with `uv`:
42
+
43
+ ```bash
44
+ uv tool install pdf-figure-table-extractor
45
+ ```
46
+
47
+ Or add it to a project:
48
+
49
+ ```bash
50
+ uv add pdf-figure-table-extractor
51
+ ```
52
+
53
+ ## Usage
54
+
55
+ ```bash
56
+ pdf-figure-table-extractor paper.pdf extracted-assets
57
+ ```
58
+
59
+ The default resolution is 400 DPI. For larger output:
60
+
61
+ ```bash
62
+ pdf-figure-table-extractor paper.pdf extracted-assets --dpi 600
63
+ ```
64
+
65
+ To omit figure captions while keeping table titles:
66
+
67
+ ```bash
68
+ pdf-figure-table-extractor paper.pdf extracted-assets --exclude-figure-captions
69
+ ```
70
+
71
+ The Python API is also available:
72
+
73
+ ```python
74
+ from pdf_figure_table_extractor import extract_assets
75
+
76
+ assets = extract_assets("paper.pdf", "extracted-assets", dpi=400)
77
+ ```
78
+
79
+ ## Output
80
+
81
+ ```text
82
+ extracted-assets/
83
+ ├── figure-001-1-page-001.png
84
+ ├── table-001-1-page-002.png
85
+ └── manifest.json
86
+ ```
87
+
88
+ `manifest.json` records the source PDF, DPI, detected caption, page number,
89
+ crop box, and output filename for every asset.
90
+
91
+ ## Scope and limitations
92
+
93
+ The extractor targets PDFs with machine-readable text and conventional labeled
94
+ captions. Scanned documents, rotated captions, unusual caption formats, and
95
+ figures spanning multiple pages may require OCR or manual correction. Always
96
+ visually inspect extracted assets before using them in publications or reports.
97
+
98
+ ## Development
99
+
100
+ ```bash
101
+ uv sync --all-groups
102
+ uv run ruff check .
103
+ uv run pytest
104
+ uv build
105
+ ```
106
+
107
+ Before publishing, choose and add an appropriate `LICENSE`, update the version,
108
+ and add repository URLs to `pyproject.toml` after the GitHub repository exists.
109
+ See [PUBLISHING.md](PUBLISHING.md) for the Trusted Publishing setup and release checklist.
@@ -0,0 +1,49 @@
1
+ # Publishing
2
+
3
+ This project is configured for PyPI Trusted Publishing through GitHub Actions.
4
+ No long-lived PyPI token is required.
5
+
6
+ ## Before the first release
7
+
8
+ 1. Choose a license, add a `LICENSE` file, and add its SPDX identifier to
9
+ `pyproject.toml`.
10
+ 2. Create the GitHub repository and push this directory as its root.
11
+ 3. Add the repository URLs to `pyproject.toml`, for example:
12
+
13
+ ```toml
14
+ [project.urls]
15
+ Homepage = "https://github.com/OWNER/pdf-figure-table-extractor"
16
+ Issues = "https://github.com/OWNER/pdf-figure-table-extractor/issues"
17
+ Source = "https://github.com/OWNER/pdf-figure-table-extractor"
18
+ ```
19
+
20
+ 4. On PyPI, create a pending trusted publisher for:
21
+ - PyPI project: `pdf-figure-table-extractor`
22
+ - GitHub owner and repository: the repository created above
23
+ - Workflow filename: `publish.yml`
24
+ - Environment name: `pypi`
25
+ 5. In the GitHub repository, create an environment named `pypi` and optionally
26
+ require manual approval for deployments.
27
+
28
+ ## Validate locally
29
+
30
+ ```bash
31
+ uv sync --locked --all-groups
32
+ uv run ruff check .
33
+ uv run pytest
34
+ uv build
35
+ uvx --from twine twine check dist/*
36
+ ```
37
+
38
+ ## Release
39
+
40
+ 1. Update `version` in `pyproject.toml` and run `uv lock`.
41
+ 2. Commit the version change and create a matching tag such as `v0.1.0`.
42
+ 3. Create a GitHub Release from that tag.
43
+ 4. The `Publish to PyPI` workflow builds the distributions and runs:
44
+
45
+ ```bash
46
+ uv publish --trusted-publishing always
47
+ ```
48
+
49
+ The package name is not reserved until the first successful PyPI upload.
@@ -0,0 +1,87 @@
1
+ # PDF Figure and Table Extractor
2
+
3
+ Extract labeled figures and tables from PDF files into tightly cropped,
4
+ high-resolution PNG images. The extractor uses PDF-native captions, vector
5
+ paths, embedded images, and text positions; it does not require OCR or a
6
+ vision model for text-based PDFs.
7
+
8
+ ## Features
9
+
10
+ - Extracts labeled `Figure`, `Fig.`, and `Table` regions.
11
+ - Handles figure captions below content and table titles above or below content.
12
+ - Supports vector figures, raster figures, ruled tables, and unruled tables.
13
+ - Includes complete captions by default.
14
+ - Produces a JSON manifest with page numbers, captions, crop boxes, and filenames.
15
+ - Re-running into the same directory removes only files recorded by the previous manifest.
16
+
17
+ ## Installation
18
+
19
+ Install the command-line tool with `uv`:
20
+
21
+ ```bash
22
+ uv tool install pdf-figure-table-extractor
23
+ ```
24
+
25
+ Or add it to a project:
26
+
27
+ ```bash
28
+ uv add pdf-figure-table-extractor
29
+ ```
30
+
31
+ ## Usage
32
+
33
+ ```bash
34
+ pdf-figure-table-extractor paper.pdf extracted-assets
35
+ ```
36
+
37
+ The default resolution is 400 DPI. For larger output:
38
+
39
+ ```bash
40
+ pdf-figure-table-extractor paper.pdf extracted-assets --dpi 600
41
+ ```
42
+
43
+ To omit figure captions while keeping table titles:
44
+
45
+ ```bash
46
+ pdf-figure-table-extractor paper.pdf extracted-assets --exclude-figure-captions
47
+ ```
48
+
49
+ The Python API is also available:
50
+
51
+ ```python
52
+ from pdf_figure_table_extractor import extract_assets
53
+
54
+ assets = extract_assets("paper.pdf", "extracted-assets", dpi=400)
55
+ ```
56
+
57
+ ## Output
58
+
59
+ ```text
60
+ extracted-assets/
61
+ ├── figure-001-1-page-001.png
62
+ ├── table-001-1-page-002.png
63
+ └── manifest.json
64
+ ```
65
+
66
+ `manifest.json` records the source PDF, DPI, detected caption, page number,
67
+ crop box, and output filename for every asset.
68
+
69
+ ## Scope and limitations
70
+
71
+ The extractor targets PDFs with machine-readable text and conventional labeled
72
+ captions. Scanned documents, rotated captions, unusual caption formats, and
73
+ figures spanning multiple pages may require OCR or manual correction. Always
74
+ visually inspect extracted assets before using them in publications or reports.
75
+
76
+ ## Development
77
+
78
+ ```bash
79
+ uv sync --all-groups
80
+ uv run ruff check .
81
+ uv run pytest
82
+ uv build
83
+ ```
84
+
85
+ Before publishing, choose and add an appropriate `LICENSE`, update the version,
86
+ and add repository URLs to `pyproject.toml` after the GitHub repository exists.
87
+ See [PUBLISHING.md](PUBLISHING.md) for the Trusted Publishing setup and release checklist.
@@ -0,0 +1,58 @@
1
+ [project]
2
+ name = "pdf-figure-table-extractor"
3
+ version = "0.1.0"
4
+ description = "Extract tightly cropped, high-resolution figures and tables from PDF files."
5
+ readme = "README.md"
6
+ requires-python = ">=3.11"
7
+ keywords = [
8
+ "academic-papers",
9
+ "figure-extraction",
10
+ "pdf",
11
+ "table-extraction",
12
+ ]
13
+ classifiers = [
14
+ "Development Status :: 4 - Beta",
15
+ "Intended Audience :: Developers",
16
+ "Intended Audience :: Science/Research",
17
+ "Operating System :: OS Independent",
18
+ "Programming Language :: Python :: 3",
19
+ "Programming Language :: Python :: 3 :: Only",
20
+ "Programming Language :: Python :: 3.11",
21
+ "Programming Language :: Python :: 3.12",
22
+ "Programming Language :: Python :: 3.13",
23
+ "Programming Language :: Python :: 3.14",
24
+ "Topic :: Scientific/Engineering :: Information Analysis",
25
+ "Topic :: Utilities",
26
+ ]
27
+ dependencies = [
28
+ "loguru>=0.7,<1",
29
+ "pymupdf>=1.26,<2",
30
+ ]
31
+
32
+ [project.scripts]
33
+ pdf-assets = "pdf_figure_table_extractor.cli:main"
34
+ pdf-figure-table-extractor = "pdf_figure_table_extractor.cli:main"
35
+
36
+ [dependency-groups]
37
+ dev = [
38
+ "pytest>=8,<10",
39
+ "ruff>=0.8",
40
+ ]
41
+
42
+ [build-system]
43
+ requires = ["hatchling>=1.27"]
44
+ build-backend = "hatchling.build"
45
+
46
+ [tool.hatch.build.targets.wheel]
47
+ packages = ["src/pdf_figure_table_extractor"]
48
+
49
+ [tool.pytest.ini_options]
50
+ pythonpath = ["src"]
51
+ testpaths = ["tests"]
52
+
53
+ [tool.ruff]
54
+ line-length = 100
55
+ target-version = "py311"
56
+
57
+ [tool.ruff.lint]
58
+ select = ["E", "F", "I", "UP"]
@@ -0,0 +1,9 @@
1
+ """High-resolution PDF figure and table extraction."""
2
+
3
+ from importlib.metadata import version
4
+
5
+ from .extractor import Asset, extract_assets
6
+
7
+ __version__ = version("pdf-figure-table-extractor")
8
+
9
+ __all__ = ["Asset", "__version__", "extract_assets"]
@@ -0,0 +1,48 @@
1
+ from __future__ import annotations
2
+
3
+ import argparse
4
+ from pathlib import Path
5
+
6
+ from loguru import logger
7
+
8
+ from .extractor import extract_assets
9
+ from .logging import configure_logging
10
+
11
+
12
+ def build_parser() -> argparse.ArgumentParser:
13
+ parser = argparse.ArgumentParser(
14
+ description="Extract figures and tables from a PDF as tightly cropped PNG files."
15
+ )
16
+ parser.add_argument("pdf", type=Path, help="Input PDF file")
17
+ parser.add_argument("output", type=Path, help="Output directory")
18
+ parser.add_argument("--dpi", type=int, default=400, help="Output resolution (default: 400)")
19
+ parser.add_argument(
20
+ "--exclude-figure-captions",
21
+ action="store_true",
22
+ help="Export figure bodies only. By default, complete figure captions are included.",
23
+ )
24
+ return parser
25
+
26
+
27
+ def main() -> None:
28
+ args = build_parser().parse_args()
29
+ configure_logging()
30
+ assets = extract_assets(
31
+ args.pdf,
32
+ args.output,
33
+ dpi=args.dpi,
34
+ include_figure_captions=not args.exclude_figure_captions,
35
+ )
36
+ figures = sum(asset.kind == "figure" for asset in assets)
37
+ tables = sum(asset.kind == "table" for asset in assets)
38
+ logger.info(
39
+ "Extracted {} assets ({} figures, {} tables) into {}",
40
+ len(assets),
41
+ figures,
42
+ tables,
43
+ args.output.resolve(),
44
+ )
45
+
46
+
47
+ if __name__ == "__main__":
48
+ main()
@@ -0,0 +1,398 @@
1
+ from __future__ import annotations
2
+
3
+ import json
4
+ import re
5
+ import statistics
6
+ from dataclasses import asdict, dataclass
7
+ from pathlib import Path
8
+
9
+ import pymupdf
10
+ from loguru import logger
11
+
12
+ CAPTION_RE = re.compile(
13
+ r"^\s*(?P<kind>fig(?:ure)?\.?|table)\s+"
14
+ r"(?P<label>[A-Z]?\d+(?:[.:-]\d+)*)\s*(?:[|:.\-\u2013\u2014])\s*",
15
+ re.IGNORECASE,
16
+ )
17
+
18
+
19
+ @dataclass(frozen=True)
20
+ class Caption:
21
+ kind: str
22
+ label: str
23
+ text: str
24
+ rect: pymupdf.Rect
25
+
26
+
27
+ @dataclass(frozen=True)
28
+ class Asset:
29
+ kind: str
30
+ label: str
31
+ page: int
32
+ caption: str
33
+ bbox: tuple[float, float, float, float]
34
+ file: str
35
+
36
+
37
+ def _line_text(line: dict) -> str:
38
+ return "".join(span["text"] for span in line["spans"]).strip()
39
+
40
+
41
+ def _find_captions(page: pymupdf.Page) -> list[Caption]:
42
+ captions: list[Caption] = []
43
+ data = page.get_text("dict", flags=pymupdf.TEXTFLAGS_TEXT)
44
+ for block in data["blocks"]:
45
+ if "lines" not in block:
46
+ continue
47
+ lines = block["lines"]
48
+ for index, line in enumerate(lines):
49
+ text = _line_text(line)
50
+ match = CAPTION_RE.match(text)
51
+ if not match:
52
+ continue
53
+
54
+ rect = pymupdf.Rect(line["bbox"])
55
+ full_text = text
56
+ for continuation in lines[index + 1 :]:
57
+ next_rect = pymupdf.Rect(continuation["bbox"])
58
+ if next_rect.y0 - rect.y1 > 4 or abs(next_rect.x0 - rect.x0) > 12:
59
+ break
60
+ full_text += " " + _line_text(continuation)
61
+ rect |= next_rect
62
+
63
+ raw_kind = match.group("kind").lower()
64
+ kind = "table" if raw_kind == "table" else "figure"
65
+ captions.append(Caption(kind, match.group("label"), full_text.strip(), rect))
66
+
67
+ text_blocks = [block for block in data["blocks"] if "lines" in block]
68
+ extended: list[Caption] = []
69
+ for caption in captions:
70
+ rect = pymupdf.Rect(caption.rect)
71
+ text = caption.text
72
+ for block in text_blocks:
73
+ block_rect = pymupdf.Rect(block["bbox"])
74
+ if (
75
+ 0 <= block_rect.y0 - rect.y1 <= 3
76
+ and abs(block_rect.x0 - rect.x0) <= 3
77
+ and block_rect.width >= rect.width * 0.75
78
+ ):
79
+ text += " " + " ".join(_line_text(line) for line in block["lines"])
80
+ rect |= block_rect
81
+ extended.append(Caption(caption.kind, caption.label, text, rect))
82
+ return extended
83
+
84
+
85
+ def _column_rect(page_rect: pymupdf.Rect, caption: Caption) -> pymupdf.Rect:
86
+ margin = max(24.0, page_rect.width * 0.045)
87
+ midpoint = page_rect.x0 + page_rect.width / 2
88
+ if caption.rect.width > page_rect.width * 0.46:
89
+ return pymupdf.Rect(
90
+ page_rect.x0 + margin,
91
+ page_rect.y0 + margin,
92
+ page_rect.x1 - margin,
93
+ page_rect.y1 - margin,
94
+ )
95
+ if caption.rect.x1 <= midpoint + 12:
96
+ return pymupdf.Rect(
97
+ page_rect.x0 + margin,
98
+ page_rect.y0 + margin,
99
+ midpoint - 7,
100
+ page_rect.y1 - margin,
101
+ )
102
+ return pymupdf.Rect(
103
+ midpoint + 7,
104
+ page_rect.y0 + margin,
105
+ page_rect.x1 - margin,
106
+ page_rect.y1 - margin,
107
+ )
108
+
109
+
110
+ def _graphic_rects(page: pymupdf.Page) -> list[pymupdf.Rect]:
111
+ rects = [pymupdf.Rect(item["rect"]) for item in page.get_drawings()]
112
+ for image in page.get_images(full=True):
113
+ rects.extend(page.get_image_rects(image[0]))
114
+ return [rect for rect in rects if rect.width > 1 and rect.height > 1]
115
+
116
+
117
+ def _distance(a: pymupdf.Rect, b: pymupdf.Rect) -> tuple[float, float]:
118
+ dx = max(a.x0 - b.x1, b.x0 - a.x1, 0)
119
+ dy = max(a.y0 - b.y1, b.y0 - a.y1, 0)
120
+ return dx, dy
121
+
122
+
123
+ def _cluster_rects(rects: list[pymupdf.Rect]) -> list[pymupdf.Rect]:
124
+ clusters = [pymupdf.Rect(rect) for rect in rects]
125
+ changed = True
126
+ while changed:
127
+ changed = False
128
+ merged: list[pymupdf.Rect] = []
129
+ while clusters:
130
+ current = clusters.pop()
131
+ index = 0
132
+ while index < len(clusters):
133
+ other = clusters[index]
134
+ dx, dy = _distance(current, other)
135
+ x_overlap = min(current.x1, other.x1) - max(current.x0, other.x0)
136
+ if (dx <= 8 and dy <= 8) or (x_overlap > 0 and dy <= 5):
137
+ current |= clusters.pop(index)
138
+ changed = True
139
+ index = 0
140
+ else:
141
+ index += 1
142
+ merged.append(current)
143
+ clusters = merged
144
+ return clusters
145
+
146
+
147
+ def _next_caption_limit(captions: list[Caption], caption: Caption, default: float) -> float:
148
+ candidates = [
149
+ other.rect.y0
150
+ for other in captions
151
+ if other is not caption
152
+ and other.rect.y0 > caption.rect.y1
153
+ and min(other.rect.x1, caption.rect.x1) > max(other.rect.x0, caption.rect.x0)
154
+ ]
155
+ return min(candidates, default=default)
156
+
157
+
158
+ def _pick_graphics(
159
+ page: pymupdf.Page,
160
+ caption: Caption,
161
+ captions: list[Caption],
162
+ ) -> pymupdf.Rect | None:
163
+ column = _column_rect(page.rect, caption)
164
+ clusters = _cluster_rects(
165
+ [rect & column for rect in _graphic_rects(page) if not (rect & column).is_empty]
166
+ )
167
+ candidates: list[tuple[float, pymupdf.Rect]] = []
168
+
169
+ if caption.kind == "figure":
170
+ for rect in clusters:
171
+ gap = caption.rect.y0 - rect.y1
172
+ if (
173
+ -8 <= gap <= 360
174
+ and rect.y0 < caption.rect.y0
175
+ and rect.width >= 35
176
+ and rect.height >= 25
177
+ ):
178
+ candidates.append((max(gap, 0), rect))
179
+ else:
180
+ limit = _next_caption_limit(captions, caption, column.y1)
181
+ for rect in clusters:
182
+ below_gap = rect.y0 - caption.rect.y1
183
+ above_gap = caption.rect.y0 - rect.y1
184
+ if -8 <= below_gap <= 90 and rect.y1 < limit and rect.width >= 35:
185
+ candidates.append((max(below_gap, 0), rect))
186
+ if -8 <= above_gap <= 90 and rect.y0 < caption.rect.y0 and rect.width >= 35:
187
+ candidates.append((max(above_gap, 0), rect))
188
+
189
+ if not candidates:
190
+ return None
191
+ candidates.sort(key=lambda item: (item[0], -item[1].get_area()))
192
+ return candidates[0][1]
193
+
194
+
195
+ def _word_rects(page: pymupdf.Page) -> list[pymupdf.Rect]:
196
+ return [pymupdf.Rect(word[:4]) for word in page.get_text("words")]
197
+
198
+
199
+ def _expand_to_content(
200
+ page: pymupdf.Page,
201
+ caption: Caption,
202
+ graphics: pymupdf.Rect,
203
+ include_figure_captions: bool,
204
+ ) -> pymupdf.Rect:
205
+ if caption.kind == "figure":
206
+ crop = pymupdf.Rect(graphics)
207
+ search = pymupdf.Rect(graphics.x0 - 8, graphics.y0 - 10, graphics.x1 + 8, graphics.y1 + 8)
208
+ if include_figure_captions:
209
+ crop |= caption.rect
210
+ search |= caption.rect
211
+ else:
212
+ crop = graphics | caption.rect
213
+ search = pymupdf.Rect(crop.x0 - 5, caption.rect.y0 - 2, crop.x1 + 5, crop.y1 + 5)
214
+
215
+ for word in _word_rects(page):
216
+ if word.intersects(search):
217
+ crop |= word
218
+ crop += (-3, -3, 3, 3)
219
+ return crop & page.rect
220
+
221
+
222
+ def _fallback_crop(
223
+ page: pymupdf.Page,
224
+ caption: Caption,
225
+ captions: list[Caption],
226
+ include_figure_captions: bool,
227
+ ) -> pymupdf.Rect | None:
228
+ column = _column_rect(page.rect, caption)
229
+ block_data: list[tuple[pymupdf.Rect, float]] = []
230
+ for block in page.get_text("dict", flags=pymupdf.TEXTFLAGS_TEXT)["blocks"]:
231
+ if "lines" not in block:
232
+ continue
233
+ rect = pymupdf.Rect(block["bbox"])
234
+ sizes = [
235
+ span["size"]
236
+ for line in block["lines"]
237
+ for span in line["spans"]
238
+ if span["text"].strip()
239
+ ]
240
+ if (
241
+ sizes
242
+ and rect.intersects(column)
243
+ and rect.y0 > page.rect.y0 + 65
244
+ and rect.y1 < page.rect.y1 - 65
245
+ ):
246
+ block_data.append((rect, statistics.median(sizes)))
247
+ blocks = [rect for rect, _ in block_data]
248
+ if caption.kind == "figure":
249
+ above = [rect for rect in blocks if rect.y1 < caption.rect.y0 - 2]
250
+ if not above:
251
+ return None
252
+ nearest = max(above, key=lambda rect: rect.y1)
253
+ top = max(column.y0, nearest.y0)
254
+ crop = pymupdf.Rect(column.x0, top, column.x1, caption.rect.y0 - 2)
255
+ if include_figure_captions:
256
+ crop |= caption.rect
257
+ return crop
258
+
259
+ bottom = _next_caption_limit(captions, caption, column.y1)
260
+ above = sorted(
261
+ (item for item in block_data if item[0].y1 < caption.rect.y0),
262
+ key=lambda item: item[0].y1,
263
+ reverse=True,
264
+ )
265
+ below = sorted(
266
+ (item for item in block_data if caption.rect.y1 < item[0].y0 < bottom),
267
+ key=lambda item: item[0].y0,
268
+ )
269
+ if not above and not below:
270
+ return None
271
+
272
+ above_gap = caption.rect.y0 - above[0][0].y1 if above else float("inf")
273
+ below_gap = below[0][0].y0 - caption.rect.y1 if below else float("inf")
274
+ above_size = above[0][1] if above else float("inf")
275
+ below_size = below[0][1] if below else float("inf")
276
+ previous_table = any(
277
+ other is not caption
278
+ and other.kind == "table"
279
+ and other.rect.y1 < caption.rect.y0
280
+ and min(other.rect.x1, caption.rect.x1) > max(other.rect.x0, caption.rect.x0)
281
+ for other in captions
282
+ )
283
+ wider_below = (
284
+ above
285
+ and below
286
+ and below[0][0].width > above[0][0].width * 1.5
287
+ and below_gap <= above_gap + 5
288
+ )
289
+ use_above = (
290
+ not previous_table
291
+ and not wider_below
292
+ and (
293
+ above_size + 0.75 < below_size
294
+ or abs(above_size - below_size) <= 0.75
295
+ and above_gap < below_gap
296
+ )
297
+ )
298
+ if use_above:
299
+ crop = caption.rect | above[0][0]
300
+ for rect, _ in above[1:]:
301
+ if crop.y0 - rect.y1 > 18:
302
+ break
303
+ crop |= rect
304
+ else:
305
+ crop = caption.rect | below[0][0]
306
+ for rect, _ in below[1:]:
307
+ if rect.y0 - crop.y1 > 18:
308
+ break
309
+ crop |= rect
310
+ return (crop + (-3, -3, 3, 3)) & page.rect
311
+
312
+
313
+ def _safe_label(label: str) -> str:
314
+ return re.sub(r"[^A-Za-z0-9.-]+", "-", label).strip("-").lower()
315
+
316
+
317
+ def _clear_previous_assets(output_dir: Path) -> None:
318
+ manifest_path = output_dir / "manifest.json"
319
+ if not manifest_path.is_file():
320
+ return
321
+ manifest = json.loads(manifest_path.read_text(encoding="utf-8"))
322
+ for asset in manifest.get("assets", []):
323
+ filename = asset.get("file", "")
324
+ if filename and Path(filename).name == filename:
325
+ (output_dir / filename).unlink(missing_ok=True)
326
+ manifest_path.unlink()
327
+
328
+
329
+ def extract_assets(
330
+ pdf_path: str | Path,
331
+ output_dir: str | Path,
332
+ *,
333
+ dpi: int = 400,
334
+ include_figure_captions: bool = True,
335
+ ) -> list[Asset]:
336
+ pdf_path = Path(pdf_path)
337
+ output_dir = Path(output_dir)
338
+ if not pdf_path.is_file() or pdf_path.suffix.lower() != ".pdf":
339
+ raise ValueError(f"Input is not a PDF file: {pdf_path}")
340
+ if dpi < 72:
341
+ raise ValueError("DPI must be at least 72")
342
+
343
+ output_dir.mkdir(parents=True, exist_ok=True)
344
+ _clear_previous_assets(output_dir)
345
+
346
+ document = pymupdf.open(pdf_path)
347
+ assets: list[Asset] = []
348
+ counters = {"figure": 0, "table": 0}
349
+ scale = dpi / 72
350
+
351
+ for page_index, page in enumerate(document):
352
+ captions = _find_captions(page)
353
+ logger.debug("Page {}: found {} captions", page_index + 1, len(captions))
354
+ for caption in captions:
355
+ graphics = _pick_graphics(page, caption, captions)
356
+ crop = (
357
+ _expand_to_content(page, caption, graphics, include_figure_captions)
358
+ if graphics
359
+ else _fallback_crop(page, caption, captions, include_figure_captions)
360
+ )
361
+ if crop is None or crop.width < 30 or crop.height < 15:
362
+ logger.warning(
363
+ "Skipped {} {} on page {}: no content boundary found",
364
+ caption.kind,
365
+ caption.label,
366
+ page_index + 1,
367
+ )
368
+ continue
369
+
370
+ counters[caption.kind] += 1
371
+ filename = (
372
+ f"{caption.kind}-{counters[caption.kind]:03d}-"
373
+ f"{_safe_label(caption.label)}-page-{page_index + 1:03d}.png"
374
+ )
375
+ pixmap = page.get_pixmap(matrix=pymupdf.Matrix(scale, scale), clip=crop, alpha=False)
376
+ pixmap.save(output_dir / filename)
377
+ assets.append(
378
+ Asset(
379
+ kind=caption.kind,
380
+ label=caption.label,
381
+ page=page_index + 1,
382
+ caption=caption.text,
383
+ bbox=tuple(round(value, 2) for value in crop),
384
+ file=filename,
385
+ )
386
+ )
387
+
388
+ document.close()
389
+ manifest = {
390
+ "source": str(pdf_path.resolve()),
391
+ "dpi": dpi,
392
+ "asset_count": len(assets),
393
+ "assets": [asdict(asset) for asset in assets],
394
+ }
395
+ (output_dir / "manifest.json").write_text(
396
+ json.dumps(manifest, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
397
+ )
398
+ return assets
@@ -0,0 +1,17 @@
1
+ from __future__ import annotations
2
+
3
+ import sys
4
+
5
+ from loguru import logger
6
+
7
+ _CONFIGURED = False
8
+
9
+
10
+ def configure_logging() -> None:
11
+ global _CONFIGURED
12
+ if _CONFIGURED:
13
+ return
14
+
15
+ logger.remove()
16
+ logger.add(sys.stderr, level="INFO", format="{time:HH:mm:ss} | {level} | {message}")
17
+ _CONFIGURED = True
@@ -0,0 +1,82 @@
1
+ import json
2
+ from pathlib import Path
3
+
4
+ import pymupdf
5
+
6
+ from pdf_figure_table_extractor import extract_assets
7
+
8
+
9
+ def create_test_pdf(path: Path) -> None:
10
+ document = pymupdf.open()
11
+
12
+ figure_page = document.new_page(width=612, height=792)
13
+ figure_page.draw_rect(pymupdf.Rect(60, 100, 280, 260), width=1)
14
+ figure_page.draw_line((80, 220), (140, 150), width=2)
15
+ figure_page.draw_line((140, 150), (240, 205), width=2)
16
+ figure_page.insert_text((90, 125), "Synthetic Architecture", fontsize=12)
17
+ figure_page.insert_textbox(
18
+ pymupdf.Rect(60, 275, 285, 310),
19
+ "Figure 1: Synthetic architecture used to verify complete caption extraction.",
20
+ fontsize=9,
21
+ )
22
+
23
+ ruled_table_page = document.new_page(width=612, height=792)
24
+ ruled_table_page.insert_textbox(
25
+ pymupdf.Rect(60, 80, 285, 115),
26
+ "Table 1: Synthetic ruled table with two columns and complete title.",
27
+ fontsize=9,
28
+ )
29
+ for y in (130, 155, 180, 205):
30
+ ruled_table_page.draw_line((60, y), (280, y), width=1)
31
+ for x in (60, 170, 280):
32
+ ruled_table_page.draw_line((x, 130), (x, 205), width=1)
33
+ ruled_table_page.insert_text((80, 148), "Method", fontsize=9)
34
+ ruled_table_page.insert_text((195, 148), "Score", fontsize=9)
35
+ ruled_table_page.insert_text((80, 173), "Baseline", fontsize=9)
36
+ ruled_table_page.insert_text((205, 173), "0.75", fontsize=9)
37
+ ruled_table_page.insert_text((80, 198), "Ours", fontsize=9)
38
+ ruled_table_page.insert_text((205, 198), "0.91", fontsize=9)
39
+
40
+ unruled_table_page = document.new_page(width=612, height=792)
41
+ unruled_table_page.insert_text((80, 120), "Model Accuracy", fontsize=9)
42
+ unruled_table_page.insert_text((80, 142), "Small 82.0", fontsize=9)
43
+ unruled_table_page.insert_text((80, 164), "Large 91.5", fontsize=9)
44
+ unruled_table_page.insert_textbox(
45
+ pymupdf.Rect(60, 185, 285, 220),
46
+ "Table 2: Synthetic unruled table whose title appears below the data.",
47
+ fontsize=9,
48
+ )
49
+
50
+ document.save(path)
51
+ document.close()
52
+
53
+
54
+ def test_extracts_figures_and_tables(tmp_path: Path) -> None:
55
+ pdf_path = tmp_path / "paper.pdf"
56
+ output_dir = tmp_path / "assets"
57
+ create_test_pdf(pdf_path)
58
+
59
+ assets = extract_assets(pdf_path, output_dir, dpi=144)
60
+
61
+ assert [asset.kind for asset in assets] == ["figure", "table", "table"]
62
+ assert all((output_dir / asset.file).is_file() for asset in assets)
63
+ manifest = json.loads((output_dir / "manifest.json").read_text(encoding="utf-8"))
64
+ assert manifest["asset_count"] == 3
65
+ assert len(list(output_dir.glob("*.png"))) == 3
66
+
67
+ figure = pymupdf.Pixmap(output_dir / assets[0].file)
68
+ assert figure.width > 400
69
+ assert figure.height > 300
70
+
71
+
72
+ def test_rerun_preserves_unrelated_files(tmp_path: Path) -> None:
73
+ pdf_path = tmp_path / "paper.pdf"
74
+ output_dir = tmp_path / "assets"
75
+ create_test_pdf(pdf_path)
76
+ extract_assets(pdf_path, output_dir, dpi=144)
77
+ unrelated = output_dir / "notes.txt"
78
+ unrelated.write_text("keep", encoding="utf-8")
79
+
80
+ extract_assets(pdf_path, output_dir, dpi=144)
81
+
82
+ assert unrelated.read_text(encoding="utf-8") == "keep"
@@ -0,0 +1,156 @@
1
+ version = 1
2
+ revision = 3
3
+ requires-python = ">=3.11"
4
+
5
+ [[package]]
6
+ name = "colorama"
7
+ version = "0.4.6"
8
+ source = { registry = "https://pypi.org/simple" }
9
+ sdist = { url = "https://files.pythonhosted.org/packages/d8/53/6f443c9a4a8358a93a6792e2acffb9d9d5cb0a5cfd8802644b7b1c9a02e4/colorama-0.4.6.tar.gz", hash = "sha256:08695f5cb7ed6e0531a20572697297273c47b8cae5a63ffc6d6ed5c201be6e44", size = 27697, upload-time = "2022-10-25T02:36:22.414Z" }
10
+ wheels = [
11
+ { url = "https://files.pythonhosted.org/packages/d1/d6/3965ed04c63042e047cb6a3e6ed1a63a35087b6a609aa3a15ed8ac56c221/colorama-0.4.6-py2.py3-none-any.whl", hash = "sha256:4f1d9991f5acc0ca119f9d443620b77f9d6b33703e51011c16baf57afb285fc6", size = 25335, upload-time = "2022-10-25T02:36:20.889Z" },
12
+ ]
13
+
14
+ [[package]]
15
+ name = "iniconfig"
16
+ version = "2.3.0"
17
+ source = { registry = "https://pypi.org/simple" }
18
+ sdist = { url = "https://files.pythonhosted.org/packages/72/34/14ca021ce8e5dfedc35312d08ba8bf51fdd999c576889fc2c24cb97f4f10/iniconfig-2.3.0.tar.gz", hash = "sha256:c76315c77db068650d49c5b56314774a7804df16fee4402c1f19d6d15d8c4730", size = 20503, upload-time = "2025-10-18T21:55:43.219Z" }
19
+ wheels = [
20
+ { url = "https://files.pythonhosted.org/packages/cb/b1/3846dd7f199d53cb17f49cba7e651e9ce294d8497c8c150530ed11865bb8/iniconfig-2.3.0-py3-none-any.whl", hash = "sha256:f631c04d2c48c52b84d0d0549c99ff3859c98df65b3101406327ecc7d53fbf12", size = 7484, upload-time = "2025-10-18T21:55:41.639Z" },
21
+ ]
22
+
23
+ [[package]]
24
+ name = "loguru"
25
+ version = "0.7.3"
26
+ source = { registry = "https://pypi.org/simple" }
27
+ dependencies = [
28
+ { name = "colorama", marker = "sys_platform == 'win32'" },
29
+ { name = "win32-setctime", marker = "sys_platform == 'win32'" },
30
+ ]
31
+ sdist = { url = "https://files.pythonhosted.org/packages/3a/05/a1dae3dffd1116099471c643b8924f5aa6524411dc6c63fdae648c4f1aca/loguru-0.7.3.tar.gz", hash = "sha256:19480589e77d47b8d85b2c827ad95d49bf31b0dcde16593892eb51dd18706eb6", size = 63559, upload-time = "2024-12-06T11:20:56.608Z" }
32
+ wheels = [
33
+ { url = "https://files.pythonhosted.org/packages/0c/29/0348de65b8cc732daa3e33e67806420b2ae89bdce2b04af740289c5c6c8c/loguru-0.7.3-py3-none-any.whl", hash = "sha256:31a33c10c8e1e10422bfd431aeb5d351c7cf7fa671e3c4df004162264b28220c", size = 61595, upload-time = "2024-12-06T11:20:54.538Z" },
34
+ ]
35
+
36
+ [[package]]
37
+ name = "packaging"
38
+ version = "26.3"
39
+ source = { registry = "https://pypi.org/simple" }
40
+ sdist = { url = "https://files.pythonhosted.org/packages/7d/fa/3944b40b07da9ce895c0e6303a5ab7d53da063554f534556b134a54d6093/packaging-26.3.tar.gz", hash = "sha256:94edc256424af38762eb31306eed28beb9f0efc50a8837492c9d6fd6004aed79", size = 313412, upload-time = "2026-08-04T18:15:28.737Z" }
41
+ wheels = [
42
+ { url = "https://files.pythonhosted.org/packages/63/34/ba1c580383c9eada3711951fef0795c80b829a078d72188184bcab9dd527/packaging-26.3-py3-none-any.whl", hash = "sha256:d7193f7c8e4e93f444fde0262bf90af30e16fa0ad0ad44cb553c87339b23cd1c", size = 129956, upload-time = "2026-08-04T18:15:27.159Z" },
43
+ ]
44
+
45
+ [[package]]
46
+ name = "pdf-figure-table-extractor"
47
+ version = "0.1.0"
48
+ source = { editable = "." }
49
+ dependencies = [
50
+ { name = "loguru" },
51
+ { name = "pymupdf" },
52
+ ]
53
+
54
+ [package.dev-dependencies]
55
+ dev = [
56
+ { name = "pytest" },
57
+ { name = "ruff" },
58
+ ]
59
+
60
+ [package.metadata]
61
+ requires-dist = [
62
+ { name = "loguru", specifier = ">=0.7,<1" },
63
+ { name = "pymupdf", specifier = ">=1.26,<2" },
64
+ ]
65
+
66
+ [package.metadata.requires-dev]
67
+ dev = [
68
+ { name = "pytest", specifier = ">=8,<10" },
69
+ { name = "ruff", specifier = ">=0.8" },
70
+ ]
71
+
72
+ [[package]]
73
+ name = "pluggy"
74
+ version = "1.6.0"
75
+ source = { registry = "https://pypi.org/simple" }
76
+ sdist = { url = "https://files.pythonhosted.org/packages/f9/e2/3e91f31a7d2b083fe6ef3fa267035b518369d9511ffab804f839851d2779/pluggy-1.6.0.tar.gz", hash = "sha256:7dcc130b76258d33b90f61b658791dede3486c3e6bfb003ee5c9bfb396dd22f3", size = 69412, upload-time = "2025-05-15T12:30:07.975Z" }
77
+ wheels = [
78
+ { url = "https://files.pythonhosted.org/packages/54/20/4d324d65cc6d9205fabedc306948156824eb9f0ee1633355a8f7ec5c66bf/pluggy-1.6.0-py3-none-any.whl", hash = "sha256:e920276dd6813095e9377c0bc5566d94c932c33b27a3e3945d8389c374dd4746", size = 20538, upload-time = "2025-05-15T12:30:06.134Z" },
79
+ ]
80
+
81
+ [[package]]
82
+ name = "pygments"
83
+ version = "2.21.0"
84
+ source = { registry = "https://pypi.org/simple" }
85
+ sdist = { url = "https://files.pythonhosted.org/packages/49/2e/ced460408999b33da6b31b0021b0f37d329e202d4169aeb164493778f25b/pygments-2.21.0.tar.gz", hash = "sha256:610ca751c9bc2492b38eb9a38a7fbc93edbbb2d7182edaf34e66ae493dee5c8c", size = 5005329, upload-time = "2026-08-17T08:02:48.824Z" }
86
+ wheels = [
87
+ { url = "https://files.pythonhosted.org/packages/71/46/17f022dd3e953bf20a04a028a21ec746d942f8d2af30fa0f124fa0e6a684/pygments-2.21.0-py3-none-any.whl", hash = "sha256:2363c69b61c4a97c838da3b130dcd6468f4848992b21a82f2a63ec34377137d9", size = 1250147, upload-time = "2026-08-17T08:02:44.912Z" },
88
+ ]
89
+
90
+ [[package]]
91
+ name = "pymupdf"
92
+ version = "1.28.2"
93
+ source = { registry = "https://pypi.org/simple" }
94
+ sdist = { url = "https://files.pythonhosted.org/packages/a3/fb/b6761fa2d5266f2cdb24c3b91f4023070ab7848381417678e7a289a1d52a/pymupdf-1.28.2.tar.gz", hash = "sha256:5e0be7908a715aa20333caddd73f1d6f01e4cd0c26e869fa2dd0b7f344da2249", size = 87903557, upload-time = "2026-08-06T21:43:23.321Z" }
95
+ wheels = [
96
+ { url = "https://files.pythonhosted.org/packages/b4/51/550c9a75c4ff3245cb4ecb7bb95cbe2ab7374230b8e2b7a1f7259444150b/pymupdf-1.28.2-cp310-abi3-macosx_10_15_x86_64.whl", hash = "sha256:5fc315b425ff1f7afdd1ea2f348205cb19b806767daae7ce4d64115799c2bae1", size = 24645079, upload-time = "2026-08-06T21:37:25.001Z" },
97
+ { url = "https://files.pythonhosted.org/packages/fa/01/3591f781b417b382a8487a2356e927acfe858b1043bab0ec47f6805bb109/pymupdf-1.28.2-cp310-abi3-macosx_11_0_arm64.whl", hash = "sha256:7113846b35dbf0a033f088e4f4fb543dabeb4b0b12c112966a1ca1ee2d5eacae", size = 23875605, upload-time = "2026-08-06T21:37:40.369Z" },
98
+ { url = "https://files.pythonhosted.org/packages/d2/86/4a68f080b71b46802178346af46486e1697508e760855ff5f3b218a6dff7/pymupdf-1.28.2-cp310-abi3-manylinux_2_28_aarch64.whl", hash = "sha256:3050a233dde1211efe89ada74e2add6238436434159f46097a1423aad2842545", size = 25095554, upload-time = "2026-08-06T21:37:58.485Z" },
99
+ { url = "https://files.pythonhosted.org/packages/c7/06/dace3e27af26690cb20bead80dbac42941b0841eb689b8aabbd67dde16f0/pymupdf-1.28.2-cp310-abi3-manylinux_2_28_x86_64.whl", hash = "sha256:397d6715c1f0df7548a92d0afd8ce370fc48fa47aeefac16be2bc04a16a8227f", size = 25762500, upload-time = "2026-08-06T21:38:17.438Z" },
100
+ { url = "https://files.pythonhosted.org/packages/e5/61/4146dfa1d8172a1ce8d59f0eed94896ddefb8deb2274534d0522fbb8abf5/pymupdf-1.28.2-cp310-abi3-musllinux_1_2_x86_64.whl", hash = "sha256:f89fb2d86d07d643a269f17a093105057e20c79c1d06c103b53600067b6d2b01", size = 25986309, upload-time = "2026-08-06T21:38:35.472Z" },
101
+ { url = "https://files.pythonhosted.org/packages/52/60/1fb6e64676f7500ebe89054b9e5bbbe14d3101c92d5f1a40ac9a35227673/pymupdf-1.28.2-cp310-abi3-win32.whl", hash = "sha256:530ef543a3885b3b81cb72a854e7c5a625a9233201221132bb6c31698c6a2bdb", size = 18525353, upload-time = "2026-08-06T21:38:47.697Z" },
102
+ { url = "https://files.pythonhosted.org/packages/4a/61/d563bbccba262f9dd6d2d35ccb72593648184d886188efb12d9ce8f34dd6/pymupdf-1.28.2-cp310-abi3-win_amd64.whl", hash = "sha256:ebd244918798502d7b4504c90410d1711a4d7675a32584ca30f1bab419ecbffe", size = 19826532, upload-time = "2026-08-06T21:39:00.213Z" },
103
+ { url = "https://files.pythonhosted.org/packages/e2/93/08f404a1f0155fe24137cf2d3aabd3e2b4b08c62053ed89c60f2611be3e9/pymupdf-1.28.2-cp310-abi3-win_arm64.whl", hash = "sha256:ffe91a24edc75c80da2a4b62f50fc0f54632d34fc8fe4cbc48e5c7ff07cf8fb4", size = 19759252, upload-time = "2026-08-06T21:39:12.937Z" },
104
+ { url = "https://files.pythonhosted.org/packages/58/8c/d897dcd32a25b58186c968b15ce4324ca029e9d96460de12325314e390be/pymupdf-1.28.2-cp313-abi3-pyemscripten_2025_0_wasm32.whl", hash = "sha256:2e1b574c0fd2cb238021033fd3c0f9c4388816638df064e4bfb56d9d81736dc8", size = 18399403, upload-time = "2026-08-06T21:39:25.008Z" },
105
+ { url = "https://files.pythonhosted.org/packages/f6/f1/de34a1c53fe2bf8c6e71db84b0ced782d408970c9810d2b456a2ae96814c/pymupdf-1.28.2-cp314-cp314t-manylinux_2_28_x86_64.whl", hash = "sha256:fd481ed48bef56305c41fb7e05a055c03345c899c7b101dad086258b438f8168", size = 25802333, upload-time = "2026-08-06T21:39:41.426Z" },
106
+ ]
107
+
108
+ [[package]]
109
+ name = "pytest"
110
+ version = "9.1.1"
111
+ source = { registry = "https://pypi.org/simple" }
112
+ dependencies = [
113
+ { name = "colorama", marker = "sys_platform == 'win32'" },
114
+ { name = "iniconfig" },
115
+ { name = "packaging" },
116
+ { name = "pluggy" },
117
+ { name = "pygments" },
118
+ ]
119
+ sdist = { url = "https://files.pythonhosted.org/packages/e4/47/b9efed96c114afcfa3c9d3fe98a76a1d14c74a9e266d397cf6eb64be5e01/pytest-9.1.1.tar.gz", hash = "sha256:1088fbde8f2b49d95a549a195707afa7a76a3ce9bcadc26b6d71f0ffda5fe313", size = 1636369, upload-time = "2026-06-19T10:58:32.857Z" }
120
+ wheels = [
121
+ { url = "https://files.pythonhosted.org/packages/24/25/1de2678b631f5a49215c6c96fff41ba892b0a34df68d6d80292b1b48aa7f/pytest-9.1.1-py3-none-any.whl", hash = "sha256:37a86b45efb9a47a61a36449063e8e18d0cab3161329fc099eb21783169c4f0c", size = 386536, upload-time = "2026-06-19T10:58:31.347Z" },
122
+ ]
123
+
124
+ [[package]]
125
+ name = "ruff"
126
+ version = "0.16.10"
127
+ source = { registry = "https://pypi.org/simple" }
128
+ sdist = { url = "https://files.pythonhosted.org/packages/c4/49/23802c45f093eb14bde54b141d2b2f058edfa63a06db7beed047308cc08f/ruff-0.16.10.tar.gz", hash = "sha256:eff4728c4eaae93f0955cd264d24b2ab348e74bf59986ccf282ba6dc16b3b017", size = 4958724, upload-time = "2026-10-01T18:03:21.697Z" }
129
+ wheels = [
130
+ { url = "https://files.pythonhosted.org/packages/2f/21/ebce22e1d90cdb2cd691026b9c6e9bec6499f0b396089481755b5d49efff/ruff-0.16.10-py3-none-linux_armv6l.whl", hash = "sha256:488b0fe3f3574210e5cf80d9f59b9e3ab17a127a8155de3f307b392589cfb511", size = 10095557, upload-time = "2026-10-01T18:02:36.072Z" },
131
+ { url = "https://files.pythonhosted.org/packages/cb/98/a54de85876a8b2612bfa0d84c7b9abfb39c6a3354aee7800b09c1649b9e3/ruff-0.16.10-py3-none-macosx_10_12_x86_64.whl", hash = "sha256:e748ff95c934c4e978783b8e687bc174e7bd84e8ad24e3243e1ecfcda5e0282d", size = 10340577, upload-time = "2026-10-01T18:02:39.258Z" },
132
+ { url = "https://files.pythonhosted.org/packages/9f/16/1a5a4a2657effe4806110f2b907313802f1367fcdbb29e8122766e407fab/ruff-0.16.10-py3-none-macosx_11_0_arm64.whl", hash = "sha256:3031a4a2e8e7b8a46f70be45f198c35a11ece509a94b80334d8d397a33c67550", size = 9774282, upload-time = "2026-10-01T18:02:41.79Z" },
133
+ { url = "https://files.pythonhosted.org/packages/6e/fb/470085af734da396e68cd80fb0a3e7c109a459716ae59588c0f6fab8a17d/ruff-0.16.10-py3-none-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:494401c86df4c4c25f69b9419605d944467ee98c42fb6ad405ef4fa40b8fb67d", size = 9920895, upload-time = "2026-10-01T18:02:44.458Z" },
134
+ { url = "https://files.pythonhosted.org/packages/57/de/f10cffe4f88a37ea6615ec460f01bf76f0bc1477737a35d9ee61a630a78b/ruff-0.16.10-py3-none-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:d203abc0ff2b773ee33d00ab8df0bb67046fbc7c07b119332f08b9b345cf8221", size = 9902707, upload-time = "2026-10-01T18:02:47.041Z" },
135
+ { url = "https://files.pythonhosted.org/packages/ff/44/3fdcedf83ae60ef837dd239170e480dce606a9142cfa2db911afdf855fd9/ruff-0.16.10-py3-none-manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:bd83d1235a5258d318477bc5b576303974cbdef5df0c01a1bff14efcc23bd12a", size = 10619287, upload-time = "2026-10-01T18:02:49.485Z" },
136
+ { url = "https://files.pythonhosted.org/packages/c1/62/02e76a5574002153618eafbb70e159468addc72a5d2c488e65cbb4639d2d/ruff-0.16.10-py3-none-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:bc2610fb269fa56dd8a68669ae470fa6272902668c0fc2ebc3aa112b2633d5b8", size = 11339942, upload-time = "2026-10-01T18:02:52.008Z" },
137
+ { url = "https://files.pythonhosted.org/packages/6b/c4/cde27d47ad8d4126c587e608c67c46e7feac11763f606d261a1bc8a489e6/ruff-0.16.10-py3-none-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:3e70175e29cc94c26ea296c80e470180b744b7419026898e58f520c6ab32578e", size = 10934316, upload-time = "2026-10-01T18:02:54.811Z" },
138
+ { url = "https://files.pythonhosted.org/packages/e4/03/17234145f302a645a123e8c3bb2411ecf4669fbc350de3b0d803230a1729/ruff-0.16.10-py3-none-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:f33f43a864a8483eebd160e713336c8bab02c934feaff0a33cf5ccb41546d09a", size = 10387968, upload-time = "2026-10-01T18:02:57.497Z" },
139
+ { url = "https://files.pythonhosted.org/packages/71/29/2493af60240ee7644b38a4b821f5f9c3b5a4fa3178770fe1d0c685217395/ruff-0.16.10-py3-none-manylinux_2_31_riscv64.whl", hash = "sha256:1dfc6f0088149fb6a362c1c446bcbb3fd2157b3852fe2fa68409276eab9ad9b3", size = 10537906, upload-time = "2026-10-01T18:03:00.006Z" },
140
+ { url = "https://files.pythonhosted.org/packages/2e/9a/f56b28f3b143bb9e518c34fb89ab191b626bca8b70dbf8af0d2e2d473572/ruff-0.16.10-py3-none-musllinux_1_2_aarch64.whl", hash = "sha256:6553498afc35f580f036030795810b9e6bcea31604b0fd9e8d352795473042e3", size = 10012938, upload-time = "2026-10-01T18:03:03.006Z" },
141
+ { url = "https://files.pythonhosted.org/packages/c2/c4/fca37362848ea4d4d80712df13632e645e7c7cfb1bdedf140699ac7a090b/ruff-0.16.10-py3-none-musllinux_1_2_armv7l.whl", hash = "sha256:a3b8471dea115d37f123882be852bed13403746d3a76c11de4a19ec5f9ff5a03", size = 9897945, upload-time = "2026-10-01T18:03:05.818Z" },
142
+ { url = "https://files.pythonhosted.org/packages/7d/c7/e0bc57664d6e0c61fd22f620af260fe9662d7e1ddb320ea7f160e76ef165/ruff-0.16.10-py3-none-musllinux_1_2_i686.whl", hash = "sha256:92e59a70bcbd9d3a5483656da906ec28edfdacfce00afd99edb8b4e9d15644be", size = 10333134, upload-time = "2026-10-01T18:03:08.25Z" },
143
+ { url = "https://files.pythonhosted.org/packages/21/aa/5c9f3b68737c0e4a33a1db5dfab44f7b786d96ca91a7233112ddab1e6dc2/ruff-0.16.10-py3-none-musllinux_1_2_x86_64.whl", hash = "sha256:7ae7375f803b5520dc9f546bed7e3a0acb70b91812e9bb4b19927de22f25b77d", size = 10741702, upload-time = "2026-10-01T18:03:10.775Z" },
144
+ { url = "https://files.pythonhosted.org/packages/78/fa/0f9c2020dc316c53d983be011720b8157cc052b97e3f4dfa7db6d0f880a6/ruff-0.16.10-py3-none-win32.whl", hash = "sha256:2a12e01cb9156c10c466f63b46eaae5ecea28dfbd21b5836353ae498e7d1349a", size = 10139176, upload-time = "2026-10-01T18:03:13.267Z" },
145
+ { url = "https://files.pythonhosted.org/packages/99/29/cfb0df9448d4d4ad48c2de029ada9ebd71baa6da983a6c77ee6c6cd0fe82/ruff-0.16.10-py3-none-win_amd64.whl", hash = "sha256:97f2015c92aa97105b0eab19eb5d224884399281cfc5da86a92db4ab5e7fb2ca", size = 10584734, upload-time = "2026-10-01T18:03:16.006Z" },
146
+ { url = "https://files.pythonhosted.org/packages/fc/05/c16957eb287c3fc062e032619a25868d93d408a844b725bcf514f7a378ff/ruff-0.16.10-py3-none-win_arm64.whl", hash = "sha256:25a65fe998c4e6861ec079ada5826a2fc605e6cbccbe9dcd7fac1f54e791621b", size = 10366440, upload-time = "2026-10-01T18:03:19.04Z" },
147
+ ]
148
+
149
+ [[package]]
150
+ name = "win32-setctime"
151
+ version = "1.2.0"
152
+ source = { registry = "https://pypi.org/simple" }
153
+ sdist = { url = "https://files.pythonhosted.org/packages/b3/8f/705086c9d734d3b663af0e9bb3d4de6578d08f46b1b101c2442fd9aecaa2/win32_setctime-1.2.0.tar.gz", hash = "sha256:ae1fdf948f5640aae05c511ade119313fb6a30d7eabe25fef9764dca5873c4c0", size = 4867, upload-time = "2024-12-07T15:28:28.314Z" }
154
+ wheels = [
155
+ { url = "https://files.pythonhosted.org/packages/e1/07/c6fe3ad3e685340704d314d765b7912993bcb8dc198f0e7a89382d37974b/win32_setctime-1.2.0-py3-none-any.whl", hash = "sha256:95d644c4e708aba81dc3704a116d8cbc974d70b3bdb8be1d150e36be6e9d1390", size = 4083, upload-time = "2024-12-07T15:28:26.465Z" },
156
+ ]