visual-parser 1.0.1__tar.gz → 2.0.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {visual_parser-1.0.1 → visual_parser-2.0.0}/PKG-INFO +13 -11
- {visual_parser-1.0.1 → visual_parser-2.0.0}/README.md +7 -7
- {visual_parser-1.0.1 → visual_parser-2.0.0}/pyproject.toml +13 -11
- {visual_parser-1.0.1 → visual_parser-2.0.0}/visual_parser/__init__.py +1 -1
- {visual_parser-1.0.1 → visual_parser-2.0.0}/visual_parser/cli.py +40 -28
- {visual_parser-1.0.1 → visual_parser-2.0.0}/visual_parser/cli_main.py +13 -1
- {visual_parser-1.0.1 → visual_parser-2.0.0}/visual_parser/config.py +10 -0
- {visual_parser-1.0.1 → visual_parser-2.0.0}/visual_parser/figure_describer.py +65 -44
- visual_parser-2.0.0/visual_parser/pipeline.py +363 -0
- {visual_parser-1.0.1 → visual_parser-2.0.0}/visual_parser/text_extractor.py +90 -100
- {visual_parser-1.0.1 → visual_parser-2.0.0}/visual_parser.egg-info/PKG-INFO +13 -11
- {visual_parser-1.0.1 → visual_parser-2.0.0}/visual_parser.egg-info/requires.txt +2 -0
- visual_parser-1.0.1/visual_parser/pipeline.py +0 -255
- {visual_parser-1.0.1 → visual_parser-2.0.0}/setup.cfg +0 -0
- {visual_parser-1.0.1 → visual_parser-2.0.0}/visual_parser/__main__.py +0 -0
- {visual_parser-1.0.1 → visual_parser-2.0.0}/visual_parser/jsonl_writer.py +0 -0
- {visual_parser-1.0.1 → visual_parser-2.0.0}/visual_parser/metadata_extractor.py +0 -0
- {visual_parser-1.0.1 → visual_parser-2.0.0}/visual_parser/nougat_engine.py +0 -0
- {visual_parser-1.0.1 → visual_parser-2.0.0}/visual_parser/pdf_tracker.py +0 -0
- {visual_parser-1.0.1 → visual_parser-2.0.0}/visual_parser/prompts.py +0 -0
- {visual_parser-1.0.1 → visual_parser-2.0.0}/visual_parser/vision_llm.py +0 -0
- {visual_parser-1.0.1 → visual_parser-2.0.0}/visual_parser.egg-info/SOURCES.txt +0 -0
- {visual_parser-1.0.1 → visual_parser-2.0.0}/visual_parser.egg-info/dependency_links.txt +0 -0
- {visual_parser-1.0.1 → visual_parser-2.0.0}/visual_parser.egg-info/entry_points.txt +0 -0
- {visual_parser-1.0.1 → visual_parser-2.0.0}/visual_parser.egg-info/top_level.txt +0 -0
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: visual-parser
|
|
3
|
-
Version:
|
|
4
|
-
Summary: Standalone Visual-RAG PDF Parser
|
|
5
|
-
License:
|
|
3
|
+
Version: 2.0.0
|
|
4
|
+
Summary: Standalone Visual-RAG PDF Parser - text extraction and Vision-LLM figure descriptions to JSONL
|
|
5
|
+
License: Apache-2.0
|
|
6
6
|
Project-URL: Homepage, https://github.com/SmartLabNuclear/RADIANT_LLM
|
|
7
7
|
Project-URL: Repository, https://github.com/SmartLabNuclear/RADIANT_LLM
|
|
8
8
|
Project-URL: Docker Hub, https://hub.docker.com/r/zev94/radiant-llm
|
|
@@ -11,7 +11,7 @@ Classifier: Programming Language :: Python :: 3
|
|
|
11
11
|
Classifier: Programming Language :: Python :: 3.10
|
|
12
12
|
Classifier: Programming Language :: Python :: 3.11
|
|
13
13
|
Classifier: Programming Language :: Python :: 3.12
|
|
14
|
-
Classifier: License :: OSI Approved ::
|
|
14
|
+
Classifier: License :: OSI Approved :: Apache Software License
|
|
15
15
|
Classifier: Operating System :: OS Independent
|
|
16
16
|
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
17
17
|
Classifier: Topic :: Text Processing :: Markup
|
|
@@ -30,6 +30,8 @@ Requires-Dist: openai==1.78.1
|
|
|
30
30
|
Requires-Dist: google-generativeai==0.8.5
|
|
31
31
|
Requires-Dist: python-dotenv==1.1.0
|
|
32
32
|
Requires-Dist: tqdm==4.67.1
|
|
33
|
+
Requires-Dist: nltk>=3.8
|
|
34
|
+
Requires-Dist: python-Levenshtein>=0.20
|
|
33
35
|
Provides-Extra: ocr
|
|
34
36
|
Requires-Dist: pytesseract==0.3.13; extra == "ocr"
|
|
35
37
|
Provides-Extra: dev
|
|
@@ -71,7 +73,7 @@ Prebuilt images are on **[zev94/radiant-llm](https://hub.docker.com/r/zev94/radi
|
|
|
71
73
|
|
|
72
74
|
| Tag | Description |
|
|
73
75
|
|-----|-------------|
|
|
74
|
-
| `visual-parser-
|
|
76
|
+
| `visual-parser-2.0` | Pinned release (v2.0) |
|
|
75
77
|
| `visual-parser-latest` | Latest visual-parser build |
|
|
76
78
|
|
|
77
79
|
### 1) Install Docker
|
|
@@ -79,7 +81,7 @@ Prebuilt images are on **[zev94/radiant-llm](https://hub.docker.com/r/zev94/radi
|
|
|
79
81
|
|
|
80
82
|
### 2) Pull the image
|
|
81
83
|
```bash
|
|
82
|
-
docker pull zev94/radiant-llm:visual-parser-
|
|
84
|
+
docker pull zev94/radiant-llm:visual-parser-2.0
|
|
83
85
|
```
|
|
84
86
|
|
|
85
87
|
### 3) Run (input + output on the same mounted folder)
|
|
@@ -87,7 +89,7 @@ Windows PowerShell:
|
|
|
87
89
|
```powershell
|
|
88
90
|
docker run --rm --env-file .env `
|
|
89
91
|
-v "C:\path\to\pdfs:/data" `
|
|
90
|
-
zev94/radiant-llm:visual-parser-
|
|
92
|
+
zev94/radiant-llm:visual-parser-2.0 `
|
|
91
93
|
--input-dir /data --output-dir /data
|
|
92
94
|
```
|
|
93
95
|
|
|
@@ -95,7 +97,7 @@ Linux / WSL:
|
|
|
95
97
|
```bash
|
|
96
98
|
docker run --rm --env-file .env \
|
|
97
99
|
-v "/path/to/pdfs:/data" \
|
|
98
|
-
zev94/radiant-llm:visual-parser-
|
|
100
|
+
zev94/radiant-llm:visual-parser-2.0 \
|
|
99
101
|
--input-dir /data --output-dir /data
|
|
100
102
|
```
|
|
101
103
|
|
|
@@ -105,7 +107,7 @@ Windows PowerShell:
|
|
|
105
107
|
docker run --rm --env-file .env `
|
|
106
108
|
-v "C:\path\to\pdfs:/data" `
|
|
107
109
|
-v "C:\path\to\out:/out" `
|
|
108
|
-
zev94/radiant-llm:visual-parser-
|
|
110
|
+
zev94/radiant-llm:visual-parser-2.0 `
|
|
109
111
|
--input-dir /data --output-dir /out
|
|
110
112
|
```
|
|
111
113
|
|
|
@@ -122,7 +124,7 @@ Default vision model is **GPT-5.5** when using `--vision-provider gpt`. Override
|
|
|
122
124
|
|
|
123
125
|
```powershell
|
|
124
126
|
docker run --rm --env-file .env -v "C:\path\to\pdfs:/data" `
|
|
125
|
-
zev94/radiant-llm:visual-parser-
|
|
127
|
+
zev94/radiant-llm:visual-parser-2.0 `
|
|
126
128
|
--input-dir /data --output-dir /data --vision-model gpt-5.4
|
|
127
129
|
```
|
|
128
130
|
|
|
@@ -138,7 +140,7 @@ python visual-parser.py --input-dir "C:\path\to\pdfs"
|
|
|
138
140
|
After pulling the image, run:
|
|
139
141
|
|
|
140
142
|
```bash
|
|
141
|
-
docker run --rm zev94/radiant-llm:visual-parser-
|
|
143
|
+
docker run --rm zev94/radiant-llm:visual-parser-2.0 --help
|
|
142
144
|
```
|
|
143
145
|
|
|
144
146
|
For copy-paste **Docker** examples (vision presets, text modes, workers, rebuild), see [`docker-usage-examples.md`](docker-usage-examples.md).
|
|
@@ -32,7 +32,7 @@ Prebuilt images are on **[zev94/radiant-llm](https://hub.docker.com/r/zev94/radi
|
|
|
32
32
|
|
|
33
33
|
| Tag | Description |
|
|
34
34
|
|-----|-------------|
|
|
35
|
-
| `visual-parser-
|
|
35
|
+
| `visual-parser-2.0` | Pinned release (v2.0) |
|
|
36
36
|
| `visual-parser-latest` | Latest visual-parser build |
|
|
37
37
|
|
|
38
38
|
### 1) Install Docker
|
|
@@ -40,7 +40,7 @@ Prebuilt images are on **[zev94/radiant-llm](https://hub.docker.com/r/zev94/radi
|
|
|
40
40
|
|
|
41
41
|
### 2) Pull the image
|
|
42
42
|
```bash
|
|
43
|
-
docker pull zev94/radiant-llm:visual-parser-
|
|
43
|
+
docker pull zev94/radiant-llm:visual-parser-2.0
|
|
44
44
|
```
|
|
45
45
|
|
|
46
46
|
### 3) Run (input + output on the same mounted folder)
|
|
@@ -48,7 +48,7 @@ Windows PowerShell:
|
|
|
48
48
|
```powershell
|
|
49
49
|
docker run --rm --env-file .env `
|
|
50
50
|
-v "C:\path\to\pdfs:/data" `
|
|
51
|
-
zev94/radiant-llm:visual-parser-
|
|
51
|
+
zev94/radiant-llm:visual-parser-2.0 `
|
|
52
52
|
--input-dir /data --output-dir /data
|
|
53
53
|
```
|
|
54
54
|
|
|
@@ -56,7 +56,7 @@ Linux / WSL:
|
|
|
56
56
|
```bash
|
|
57
57
|
docker run --rm --env-file .env \
|
|
58
58
|
-v "/path/to/pdfs:/data" \
|
|
59
|
-
zev94/radiant-llm:visual-parser-
|
|
59
|
+
zev94/radiant-llm:visual-parser-2.0 \
|
|
60
60
|
--input-dir /data --output-dir /data
|
|
61
61
|
```
|
|
62
62
|
|
|
@@ -66,7 +66,7 @@ Windows PowerShell:
|
|
|
66
66
|
docker run --rm --env-file .env `
|
|
67
67
|
-v "C:\path\to\pdfs:/data" `
|
|
68
68
|
-v "C:\path\to\out:/out" `
|
|
69
|
-
zev94/radiant-llm:visual-parser-
|
|
69
|
+
zev94/radiant-llm:visual-parser-2.0 `
|
|
70
70
|
--input-dir /data --output-dir /out
|
|
71
71
|
```
|
|
72
72
|
|
|
@@ -83,7 +83,7 @@ Default vision model is **GPT-5.5** when using `--vision-provider gpt`. Override
|
|
|
83
83
|
|
|
84
84
|
```powershell
|
|
85
85
|
docker run --rm --env-file .env -v "C:\path\to\pdfs:/data" `
|
|
86
|
-
zev94/radiant-llm:visual-parser-
|
|
86
|
+
zev94/radiant-llm:visual-parser-2.0 `
|
|
87
87
|
--input-dir /data --output-dir /data --vision-model gpt-5.4
|
|
88
88
|
```
|
|
89
89
|
|
|
@@ -99,7 +99,7 @@ python visual-parser.py --input-dir "C:\path\to\pdfs"
|
|
|
99
99
|
After pulling the image, run:
|
|
100
100
|
|
|
101
101
|
```bash
|
|
102
|
-
docker run --rm zev94/radiant-llm:visual-parser-
|
|
102
|
+
docker run --rm zev94/radiant-llm:visual-parser-2.0 --help
|
|
103
103
|
```
|
|
104
104
|
|
|
105
105
|
For copy-paste **Docker** examples (vision presets, text modes, workers, rebuild), see [`docker-usage-examples.md`](docker-usage-examples.md).
|
|
@@ -1,14 +1,14 @@
|
|
|
1
1
|
[build-system]
|
|
2
|
-
requires
|
|
2
|
+
requires = ["setuptools>=61", "wheel"]
|
|
3
3
|
build-backend = "setuptools.build_meta"
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
|
-
name
|
|
7
|
-
version
|
|
8
|
-
description = "Standalone Visual-RAG PDF Parser
|
|
9
|
-
readme
|
|
6
|
+
name = "visual-parser"
|
|
7
|
+
version = "2.0.0"
|
|
8
|
+
description = "Standalone Visual-RAG PDF Parser - text extraction and Vision-LLM figure descriptions to JSONL"
|
|
9
|
+
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.10"
|
|
11
|
-
license
|
|
11
|
+
license = { text = "Apache-2.0" }
|
|
12
12
|
|
|
13
13
|
keywords = [
|
|
14
14
|
"pdf", "rag", "nougat", "vision-llm", "ocr",
|
|
@@ -20,7 +20,7 @@ classifiers = [
|
|
|
20
20
|
"Programming Language :: Python :: 3.10",
|
|
21
21
|
"Programming Language :: Python :: 3.11",
|
|
22
22
|
"Programming Language :: Python :: 3.12",
|
|
23
|
-
"License :: OSI Approved ::
|
|
23
|
+
"License :: OSI Approved :: Apache Software License",
|
|
24
24
|
"Operating System :: OS Independent",
|
|
25
25
|
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
26
26
|
"Topic :: Text Processing :: Markup",
|
|
@@ -40,16 +40,18 @@ dependencies = [
|
|
|
40
40
|
"google-generativeai==0.8.5",
|
|
41
41
|
"python-dotenv==1.1.0",
|
|
42
42
|
"tqdm==4.67.1",
|
|
43
|
+
"nltk>=3.8",
|
|
44
|
+
"python-Levenshtein>=0.20",
|
|
43
45
|
]
|
|
44
46
|
|
|
45
47
|
[project.optional-dependencies]
|
|
46
|
-
ocr = ["pytesseract==0.3.13"]
|
|
48
|
+
ocr = ["pytesseract==0.3.13"]
|
|
47
49
|
dev = ["pytest", "ruff", "mypy"]
|
|
48
50
|
|
|
49
51
|
[project.urls]
|
|
50
|
-
Homepage
|
|
51
|
-
Repository
|
|
52
|
-
"Docker Hub"
|
|
52
|
+
Homepage = "https://github.com/SmartLabNuclear/RADIANT_LLM"
|
|
53
|
+
Repository = "https://github.com/SmartLabNuclear/RADIANT_LLM"
|
|
54
|
+
"Docker Hub" = "https://hub.docker.com/r/zev94/radiant-llm"
|
|
53
55
|
|
|
54
56
|
[project.scripts]
|
|
55
57
|
visual-parser = "visual_parser.cli_main:main"
|
|
@@ -17,8 +17,8 @@ import sys
|
|
|
17
17
|
USAGE_EXAMPLES = """
|
|
18
18
|
Examples
|
|
19
19
|
--------
|
|
20
|
-
# Nougat (default) + GPT-5.
|
|
21
|
-
python visual-parser.py --input-dir ./my_pdfs
|
|
20
|
+
# Nougat (default) + GPT-5.4 vision
|
|
21
|
+
python visual-parser.py --input-dir ./my_pdfs
|
|
22
22
|
|
|
23
23
|
# Fast lightweight extraction + Gemini
|
|
24
24
|
python visual-parser.py --input-dir ./my_pdfs \\
|
|
@@ -47,7 +47,7 @@ def _build_arg_parser() -> argparse.ArgumentParser:
|
|
|
47
47
|
"Visual-RAG PDF Parser — detects new PDFs, extracts text and "
|
|
48
48
|
"figure descriptions, and writes three JSONL knowledge bases:\n"
|
|
49
49
|
" 01_chunks_kb.jsonl text chunks\n"
|
|
50
|
-
" 02_visuals_kb.jsonl visual descriptions\n"
|
|
50
|
+
" 02_visuals_kb.jsonl visual descriptions\n"
|
|
51
51
|
" 03_metadata_kb.jsonl document metadata"
|
|
52
52
|
),
|
|
53
53
|
formatter_class=argparse.RawDescriptionHelpFormatter,
|
|
@@ -108,20 +108,20 @@ def _build_arg_parser() -> argparse.ArgumentParser:
|
|
|
108
108
|
choices=["gpt", "gemini"],
|
|
109
109
|
default="gpt",
|
|
110
110
|
help=(
|
|
111
|
-
"gpt — OpenAI GPT-5.
|
|
112
|
-
"gemini — Google Gemini (set GEMINI_API_KEY in .env)."
|
|
113
|
-
),
|
|
114
|
-
)
|
|
111
|
+
"gpt — OpenAI GPT-5.4 (set OPENAI_API_KEY in .env).\n"
|
|
112
|
+
"gemini — Google Gemini (set GEMINI_API_KEY in .env)."
|
|
113
|
+
),
|
|
114
|
+
)
|
|
115
115
|
vision_group.add_argument(
|
|
116
116
|
"--vision-model",
|
|
117
117
|
default=None,
|
|
118
118
|
metavar="MODEL_NAME",
|
|
119
|
-
help=(
|
|
120
|
-
"Vision model name. Omit to use the latest for each provider:\n"
|
|
121
|
-
" gpt → gpt-5.
|
|
122
|
-
" gemini → gemini-3-pro-preview (also: gemini-2.5-flash, gemini-1.5-pro)"
|
|
123
|
-
),
|
|
124
|
-
)
|
|
119
|
+
help=(
|
|
120
|
+
"Vision model name. Omit to use the latest for each provider:\n"
|
|
121
|
+
" gpt → gpt-5.4 (also: gpt-5.5, gpt-5.3-chat-latest, gpt-5.2, gpt-5.1, gpt-5, gpt-4o, gpt-4.1)\n"
|
|
122
|
+
" gemini → gemini-3-pro-preview (also: gemini-2.5-flash, gemini-1.5-pro)"
|
|
123
|
+
),
|
|
124
|
+
)
|
|
125
125
|
vision_group.add_argument(
|
|
126
126
|
"--vision-detail",
|
|
127
127
|
choices=["low", "high", "auto"],
|
|
@@ -134,17 +134,17 @@ def _build_arg_parser() -> argparse.ArgumentParser:
|
|
|
134
134
|
)
|
|
135
135
|
vision_group.add_argument(
|
|
136
136
|
"--reasoning-effort",
|
|
137
|
-
choices=["minimal", "none", "low", "medium", "high", "xhigh"],
|
|
137
|
+
choices=["minimal", "none", "low", "medium", "high", "xhigh"],
|
|
138
138
|
default="medium",
|
|
139
|
-
help=(
|
|
140
|
-
"Reasoning effort for GPT-5.x models (ignored for Gemini and older GPT).\n"
|
|
141
|
-
" minimal/none — minimum reasoning, depending on model.\n"
|
|
142
|
-
" low — light reasoning.\n"
|
|
143
|
-
" medium — balanced (default).\n"
|
|
144
|
-
" high — deeper reasoning, slower.\n"
|
|
145
|
-
" xhigh — maximum depth (gpt-5.2, gpt-5.4, and gpt-5.5)."
|
|
146
|
-
),
|
|
147
|
-
)
|
|
139
|
+
help=(
|
|
140
|
+
"Reasoning effort for GPT-5.x models (ignored for Gemini and older GPT).\n"
|
|
141
|
+
" minimal/none — minimum reasoning, depending on model.\n"
|
|
142
|
+
" low — light reasoning.\n"
|
|
143
|
+
" medium — balanced (default).\n"
|
|
144
|
+
" high — deeper reasoning, slower.\n"
|
|
145
|
+
" xhigh — maximum depth (gpt-5.2, gpt-5.4, and gpt-5.5)."
|
|
146
|
+
),
|
|
147
|
+
)
|
|
148
148
|
vision_group.add_argument(
|
|
149
149
|
"--metadata-pages",
|
|
150
150
|
type=int,
|
|
@@ -173,6 +173,17 @@ def _build_arg_parser() -> argparse.ArgumentParser:
|
|
|
173
173
|
"Use after changing prompts, chunking strategy, or switching models."
|
|
174
174
|
),
|
|
175
175
|
)
|
|
176
|
+
misc_group.add_argument(
|
|
177
|
+
"--skip-text",
|
|
178
|
+
action="store_true",
|
|
179
|
+
help=(
|
|
180
|
+
"Skip text extraction (Step 1) and resume only the vision steps "
|
|
181
|
+
"(figure descriptions + metadata). Use when chunking already completed "
|
|
182
|
+
"but the run was interrupted mid-vision (e.g. API credit exhaustion). "
|
|
183
|
+
"PDFs already present in 02_visuals_kb.jsonl / 03_metadata_kb.jsonl "
|
|
184
|
+
"are skipped automatically — no duplicates."
|
|
185
|
+
),
|
|
186
|
+
)
|
|
176
187
|
misc_group.add_argument(
|
|
177
188
|
"--log-level",
|
|
178
189
|
choices=["DEBUG", "INFO", "WARNING", "ERROR"],
|
|
@@ -194,10 +205,10 @@ def main(argv=None) -> int:
|
|
|
194
205
|
args = parser.parse_args(argv)
|
|
195
206
|
|
|
196
207
|
# Default vision model per provider when not explicitly set
|
|
197
|
-
if args.vision_model is None:
|
|
198
|
-
args.vision_model = (
|
|
199
|
-
"gpt-5.
|
|
200
|
-
)
|
|
208
|
+
if args.vision_model is None:
|
|
209
|
+
args.vision_model = (
|
|
210
|
+
"gpt-5.4" if args.vision_provider == "gpt" else "gemini-3-pro-preview"
|
|
211
|
+
)
|
|
201
212
|
|
|
202
213
|
from visual_parser.config import ParserConfig
|
|
203
214
|
|
|
@@ -209,13 +220,14 @@ def main(argv=None) -> int:
|
|
|
209
220
|
chunk_size = args.chunk_size,
|
|
210
221
|
chunk_overlap = args.chunk_overlap,
|
|
211
222
|
vision_provider = args.vision_provider,
|
|
212
|
-
gpt_vision_model = args.vision_model if args.vision_provider == "gpt" else "gpt-5.
|
|
223
|
+
gpt_vision_model = args.vision_model if args.vision_provider == "gpt" else "gpt-5.4",
|
|
213
224
|
gemini_vision_model = args.vision_model if args.vision_provider == "gemini" else "gemini-3-pro-preview",
|
|
214
225
|
gpt_reasoning_effort = args.reasoning_effort,
|
|
215
226
|
vision_detail = args.vision_detail,
|
|
216
227
|
metadata_pages = args.metadata_pages,
|
|
217
228
|
max_workers = args.max_workers,
|
|
218
229
|
rebuild = args.rebuild,
|
|
230
|
+
skip_text = args.skip_text,
|
|
219
231
|
log_level = args.log_level,
|
|
220
232
|
)
|
|
221
233
|
|
|
@@ -15,7 +15,7 @@ import sys
|
|
|
15
15
|
USAGE_EXAMPLES = """
|
|
16
16
|
Examples
|
|
17
17
|
--------
|
|
18
|
-
# Nougat (default) + GPT-5.
|
|
18
|
+
# Nougat (default) + GPT-5.4 vision
|
|
19
19
|
python visual-parser.py --input-dir ./my_pdfs
|
|
20
20
|
|
|
21
21
|
# Fast lightweight extraction + Gemini
|
|
@@ -168,6 +168,17 @@ def _build_arg_parser() -> argparse.ArgumentParser:
|
|
|
168
168
|
"Use after changing prompts, chunking strategy, or switching models."
|
|
169
169
|
),
|
|
170
170
|
)
|
|
171
|
+
misc_group.add_argument(
|
|
172
|
+
"--skip-text",
|
|
173
|
+
action="store_true",
|
|
174
|
+
help=(
|
|
175
|
+
"Skip text extraction (Step 1) and resume only the vision steps "
|
|
176
|
+
"(figure descriptions + metadata). Use when chunking already completed "
|
|
177
|
+
"but the run was interrupted mid-vision (e.g. API credit exhaustion). "
|
|
178
|
+
"PDFs already present in 02_visuals_kb.jsonl / 03_metadata_kb.jsonl "
|
|
179
|
+
"are skipped automatically — no duplicates."
|
|
180
|
+
),
|
|
181
|
+
)
|
|
171
182
|
misc_group.add_argument(
|
|
172
183
|
"--log-level",
|
|
173
184
|
choices=["DEBUG", "INFO", "WARNING", "ERROR"],
|
|
@@ -206,6 +217,7 @@ def main(argv=None) -> int:
|
|
|
206
217
|
metadata_pages=args.metadata_pages,
|
|
207
218
|
max_workers=args.max_workers,
|
|
208
219
|
rebuild=args.rebuild,
|
|
220
|
+
skip_text=args.skip_text,
|
|
209
221
|
log_level=args.log_level,
|
|
210
222
|
)
|
|
211
223
|
|
|
@@ -121,6 +121,15 @@ class ParserConfig:
|
|
|
121
121
|
rebuild: bool = False
|
|
122
122
|
"""If True, reprocess all PDFs even if already recorded in 04_processed_pdfs.txt."""
|
|
123
123
|
|
|
124
|
+
skip_text: bool = False
|
|
125
|
+
"""
|
|
126
|
+
If True, skip text extraction (Step 1) entirely.
|
|
127
|
+
Use when chunking already completed but the vision steps (figures / metadata)
|
|
128
|
+
failed mid-run (e.g. API credit exhaustion). All PDFs in input_dir are
|
|
129
|
+
re-queued for vision steps; PDFs already present in 02_visuals_kb.jsonl /
|
|
130
|
+
03_metadata_kb.jsonl are skipped automatically.
|
|
131
|
+
"""
|
|
132
|
+
|
|
124
133
|
log_level: str = "ERROR"
|
|
125
134
|
|
|
126
135
|
# -------------------------------------------------------------------------
|
|
@@ -145,6 +154,7 @@ class ParserConfig:
|
|
|
145
154
|
metadata_pages = int(os.getenv("VISUAL_PARSER_METADATA_PAGES", "2")),
|
|
146
155
|
max_workers = int(os.getenv("VISUAL_PARSER_MAX_WORKERS", "4")),
|
|
147
156
|
rebuild = os.getenv("VISUAL_PARSER_REBUILD", "false").lower() == "true",
|
|
157
|
+
skip_text = os.getenv("VISUAL_PARSER_SKIP_TEXT", "false").lower() == "true",
|
|
148
158
|
log_level = os.getenv("VISUAL_PARSER_LOG_LEVEL", "ERROR"),
|
|
149
159
|
)
|
|
150
160
|
|
|
@@ -8,7 +8,7 @@ function in PDFAnalyser.py.
|
|
|
8
8
|
Output
|
|
9
9
|
------
|
|
10
10
|
One record per figure (or per page that contains at least one figure) is
|
|
11
|
-
appended to ``02_visuals_kb.jsonl`` in *output_dir*:
|
|
11
|
+
appended to ``02_visuals_kb.jsonl`` in *output_dir*:
|
|
12
12
|
|
|
13
13
|
{
|
|
14
14
|
"source": "myreport.pdf",
|
|
@@ -84,16 +84,16 @@ def describe_figures_for_new_pdfs(
|
|
|
84
84
|
vision_detail: str = "low",
|
|
85
85
|
raster_dpi: int = 200,
|
|
86
86
|
figure_prompt: str = FIGURE_PROMPT,
|
|
87
|
-
reasoning_effort: Optional[str] = "medium",
|
|
88
|
-
) -> None:
|
|
87
|
+
reasoning_effort: Optional[str] = "medium",
|
|
88
|
+
) -> None:
|
|
89
89
|
"""
|
|
90
90
|
For each PDF in *new_pdf_paths*, rasterise every page at *raster_dpi* DPI,
|
|
91
91
|
call the Vision LLM page-by-page, parse the figure descriptions, and
|
|
92
|
-
append the results to ``02_visuals_kb.jsonl`` in *output_dir*.
|
|
92
|
+
append the results to ``02_visuals_kb.jsonl`` in *output_dir*.
|
|
93
93
|
|
|
94
94
|
Args:
|
|
95
95
|
new_pdf_paths: Full paths of PDFs to describe.
|
|
96
|
-
output_dir: Directory where ``02_visuals_kb.jsonl`` is written.
|
|
96
|
+
output_dir: Directory where ``02_visuals_kb.jsonl`` is written.
|
|
97
97
|
vision_provider: ``'gpt'`` or ``'gemini'``.
|
|
98
98
|
vision_api_key: API key for the chosen provider.
|
|
99
99
|
vision_model: Vision model name string.
|
|
@@ -125,7 +125,7 @@ def describe_figures_for_new_pdfs(
|
|
|
125
125
|
|
|
126
126
|
if not page_images:
|
|
127
127
|
logger.info("No pages to describe (all PDFs failed to rasterise).")
|
|
128
|
-
return
|
|
128
|
+
return
|
|
129
129
|
|
|
130
130
|
# -----------------------------------------------------------------------
|
|
131
131
|
# Step 2 – Group page images by PDF name
|
|
@@ -135,16 +135,47 @@ def describe_figures_for_new_pdfs(
|
|
|
135
135
|
pages_by_pdf[record["pdf"]].append(record)
|
|
136
136
|
|
|
137
137
|
# -----------------------------------------------------------------------
|
|
138
|
-
# Step
|
|
138
|
+
# Step 2.5 – Build set of (source, page) pairs already on disk so a
|
|
139
|
+
# mid-run crash can be resumed at page granularity.
|
|
139
140
|
# -----------------------------------------------------------------------
|
|
140
|
-
|
|
141
|
+
figures_path = os.path.join(output_dir, "02_visuals_kb.jsonl")
|
|
142
|
+
done_pages: set = set()
|
|
143
|
+
if os.path.exists(figures_path):
|
|
144
|
+
with open(figures_path, encoding="utf-8") as _fh:
|
|
145
|
+
for _line in _fh:
|
|
146
|
+
_line = _line.strip()
|
|
147
|
+
if not _line:
|
|
148
|
+
continue
|
|
149
|
+
try:
|
|
150
|
+
_rec = json.loads(_line)
|
|
151
|
+
_src = _rec.get("source", "")
|
|
152
|
+
_pg = _rec.get("page")
|
|
153
|
+
if _src and _pg is not None:
|
|
154
|
+
done_pages.add((_src, _pg))
|
|
155
|
+
except Exception:
|
|
156
|
+
pass
|
|
157
|
+
if done_pages:
|
|
158
|
+
logger.info(
|
|
159
|
+
"Resuming: %d page(s) already in 02_visuals_kb.jsonl — will skip.",
|
|
160
|
+
len(done_pages),
|
|
161
|
+
)
|
|
162
|
+
|
|
163
|
+
# -----------------------------------------------------------------------
|
|
164
|
+
# Step 3 – Call Vision LLM per page; flush each page to disk immediately.
|
|
165
|
+
# Figure records are independent so per-page atomicity is safe.
|
|
166
|
+
# -----------------------------------------------------------------------
|
|
167
|
+
total_written = 0
|
|
141
168
|
|
|
142
169
|
for pdf_name, image_records in pages_by_pdf.items():
|
|
143
|
-
|
|
170
|
+
pdf_written = 0
|
|
144
171
|
|
|
145
172
|
for record in image_records:
|
|
146
|
-
page_number
|
|
147
|
-
image_bytes
|
|
173
|
+
page_number = record["page"]
|
|
174
|
+
image_bytes = record["bytes"]
|
|
175
|
+
|
|
176
|
+
if (pdf_name, page_number) in done_pages:
|
|
177
|
+
logger.debug("Skipping %s page %d (already in KB).", pdf_name, page_number)
|
|
178
|
+
continue
|
|
148
179
|
|
|
149
180
|
try:
|
|
150
181
|
raw_response = call_vision_llm(
|
|
@@ -159,8 +190,6 @@ def describe_figures_for_new_pdfs(
|
|
|
159
190
|
|
|
160
191
|
captions = _parse_llm_response(raw_response, pdf_name, page_number)
|
|
161
192
|
|
|
162
|
-
# Normalise: the model should return a list, but sometimes
|
|
163
|
-
# returns a single dict for single-figure pages.
|
|
164
193
|
if isinstance(captions, dict):
|
|
165
194
|
captions = [captions]
|
|
166
195
|
|
|
@@ -171,15 +200,27 @@ def describe_figures_for_new_pdfs(
|
|
|
171
200
|
)
|
|
172
201
|
continue
|
|
173
202
|
|
|
174
|
-
|
|
203
|
+
document_id = make_document_id(pdf_name)
|
|
204
|
+
page_rows: List[Dict] = []
|
|
205
|
+
for fig_idx, caption in enumerate(captions):
|
|
175
206
|
if not isinstance(caption, dict):
|
|
176
207
|
continue
|
|
177
208
|
description = caption.get("description")
|
|
178
209
|
if description is None:
|
|
179
210
|
continue
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
|
|
211
|
+
page_rows.append({
|
|
212
|
+
"source": pdf_name,
|
|
213
|
+
"page": page_number,
|
|
214
|
+
"document_id": document_id,
|
|
215
|
+
"figure_index": fig_idx,
|
|
216
|
+
"figure_id": f"{document_id}:p{page_number}:f{fig_idx}",
|
|
217
|
+
"description": description,
|
|
218
|
+
})
|
|
219
|
+
|
|
220
|
+
if page_rows:
|
|
221
|
+
append_to_jsonl(figures_path, page_rows)
|
|
222
|
+
pdf_written += len(page_rows)
|
|
223
|
+
total_written += len(page_rows)
|
|
183
224
|
|
|
184
225
|
except Exception as exc:
|
|
185
226
|
logger.error(
|
|
@@ -187,32 +228,12 @@ def describe_figures_for_new_pdfs(
|
|
|
187
228
|
pdf_name, page_number, exc,
|
|
188
229
|
)
|
|
189
230
|
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
len({k[0] for k in descriptions_by_page}))
|
|
231
|
+
if pdf_written:
|
|
232
|
+
logger.info("[FIGURES] %s: %d figure(s) written.", pdf_name, pdf_written)
|
|
233
|
+
else:
|
|
234
|
+
logger.info("[FIGURES] %s: no new figures (all pages done or none detected).", pdf_name)
|
|
195
235
|
|
|
196
|
-
|
|
197
|
-
|
|
198
|
-
# -----------------------------------------------------------------------
|
|
199
|
-
figure_rows: List[Dict] = []
|
|
200
|
-
|
|
201
|
-
for (pdf_name, page_number), descriptions in descriptions_by_page.items():
|
|
202
|
-
document_id = make_document_id(pdf_name)
|
|
203
|
-
for fig_idx, description in enumerate(descriptions):
|
|
204
|
-
figure_rows.append({
|
|
205
|
-
"source": pdf_name,
|
|
206
|
-
"page": page_number,
|
|
207
|
-
"document_id": document_id,
|
|
208
|
-
"figure_index": fig_idx,
|
|
209
|
-
"figure_id": f"{document_id}:p{page_number}:f{fig_idx}",
|
|
210
|
-
"description": description,
|
|
211
|
-
})
|
|
212
|
-
|
|
213
|
-
if figure_rows:
|
|
214
|
-
figures_path = os.path.join(output_dir, "02_visuals_kb.jsonl")
|
|
215
|
-
append_to_jsonl(figures_path, figure_rows)
|
|
216
|
-
print(f"[FIGURES] Wrote {len(figure_rows)} figure record(s) to 02_visuals_kb.jsonl.")
|
|
236
|
+
if total_written:
|
|
237
|
+
print(f"[FIGURES] Wrote {total_written} figure record(s) to 02_visuals_kb.jsonl.")
|
|
217
238
|
else:
|
|
218
|
-
logger.info("No figures
|
|
239
|
+
logger.info("No new figures written — all pages already processed or none detected.")
|