visual-parser 2.1.0__tar.gz → 2.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. visual_parser-2.2.0/PKG-INFO +331 -0
  2. visual_parser-2.2.0/README.md +286 -0
  3. {visual_parser-2.1.0 → visual_parser-2.2.0}/pyproject.toml +9 -1
  4. visual_parser-2.2.0/tests/test_cli_parity.py +193 -0
  5. visual_parser-2.2.0/tests/test_config.py +117 -0
  6. visual_parser-2.2.0/tests/test_figure_describer.py +176 -0
  7. visual_parser-2.2.0/tests/test_integration.py +113 -0
  8. visual_parser-2.2.0/tests/test_model_catalog.py +91 -0
  9. visual_parser-2.2.0/tests/test_ollama_local.py +292 -0
  10. visual_parser-2.2.0/tests/test_pipeline.py +168 -0
  11. visual_parser-2.2.0/tests/test_vision_llm.py +226 -0
  12. {visual_parser-2.1.0 → visual_parser-2.2.0}/visual_parser/__init__.py +3 -3
  13. {visual_parser-2.1.0 → visual_parser-2.2.0}/visual_parser/cli.py +61 -11
  14. {visual_parser-2.1.0 → visual_parser-2.2.0}/visual_parser/cli_main.py +35 -10
  15. {visual_parser-2.1.0 → visual_parser-2.2.0}/visual_parser/config.py +35 -8
  16. {visual_parser-2.1.0 → visual_parser-2.2.0}/visual_parser/figure_describer.py +302 -239
  17. {visual_parser-2.1.0 → visual_parser-2.2.0}/visual_parser/model_catalog.py +5 -4
  18. visual_parser-2.2.0/visual_parser/ollama_local.py +349 -0
  19. {visual_parser-2.1.0 → visual_parser-2.2.0}/visual_parser/pipeline.py +111 -22
  20. {visual_parser-2.1.0 → visual_parser-2.2.0}/visual_parser/prompts.py +33 -0
  21. {visual_parser-2.1.0 → visual_parser-2.2.0}/visual_parser/vision_llm.py +139 -40
  22. visual_parser-2.2.0/visual_parser.egg-info/PKG-INFO +331 -0
  23. {visual_parser-2.1.0 → visual_parser-2.2.0}/visual_parser.egg-info/SOURCES.txt +9 -0
  24. {visual_parser-2.1.0 → visual_parser-2.2.0}/visual_parser.egg-info/requires.txt +1 -0
  25. visual_parser-2.1.0/PKG-INFO +0 -231
  26. visual_parser-2.1.0/README.md +0 -187
  27. visual_parser-2.1.0/visual_parser.egg-info/PKG-INFO +0 -231
  28. {visual_parser-2.1.0 → visual_parser-2.2.0}/LICENSE +0 -0
  29. {visual_parser-2.1.0 → visual_parser-2.2.0}/setup.cfg +0 -0
  30. {visual_parser-2.1.0 → visual_parser-2.2.0}/visual_parser/__main__.py +0 -0
  31. {visual_parser-2.1.0 → visual_parser-2.2.0}/visual_parser/jsonl_writer.py +0 -0
  32. {visual_parser-2.1.0 → visual_parser-2.2.0}/visual_parser/metadata_extractor.py +0 -0
  33. {visual_parser-2.1.0 → visual_parser-2.2.0}/visual_parser/nougat_engine.py +0 -0
  34. {visual_parser-2.1.0 → visual_parser-2.2.0}/visual_parser/openai_gateway.py +0 -0
  35. {visual_parser-2.1.0 → visual_parser-2.2.0}/visual_parser/pdf_tracker.py +0 -0
  36. {visual_parser-2.1.0 → visual_parser-2.2.0}/visual_parser/text_extractor.py +0 -0
  37. {visual_parser-2.1.0 → visual_parser-2.2.0}/visual_parser.egg-info/dependency_links.txt +0 -0
  38. {visual_parser-2.1.0 → visual_parser-2.2.0}/visual_parser.egg-info/entry_points.txt +0 -0
  39. {visual_parser-2.1.0 → visual_parser-2.2.0}/visual_parser.egg-info/top_level.txt +0 -0
@@ -0,0 +1,331 @@
1
+ Metadata-Version: 2.4
2
+ Name: visual-parser
3
+ Version: 2.2.0
4
+ Summary: Standalone Visual-RAG PDF Parser - text extraction and Vision-LLM figure descriptions to JSONL
5
+ Author: Zavier N. Ndum
6
+ Author-email: zavier.ndum@tamu.edu
7
+ License-Expression: Apache-2.0
8
+ Project-URL: Homepage, https://github.com/SmartLabNuclear/RADIANT_LLM
9
+ Project-URL: Repository, https://github.com/SmartLabNuclear/RADIANT_LLM
10
+ Project-URL: Docker Hub, https://hub.docker.com/r/zev94/radiant-llm
11
+ Keywords: pdf,rag,nougat,vision-llm,ocr,document-parsing,jsonl,knowledge-base
12
+ Classifier: Programming Language :: Python :: 3
13
+ Classifier: Programming Language :: Python :: 3.10
14
+ Classifier: Programming Language :: Python :: 3.11
15
+ Classifier: Programming Language :: Python :: 3.12
16
+ Classifier: Operating System :: OS Independent
17
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
18
+ Classifier: Topic :: Text Processing :: Markup
19
+ Classifier: Intended Audience :: Science/Research
20
+ Requires-Python: >=3.10
21
+ Description-Content-Type: text/markdown
22
+ License-File: LICENSE
23
+ Requires-Dist: PyMuPDF>=1.23.0
24
+ Requires-Dist: Pillow>=10.0.0
25
+ Requires-Dist: torch>=2.1.0
26
+ Requires-Dist: transformers>=4.38.0
27
+ Requires-Dist: huggingface-hub>=0.36.0
28
+ Requires-Dist: langchain-community>=0.0.20
29
+ Requires-Dist: langchain>=0.1.0
30
+ Requires-Dist: langchain-text-splitters>=0.0.1
31
+ Requires-Dist: openai>=1.14.0
32
+ Requires-Dist: google-generativeai>=0.5.0
33
+ Requires-Dist: requests>=2.31.0
34
+ Requires-Dist: python-dotenv>=1.0.0
35
+ Requires-Dist: tqdm>=4.66.0
36
+ Requires-Dist: nltk>=3.8
37
+ Requires-Dist: python-Levenshtein>=0.20
38
+ Provides-Extra: ocr
39
+ Requires-Dist: pytesseract>=0.3.10; extra == "ocr"
40
+ Provides-Extra: dev
41
+ Requires-Dist: pytest; extra == "dev"
42
+ Requires-Dist: ruff; extra == "dev"
43
+ Requires-Dist: mypy; extra == "dev"
44
+ Dynamic: license-file
45
+
46
+ # visual-parser (Standalone Visual-RAG PDF Ingestion)
47
+
48
+ ![Python 3.10+](https://img.shields.io/badge/Python-3.10%2B-brightgreen.svg)
49
+ ![PyTorch](https://img.shields.io/badge/PyTorch-2.1%2B-ee4c2c.svg)
50
+ ![LangChain](https://img.shields.io/badge/LangChain-0.1%2B-1C3C3C.svg)
51
+ ![CUDA](https://img.shields.io/badge/CUDA-optional-76B900.svg)
52
+
53
+ `visual-parser` converts PDFs into a multi-modal JSONL knowledge base (text chunks, figure descriptions, metadata). It was extracted from [RADIANT-LLM](https://github.com/SmartLabNuclear/RADIANT_LLM) as a standalone PDF-ingestion tool, independent of any chatbot.
54
+
55
+ 1) Run `visual-parser` on curated PDFs to generate JSONL KB files.
56
+ 2) Point any downstream RAG system at the generated KB — RADIANT-LLM, AutoSAM, and AutoFLUKA all consume the identical JSONL/registry format, so the same KB works with any of them without re-parsing.
57
+
58
+ ## Table of Contents
59
+
60
+ - [Outputs (JSONL KB)](#outputs-jsonl-kb)
61
+ - [API Keys (.env)](#api-keys-env)
62
+ - [Install and Run](#install-and-run)
63
+ - [Via pip](#via-pip)
64
+ - [Via Docker](#via-docker)
65
+ - [From source](#from-source)
66
+ - [Common configuration flags](#common-configuration-flags)
67
+ - [Local Ollama vision provider](#local-ollama-vision-provider)
68
+ - [Citation](#citation)
69
+ - [License](#license)
70
+
71
+ ## Outputs (JSONL KB)
72
+
73
+ By default, the pipeline writes:
74
+ - `01_chunks_kb.jsonl`: chunked text extracted from PDFs (Nougat by default).
75
+ - `02_visuals_kb.jsonl`: figure/page visual descriptions (Vision LLM).
76
+ - `03_metadata_kb.jsonl`: document metadata rows (title/author/etc.).
77
+
78
+ Alongside the KB, the pipeline also writes two bookkeeping files: `04_processed_pdfs.txt` (tracks which PDFs have already been processed, so re-runs skip them unless `--rebuild`) and `05_pipeline.log` (run log at the verbosity set by `--log-level`, default `ERROR`).
79
+
80
+ ## API Keys (.env)
81
+
82
+ Provide at least one provider:
83
+ - `OPENAI_API_KEY` (OpenAI)
84
+ - `GEMINI_API_KEY` (Gemini)
85
+
86
+ Set these via a `.env` file (checked, in order: `~/.config/visual-parser/.env`, a `.env` in the current directory, `$VISUAL_PARSER_ENV_FILE` if set) or as regular OS environment variables (e.g. `setx` on Windows).
87
+
88
+ Optional:
89
+ - `HF_TOKEN` (for gated Hugging Face models)
90
+ - `PORTKEY_API_KEY` + `PORTKEY_OPENAI_PROVIDER_SLUG` — routes OpenAI-family calls (vision LLM + `--list-models`) through a [Portkey](https://portkey.ai) gateway instead of OpenAI directly. Only takes effect when `OPENAI_API_KEY` is absent/empty; direct OpenAI wins when both are set. Set `VISUAL_PARSER_FORCE_PORTKEY=true` to force Portkey even when `OPENAI_API_KEY` is set at the OS/system level.
91
+
92
+ ## Install and Run
93
+
94
+ Three ways to run `visual-parser`, in order of setup effort: [pip install](#via-pip), [Docker](#via-docker), or [from source](#from-source). All three share the identical CLI and flags — see [Common configuration flags](#common-configuration-flags).
95
+
96
+ ### Via pip
97
+
98
+ ```bash
99
+ pip install visual-parser
100
+ ```
101
+
102
+ Check the install:
103
+
104
+ ```bash
105
+ visual-parser --version
106
+ ```
107
+
108
+ Run against a folder of PDFs:
109
+
110
+ ```bash
111
+ visual-parser --input-dir /path/to/pdfs --output-dir /path/to/pdfs
112
+ ```
113
+
114
+ #### GPU acceleration
115
+
116
+ `--text-mode nougat` (the default) uses a CUDA GPU automatically when one is available (`torch.cuda.is_available()`). A plain `pip install visual-parser` (or `pip install torch`) installs PyPI's default CPU-only torch wheel, even on a GPU machine. Install the matching CUDA build from PyTorch's index instead:
117
+
118
+ ```bash
119
+ pip install torch --index-url https://download.pytorch.org/whl/cu130
120
+ ```
121
+
122
+ Use `cu130` or newer, not `cu126` — `cu126` only covers compute capability up to `sm_90` (Hopper); on a newer GPU (e.g. Blackwell, `sm_120`), `torch.cuda.is_available()` still reports `True` but the first kernel launch fails. `cu130` covers Blackwell and every architecture `cu126` covered, and requires an r580+ NVIDIA driver. Pick the exact tag for your driver at [pytorch.org/get-started](https://pytorch.org/get-started/locally/) if `cu130` doesn't apply. Reinstalling `visual-parser` afterward does not downgrade this back to CPU. Verify with:
123
+
124
+ ```bash
125
+ python -c "import torch; print(torch.cuda.is_available(), torch.cuda.get_device_name(0) if torch.cuda.is_available() else None)"
126
+ ```
127
+
128
+ ### Via Docker
129
+
130
+ Prebuilt images are on **[zev94/radiant-llm](https://hub.docker.com/r/zev94/radiant-llm)** under the **visual-parser** tags:
131
+
132
+ | Tag | Description |
133
+ |-----|-------------|
134
+ | `visual-parser-latest` | Always latest build (rolling) |
135
+ | `visual-parser-2.1.0` | Pinned release |
136
+ | `visual-parser-2.0` | Pinned release |
137
+ | `visual-parser-1.0.2` | Legacy |
138
+ | `visual-parser-1.0` | Legacy — v1.0.0, stale |
139
+
140
+ #### Install Docker
141
+
142
+ Docker Desktop (Windows/macOS) or Docker Engine (Linux).
143
+
144
+ #### Pull the image
145
+
146
+ ```bash
147
+ docker pull zev94/radiant-llm:visual-parser-latest
148
+ ```
149
+
150
+ #### Run (input and output on the same mounted folder)
151
+
152
+ Windows PowerShell:
153
+ ```powershell
154
+ docker run --rm --env-file .env `
155
+ -v "C:\path\to\pdfs:/data" `
156
+ zev94/radiant-llm:visual-parser-latest `
157
+ --input-dir /data --output-dir /data
158
+ ```
159
+
160
+ Linux / WSL:
161
+ ```bash
162
+ docker run --rm --env-file .env \
163
+ -v "/path/to/pdfs:/data" \
164
+ zev94/radiant-llm:visual-parser-latest \
165
+ --input-dir /data --output-dir /data
166
+ ```
167
+
168
+ #### Run (separate output directory)
169
+
170
+ Windows PowerShell:
171
+ ```powershell
172
+ docker run --rm --env-file .env `
173
+ -v "C:\path\to\pdfs:/data" `
174
+ -v "C:\path\to\out:/out" `
175
+ zev94/radiant-llm:visual-parser-latest `
176
+ --input-dir /data --output-dir /out
177
+ ```
178
+
179
+ #### GPU acceleration
180
+
181
+ Add `--gpus all` to any run command above to use an NVIDIA GPU instead of CPU for `--text-mode nougat` (the default). Works out of the box with Docker Desktop's WSL2 backend; no extra `nvidia-container-toolkit` install needed on most machines. Omit it (or run on a machine with no GPU) to fall back to CPU automatically.
182
+
183
+ Requires an NVIDIA driver supporting CUDA 13.0+ (r580 or newer) on the host — check with `nvidia-smi`; update from [nvidia.com/Download](https://www.nvidia.com/Download/index.aspx) if it reports an older version. Without it, `torch.cuda.is_available()` still reports `True`, but the first Nougat inference call fails with `CUDA error: no kernel image is available`:
184
+
185
+ ```powershell
186
+ docker run --rm --gpus all --env-file .env `
187
+ -v "C:\path\to\pdfs:/data" `
188
+ zev94/radiant-llm:visual-parser-latest `
189
+ --input-dir /data --output-dir /data
190
+ ```
191
+
192
+ #### Troubleshooting: GPU not detected
193
+
194
+ - Confirm the host sees the GPU at all: `nvidia-smi` (run from WSL if on Windows).
195
+ - Confirm the driver supports CUDA 13.0+: check the `CUDA Version` line in that output — needs to read 13.0 or higher (roughly r580+). Update from [nvidia.com/Download](https://www.nvidia.com/Download/index.aspx) if it reads lower.
196
+ - Confirm Docker can reach the GPU: `docker run --rm --gpus all nvidia/cuda:12.2.0-base-ubuntu22.04 nvidia-smi`.
197
+ - If Docker Desktop's WSL2 backend still doesn't see the GPU after a driver update, restart the WSL VM: `wsl --shutdown`, then restart Docker Desktop.
198
+
199
+ #### Offline install (legacy .tar)
200
+
201
+ ```powershell
202
+ docker load -i .\visual-parser_2.1.0.tar
203
+ docker images # use the tag printed by Docker
204
+ ```
205
+
206
+ #### Model overrides (optional)
207
+
208
+ Default vision model is **GPT-5.4** when using `--vision-provider gpt`. Override on the command line (works identically via Docker or a plain CLI install):
209
+
210
+ ```powershell
211
+ docker run --rm --env-file .env -v "C:\path\to\pdfs:/data" `
212
+ zev94/radiant-llm:visual-parser-latest `
213
+ --input-dir /data --output-dir /data --vision-model gpt-5.6
214
+ ```
215
+
216
+ ### From source
217
+
218
+ Clone the repository and install dependencies directly, without packaging:
219
+
220
+ ```bash
221
+ git clone https://github.com/SmartLabNuclear/RADIANT_LLM.git
222
+ cd RADIANT_LLM/Visual-Parser
223
+ pip install -r requirements.txt
224
+ ```
225
+
226
+ Run the top-level script:
227
+
228
+ ```bash
229
+ python visual-parser.py --input-dir ./my_pdfs
230
+ ```
231
+
232
+ or as a module:
233
+
234
+ ```bash
235
+ python -m visual_parser --input-dir ./my_pdfs
236
+ ```
237
+
238
+ Both use the same CLI and flags as [Via pip](#via-pip). GPU acceleration follows the same rule: a plain `pip install -r requirements.txt` also resolves to the CPU-only torch wheel by default — see [GPU acceleration](#gpu-acceleration) under Via pip for the CUDA-build install command.
239
+
240
+ ## Common configuration flags
241
+
242
+ Full flag list via `--help` — Docker:
243
+
244
+ ```bash
245
+ docker run --rm zev94/radiant-llm:visual-parser-latest --help
246
+ ```
247
+
248
+ or plain CLI:
249
+
250
+ ```bash
251
+ visual-parser --help
252
+ ```
253
+
254
+ For copy-paste Docker examples (vision presets, text modes, workers, rebuild), see [`docker-usage-examples.md`](docker-usage-examples.md).
255
+
256
+ Paths:
257
+ - `--input-dir` / `-i` (required unless `--list-models` is given)
258
+ - `--output-dir` / `-o` (default: same as input)
259
+
260
+ Text extraction:
261
+ - `--text-mode nougat|lightweight` (default: `nougat`)
262
+ - `--nougat-model facebook/nougat-small`
263
+ - `--chunk-size 500`
264
+ - `--chunk-overlap 100`
265
+
266
+ Vision LLM:
267
+ - `--vision-provider gpt|gemini|ollama` (default: `gpt`) — see [Local Ollama vision provider](#local-ollama-vision-provider) for the `ollama` option
268
+ - `--vision-model gpt-5.6` (or `gpt-4o`, `gemini-2.5-flash`, etc. — run `--list-models` for what's actually live on your account; omit entirely for `ollama` to auto-select)
269
+ - `--vision-detail low|high|auto` (default: `low`)
270
+ - `--reasoning-effort minimal|none|low|medium|high|xhigh` (default: `medium`)
271
+ - `--metadata-pages 2`
272
+ - `--vision-context-pages 0` — include N adjacent pages (before and after) as context in each figure-description call, to help with figures/captions that span a page break. 0 (default) is today's single-page behavior; more context pages means more tokens per call
273
+
274
+ Performance / misc:
275
+ - `--max-workers 4`
276
+ - `--rebuild` (reprocess everything; ignore `04_processed_pdfs.txt`)
277
+ - `--skip-text` (skip text extraction and resume only the vision steps; use after an interrupted run)
278
+ - `--list-models` — print the live, currently-available vision models for whichever provider(s) you have a key configured for, then exit (no PDF processing). Reflects Portkey's catalog too when that's active.
279
+ - `--version` / `-V`
280
+ - `--log-level DEBUG|INFO|WARNING|ERROR` (default: `ERROR`)
281
+
282
+ ## Local Ollama vision provider
283
+
284
+ `--vision-provider ollama` runs figure description and metadata extraction against a locally-pulled model served by [Ollama](https://ollama.com), instead of a cloud API — no API key needed, no per-call cost.
285
+
286
+ ```bash
287
+ visual-parser --input-dir /path/to/pdfs --vision-provider ollama
288
+ ```
289
+
290
+ Omit `--vision-model` and the largest locally-pulled, vision-capable model that fits in currently-free GPU VRAM is auto-selected (logged to the console/`05_pipeline.log` as it's picked). To use a specific model instead, pass it explicitly:
291
+
292
+ ```bash
293
+ visual-parser --input-dir /path/to/pdfs --vision-provider ollama --vision-model qwen2.5-vl:32b
294
+ ```
295
+
296
+ Auto-selection:
297
+ - Pull at least one vision-capable model first: `ollama pull llava` (or any other vision-capable tag).
298
+ - Vision capability is detected via Ollama's reported model capabilities, falling back to a name heuristic (`llava`, `vision`, `-vl`, `pixtral`, `moondream`, `minicpm-v`, `bakllava`) on older Ollama versions that don't report it.
299
+ - The selection budget is 85% of currently-free VRAM (not total VRAM) — this correctly leaves room for whatever else is already resident (e.g. Nougat's model during the same pipeline run) and for the model's own KV-cache/context overhead, which isn't included in its on-disk size.
300
+ - If no CUDA GPU is visible to the process at all — including the case of running in a Docker container without `--gpus all` while Ollama runs on the host with a real GPU — CPU-only Ollama is treated as a legitimate setup: auto-selection falls back to the smallest pulled vision-capable model instead of refusing, printing a clear warning that it's running CPU-only and will be slow. Pass an explicit `--vision-model` to pick a different one.
301
+ - If a real GPU *is* visible but no vision-capable model fits the free-VRAM budget, that case stays strict — auto-selection raises an error asking for an explicit `--vision-model`, rather than silently falling back to heavy CPU offload.
302
+
303
+ Reachability: checks `OLLAMA_BASE_URL` (if set) → `http://localhost:11434` → `http://host.docker.internal:11434` (the containerized case). On Docker Desktop (Windows/macOS) the container reaches a host-installed Ollama automatically; on native Linux Docker Engine, add `--add-host=host.docker.internal:host-gateway` to the `docker run` command, or set `OLLAMA_BASE_URL` directly.
304
+
305
+ ---
306
+
307
+ ## Citation
308
+
309
+ If you use RADIANT-LLM or the accompanying evaluation materials, please cite the journal article:
310
+
311
+ ```bibtex
312
+ @article{ndum2026retrieval,
313
+ title={A retrieval-augmented, domain-intelligent agentic framework for reliable decision support in safety-critical nuclear engineering},
314
+ author={Ndum, Zavier Ndum and Tao, Jian and Ford, John and Yim, Mansung and Liu, Yang},
315
+ journal={Reliability Engineering \& System Safety},
316
+ pages={113057},
317
+ year={2026},
318
+ publisher={Elsevier}
319
+ }
320
+ ```
321
+
322
+ Journal: *Reliability Engineering & System Safety* (2026), article 113057
323
+ Preprint: https://arxiv.org/abs/2604.22755
324
+
325
+ ---
326
+
327
+ ## License
328
+
329
+ Copyright 2026 Zavier N. Ndum
330
+
331
+ This project is licensed under the Apache License 2.0, the same license as its parent project, [RADIANT-LLM](https://github.com/SmartLabNuclear/RADIANT_LLM). See the [LICENSE](https://github.com/SmartLabNuclear/RADIANT_LLM/blob/main/LICENSE) file in the RADIANT_LLM repository, or the `LICENSE` file bundled with this package, for the full license text.
@@ -0,0 +1,286 @@
1
+ # visual-parser (Standalone Visual-RAG PDF Ingestion)
2
+
3
+ ![Python 3.10+](https://img.shields.io/badge/Python-3.10%2B-brightgreen.svg)
4
+ ![PyTorch](https://img.shields.io/badge/PyTorch-2.1%2B-ee4c2c.svg)
5
+ ![LangChain](https://img.shields.io/badge/LangChain-0.1%2B-1C3C3C.svg)
6
+ ![CUDA](https://img.shields.io/badge/CUDA-optional-76B900.svg)
7
+
8
+ `visual-parser` converts PDFs into a multi-modal JSONL knowledge base (text chunks, figure descriptions, metadata). It was extracted from [RADIANT-LLM](https://github.com/SmartLabNuclear/RADIANT_LLM) as a standalone PDF-ingestion tool, independent of any chatbot.
9
+
10
+ 1) Run `visual-parser` on curated PDFs to generate JSONL KB files.
11
+ 2) Point any downstream RAG system at the generated KB — RADIANT-LLM, AutoSAM, and AutoFLUKA all consume the identical JSONL/registry format, so the same KB works with any of them without re-parsing.
12
+
13
+ ## Table of Contents
14
+
15
+ - [Outputs (JSONL KB)](#outputs-jsonl-kb)
16
+ - [API Keys (.env)](#api-keys-env)
17
+ - [Install and Run](#install-and-run)
18
+ - [Via pip](#via-pip)
19
+ - [Via Docker](#via-docker)
20
+ - [From source](#from-source)
21
+ - [Common configuration flags](#common-configuration-flags)
22
+ - [Local Ollama vision provider](#local-ollama-vision-provider)
23
+ - [Citation](#citation)
24
+ - [License](#license)
25
+
26
+ ## Outputs (JSONL KB)
27
+
28
+ By default, the pipeline writes:
29
+ - `01_chunks_kb.jsonl`: chunked text extracted from PDFs (Nougat by default).
30
+ - `02_visuals_kb.jsonl`: figure/page visual descriptions (Vision LLM).
31
+ - `03_metadata_kb.jsonl`: document metadata rows (title/author/etc.).
32
+
33
+ Alongside the KB, the pipeline also writes two bookkeeping files: `04_processed_pdfs.txt` (tracks which PDFs have already been processed, so re-runs skip them unless `--rebuild`) and `05_pipeline.log` (run log at the verbosity set by `--log-level`, default `ERROR`).
34
+
35
+ ## API Keys (.env)
36
+
37
+ Provide at least one provider:
38
+ - `OPENAI_API_KEY` (OpenAI)
39
+ - `GEMINI_API_KEY` (Gemini)
40
+
41
+ Set these via a `.env` file (checked, in order: `~/.config/visual-parser/.env`, a `.env` in the current directory, `$VISUAL_PARSER_ENV_FILE` if set) or as regular OS environment variables (e.g. `setx` on Windows).
42
+
43
+ Optional:
44
+ - `HF_TOKEN` (for gated Hugging Face models)
45
+ - `PORTKEY_API_KEY` + `PORTKEY_OPENAI_PROVIDER_SLUG` — routes OpenAI-family calls (vision LLM + `--list-models`) through a [Portkey](https://portkey.ai) gateway instead of OpenAI directly. Only takes effect when `OPENAI_API_KEY` is absent/empty; direct OpenAI wins when both are set. Set `VISUAL_PARSER_FORCE_PORTKEY=true` to force Portkey even when `OPENAI_API_KEY` is set at the OS/system level.
46
+
47
+ ## Install and Run
48
+
49
+ Three ways to run `visual-parser`, in order of setup effort: [pip install](#via-pip), [Docker](#via-docker), or [from source](#from-source). All three share the identical CLI and flags — see [Common configuration flags](#common-configuration-flags).
50
+
51
+ ### Via pip
52
+
53
+ ```bash
54
+ pip install visual-parser
55
+ ```
56
+
57
+ Check the install:
58
+
59
+ ```bash
60
+ visual-parser --version
61
+ ```
62
+
63
+ Run against a folder of PDFs:
64
+
65
+ ```bash
66
+ visual-parser --input-dir /path/to/pdfs --output-dir /path/to/pdfs
67
+ ```
68
+
69
+ #### GPU acceleration
70
+
71
+ `--text-mode nougat` (the default) uses a CUDA GPU automatically when one is available (`torch.cuda.is_available()`). A plain `pip install visual-parser` (or `pip install torch`) installs PyPI's default CPU-only torch wheel, even on a GPU machine. Install the matching CUDA build from PyTorch's index instead:
72
+
73
+ ```bash
74
+ pip install torch --index-url https://download.pytorch.org/whl/cu130
75
+ ```
76
+
77
+ Use `cu130` or newer, not `cu126` — `cu126` only covers compute capability up to `sm_90` (Hopper); on a newer GPU (e.g. Blackwell, `sm_120`), `torch.cuda.is_available()` still reports `True` but the first kernel launch fails. `cu130` covers Blackwell and every architecture `cu126` covered, and requires an r580+ NVIDIA driver. Pick the exact tag for your driver at [pytorch.org/get-started](https://pytorch.org/get-started/locally/) if `cu130` doesn't apply. Reinstalling `visual-parser` afterward does not downgrade this back to CPU. Verify with:
78
+
79
+ ```bash
80
+ python -c "import torch; print(torch.cuda.is_available(), torch.cuda.get_device_name(0) if torch.cuda.is_available() else None)"
81
+ ```
82
+
83
+ ### Via Docker
84
+
85
+ Prebuilt images are on **[zev94/radiant-llm](https://hub.docker.com/r/zev94/radiant-llm)** under the **visual-parser** tags:
86
+
87
+ | Tag | Description |
88
+ |-----|-------------|
89
+ | `visual-parser-latest` | Always latest build (rolling) |
90
+ | `visual-parser-2.1.0` | Pinned release |
91
+ | `visual-parser-2.0` | Pinned release |
92
+ | `visual-parser-1.0.2` | Legacy |
93
+ | `visual-parser-1.0` | Legacy — v1.0.0, stale |
94
+
95
+ #### Install Docker
96
+
97
+ Docker Desktop (Windows/macOS) or Docker Engine (Linux).
98
+
99
+ #### Pull the image
100
+
101
+ ```bash
102
+ docker pull zev94/radiant-llm:visual-parser-latest
103
+ ```
104
+
105
+ #### Run (input and output on the same mounted folder)
106
+
107
+ Windows PowerShell:
108
+ ```powershell
109
+ docker run --rm --env-file .env `
110
+ -v "C:\path\to\pdfs:/data" `
111
+ zev94/radiant-llm:visual-parser-latest `
112
+ --input-dir /data --output-dir /data
113
+ ```
114
+
115
+ Linux / WSL:
116
+ ```bash
117
+ docker run --rm --env-file .env \
118
+ -v "/path/to/pdfs:/data" \
119
+ zev94/radiant-llm:visual-parser-latest \
120
+ --input-dir /data --output-dir /data
121
+ ```
122
+
123
+ #### Run (separate output directory)
124
+
125
+ Windows PowerShell:
126
+ ```powershell
127
+ docker run --rm --env-file .env `
128
+ -v "C:\path\to\pdfs:/data" `
129
+ -v "C:\path\to\out:/out" `
130
+ zev94/radiant-llm:visual-parser-latest `
131
+ --input-dir /data --output-dir /out
132
+ ```
133
+
134
+ #### GPU acceleration
135
+
136
+ Add `--gpus all` to any run command above to use an NVIDIA GPU instead of CPU for `--text-mode nougat` (the default). Works out of the box with Docker Desktop's WSL2 backend; no extra `nvidia-container-toolkit` install needed on most machines. Omit it (or run on a machine with no GPU) to fall back to CPU automatically.
137
+
138
+ Requires an NVIDIA driver supporting CUDA 13.0+ (r580 or newer) on the host — check with `nvidia-smi`; update from [nvidia.com/Download](https://www.nvidia.com/Download/index.aspx) if it reports an older version. Without it, `torch.cuda.is_available()` still reports `True`, but the first Nougat inference call fails with `CUDA error: no kernel image is available`:
139
+
140
+ ```powershell
141
+ docker run --rm --gpus all --env-file .env `
142
+ -v "C:\path\to\pdfs:/data" `
143
+ zev94/radiant-llm:visual-parser-latest `
144
+ --input-dir /data --output-dir /data
145
+ ```
146
+
147
+ #### Troubleshooting: GPU not detected
148
+
149
+ - Confirm the host sees the GPU at all: `nvidia-smi` (run from WSL if on Windows).
150
+ - Confirm the driver supports CUDA 13.0+: check the `CUDA Version` line in that output — needs to read 13.0 or higher (roughly r580+). Update from [nvidia.com/Download](https://www.nvidia.com/Download/index.aspx) if it reads lower.
151
+ - Confirm Docker can reach the GPU: `docker run --rm --gpus all nvidia/cuda:12.2.0-base-ubuntu22.04 nvidia-smi`.
152
+ - If Docker Desktop's WSL2 backend still doesn't see the GPU after a driver update, restart the WSL VM: `wsl --shutdown`, then restart Docker Desktop.
153
+
154
+ #### Offline install (legacy .tar)
155
+
156
+ ```powershell
157
+ docker load -i .\visual-parser_2.1.0.tar
158
+ docker images # use the tag printed by Docker
159
+ ```
160
+
161
+ #### Model overrides (optional)
162
+
163
+ Default vision model is **GPT-5.4** when using `--vision-provider gpt`. Override on the command line (works identically via Docker or a plain CLI install):
164
+
165
+ ```powershell
166
+ docker run --rm --env-file .env -v "C:\path\to\pdfs:/data" `
167
+ zev94/radiant-llm:visual-parser-latest `
168
+ --input-dir /data --output-dir /data --vision-model gpt-5.6
169
+ ```
170
+
171
+ ### From source
172
+
173
+ Clone the repository and install dependencies directly, without packaging:
174
+
175
+ ```bash
176
+ git clone https://github.com/SmartLabNuclear/RADIANT_LLM.git
177
+ cd RADIANT_LLM/Visual-Parser
178
+ pip install -r requirements.txt
179
+ ```
180
+
181
+ Run the top-level script:
182
+
183
+ ```bash
184
+ python visual-parser.py --input-dir ./my_pdfs
185
+ ```
186
+
187
+ or as a module:
188
+
189
+ ```bash
190
+ python -m visual_parser --input-dir ./my_pdfs
191
+ ```
192
+
193
+ Both use the same CLI and flags as [Via pip](#via-pip). GPU acceleration follows the same rule: a plain `pip install -r requirements.txt` also resolves to the CPU-only torch wheel by default — see [GPU acceleration](#gpu-acceleration) under Via pip for the CUDA-build install command.
194
+
195
+ ## Common configuration flags
196
+
197
+ Full flag list via `--help` — Docker:
198
+
199
+ ```bash
200
+ docker run --rm zev94/radiant-llm:visual-parser-latest --help
201
+ ```
202
+
203
+ or plain CLI:
204
+
205
+ ```bash
206
+ visual-parser --help
207
+ ```
208
+
209
+ For copy-paste Docker examples (vision presets, text modes, workers, rebuild), see [`docker-usage-examples.md`](docker-usage-examples.md).
210
+
211
+ Paths:
212
+ - `--input-dir` / `-i` (required unless `--list-models` is given)
213
+ - `--output-dir` / `-o` (default: same as input)
214
+
215
+ Text extraction:
216
+ - `--text-mode nougat|lightweight` (default: `nougat`)
217
+ - `--nougat-model facebook/nougat-small`
218
+ - `--chunk-size 500`
219
+ - `--chunk-overlap 100`
220
+
221
+ Vision LLM:
222
+ - `--vision-provider gpt|gemini|ollama` (default: `gpt`) — see [Local Ollama vision provider](#local-ollama-vision-provider) for the `ollama` option
223
+ - `--vision-model gpt-5.6` (or `gpt-4o`, `gemini-2.5-flash`, etc. — run `--list-models` for what's actually live on your account; omit entirely for `ollama` to auto-select)
224
+ - `--vision-detail low|high|auto` (default: `low`)
225
+ - `--reasoning-effort minimal|none|low|medium|high|xhigh` (default: `medium`)
226
+ - `--metadata-pages 2`
227
+ - `--vision-context-pages 0` — include N adjacent pages (before and after) as context in each figure-description call, to help with figures/captions that span a page break. 0 (default) is today's single-page behavior; more context pages means more tokens per call
228
+
229
+ Performance / misc:
230
+ - `--max-workers 4`
231
+ - `--rebuild` (reprocess everything; ignore `04_processed_pdfs.txt`)
232
+ - `--skip-text` (skip text extraction and resume only the vision steps; use after an interrupted run)
233
+ - `--list-models` — print the live, currently-available vision models for whichever provider(s) you have a key configured for, then exit (no PDF processing). Reflects Portkey's catalog too when that's active.
234
+ - `--version` / `-V`
235
+ - `--log-level DEBUG|INFO|WARNING|ERROR` (default: `ERROR`)
236
+
237
+ ## Local Ollama vision provider
238
+
239
+ `--vision-provider ollama` runs figure description and metadata extraction against a locally-pulled model served by [Ollama](https://ollama.com), instead of a cloud API — no API key needed, no per-call cost.
240
+
241
+ ```bash
242
+ visual-parser --input-dir /path/to/pdfs --vision-provider ollama
243
+ ```
244
+
245
+ Omit `--vision-model` and the largest locally-pulled, vision-capable model that fits in currently-free GPU VRAM is auto-selected (logged to the console/`05_pipeline.log` as it's picked). To use a specific model instead, pass it explicitly:
246
+
247
+ ```bash
248
+ visual-parser --input-dir /path/to/pdfs --vision-provider ollama --vision-model qwen2.5-vl:32b
249
+ ```
250
+
251
+ Auto-selection:
252
+ - Pull at least one vision-capable model first: `ollama pull llava` (or any other vision-capable tag).
253
+ - Vision capability is detected via Ollama's reported model capabilities, falling back to a name heuristic (`llava`, `vision`, `-vl`, `pixtral`, `moondream`, `minicpm-v`, `bakllava`) on older Ollama versions that don't report it.
254
+ - The selection budget is 85% of currently-free VRAM (not total VRAM) — this correctly leaves room for whatever else is already resident (e.g. Nougat's model during the same pipeline run) and for the model's own KV-cache/context overhead, which isn't included in its on-disk size.
255
+ - If no CUDA GPU is visible to the process at all — including the case of running in a Docker container without `--gpus all` while Ollama runs on the host with a real GPU — CPU-only Ollama is treated as a legitimate setup: auto-selection falls back to the smallest pulled vision-capable model instead of refusing, printing a clear warning that it's running CPU-only and will be slow. Pass an explicit `--vision-model` to pick a different one.
256
+ - If a real GPU *is* visible but no vision-capable model fits the free-VRAM budget, that case stays strict — auto-selection raises an error asking for an explicit `--vision-model`, rather than silently falling back to heavy CPU offload.
257
+
258
+ Reachability: checks `OLLAMA_BASE_URL` (if set) → `http://localhost:11434` → `http://host.docker.internal:11434` (the containerized case). On Docker Desktop (Windows/macOS) the container reaches a host-installed Ollama automatically; on native Linux Docker Engine, add `--add-host=host.docker.internal:host-gateway` to the `docker run` command, or set `OLLAMA_BASE_URL` directly.
259
+
260
+ ---
261
+
262
+ ## Citation
263
+
264
+ If you use RADIANT-LLM or the accompanying evaluation materials, please cite the journal article:
265
+
266
+ ```bibtex
267
+ @article{ndum2026retrieval,
268
+ title={A retrieval-augmented, domain-intelligent agentic framework for reliable decision support in safety-critical nuclear engineering},
269
+ author={Ndum, Zavier Ndum and Tao, Jian and Ford, John and Yim, Mansung and Liu, Yang},
270
+ journal={Reliability Engineering \& System Safety},
271
+ pages={113057},
272
+ year={2026},
273
+ publisher={Elsevier}
274
+ }
275
+ ```
276
+
277
+ Journal: *Reliability Engineering & System Safety* (2026), article 113057
278
+ Preprint: https://arxiv.org/abs/2604.22755
279
+
280
+ ---
281
+
282
+ ## License
283
+
284
+ Copyright 2026 Zavier N. Ndum
285
+
286
+ This project is licensed under the Apache License 2.0, the same license as its parent project, [RADIANT-LLM](https://github.com/SmartLabNuclear/RADIANT_LLM). See the [LICENSE](https://github.com/SmartLabNuclear/RADIANT_LLM/blob/main/LICENSE) file in the RADIANT_LLM repository, or the `LICENSE` file bundled with this package, for the full license text.
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "visual-parser"
7
- version = "2.1.0"
7
+ version = "2.2.0"
8
8
  description = "Standalone Visual-RAG PDF Parser - text extraction and Vision-LLM figure descriptions to JSONL"
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.10"
@@ -43,6 +43,7 @@ dependencies = [
43
43
  "langchain-text-splitters>=0.0.1", # RecursiveCharacterTextSplitter (split out from langchain in v0.2+)
44
44
  "openai>=1.14.0",
45
45
  "google-generativeai>=0.5.0",
46
+ "requests>=2.31.0", # local Ollama vision provider (model listing, capability detection)
46
47
  "python-dotenv>=1.0.0",
47
48
  "tqdm>=4.66.0",
48
49
  "nltk>=3.8", # required by NougatTokenizerFast
@@ -65,6 +66,13 @@ visual-parser = "visual_parser.cli_main:main"
65
66
  where = ["."]
66
67
  include = ["visual_parser*"]
67
68
 
69
+ [tool.pytest.ini_options]
70
+ testpaths = ["tests"]
71
+ addopts = "-m 'not integration'"
72
+ markers = [
73
+ "integration: requires a live OPENAI_API_KEY/GEMINI_API_KEY/Ollama instance and network access; excluded by default, run explicitly with 'pytest -m integration'",
74
+ ]
75
+
68
76
  [tool.ruff]
69
77
  line-length = 100
70
78
  target-version = "py310"