visual-parser 1.0.2__tar.gz → 2.0.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (25) hide show
  1. {visual_parser-1.0.2 → visual_parser-2.0.1}/PKG-INFO +12 -11
  2. {visual_parser-1.0.2 → visual_parser-2.0.1}/README.md +7 -7
  3. {visual_parser-1.0.2 → visual_parser-2.0.1}/pyproject.toml +14 -11
  4. {visual_parser-1.0.2 → visual_parser-2.0.1}/visual_parser/__init__.py +1 -1
  5. {visual_parser-1.0.2 → visual_parser-2.0.1}/visual_parser/cli.py +12 -0
  6. {visual_parser-1.0.2 → visual_parser-2.0.1}/visual_parser/cli_main.py +12 -0
  7. {visual_parser-1.0.2 → visual_parser-2.0.1}/visual_parser/config.py +10 -0
  8. {visual_parser-1.0.2 → visual_parser-2.0.1}/visual_parser/figure_describer.py +65 -44
  9. visual_parser-2.0.1/visual_parser/pipeline.py +363 -0
  10. {visual_parser-1.0.2 → visual_parser-2.0.1}/visual_parser/text_extractor.py +90 -100
  11. {visual_parser-1.0.2 → visual_parser-2.0.1}/visual_parser.egg-info/PKG-INFO +12 -11
  12. visual_parser-1.0.2/visual_parser/pipeline.py +0 -255
  13. {visual_parser-1.0.2 → visual_parser-2.0.1}/setup.cfg +0 -0
  14. {visual_parser-1.0.2 → visual_parser-2.0.1}/visual_parser/__main__.py +0 -0
  15. {visual_parser-1.0.2 → visual_parser-2.0.1}/visual_parser/jsonl_writer.py +0 -0
  16. {visual_parser-1.0.2 → visual_parser-2.0.1}/visual_parser/metadata_extractor.py +0 -0
  17. {visual_parser-1.0.2 → visual_parser-2.0.1}/visual_parser/nougat_engine.py +0 -0
  18. {visual_parser-1.0.2 → visual_parser-2.0.1}/visual_parser/pdf_tracker.py +0 -0
  19. {visual_parser-1.0.2 → visual_parser-2.0.1}/visual_parser/prompts.py +0 -0
  20. {visual_parser-1.0.2 → visual_parser-2.0.1}/visual_parser/vision_llm.py +0 -0
  21. {visual_parser-1.0.2 → visual_parser-2.0.1}/visual_parser.egg-info/SOURCES.txt +0 -0
  22. {visual_parser-1.0.2 → visual_parser-2.0.1}/visual_parser.egg-info/dependency_links.txt +0 -0
  23. {visual_parser-1.0.2 → visual_parser-2.0.1}/visual_parser.egg-info/entry_points.txt +0 -0
  24. {visual_parser-1.0.2 → visual_parser-2.0.1}/visual_parser.egg-info/requires.txt +0 -0
  25. {visual_parser-1.0.2 → visual_parser-2.0.1}/visual_parser.egg-info/top_level.txt +0 -0
@@ -1,8 +1,9 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: visual-parser
3
- Version: 1.0.2
4
- Summary: Standalone Visual-RAG PDF Parser — text extraction + Vision-LLM figure descriptions → JSONL
5
- License: MIT
3
+ Version: 2.0.1
4
+ Summary: Standalone Visual-RAG PDF Parser - text extraction and Vision-LLM figure descriptions to JSONL
5
+ Author-email: "Zavier N. Ndum" <zavier.ndum@tamu.edu>
6
+ License: Apache-2.0
6
7
  Project-URL: Homepage, https://github.com/SmartLabNuclear/RADIANT_LLM
7
8
  Project-URL: Repository, https://github.com/SmartLabNuclear/RADIANT_LLM
8
9
  Project-URL: Docker Hub, https://hub.docker.com/r/zev94/radiant-llm
@@ -11,7 +12,7 @@ Classifier: Programming Language :: Python :: 3
11
12
  Classifier: Programming Language :: Python :: 3.10
12
13
  Classifier: Programming Language :: Python :: 3.11
13
14
  Classifier: Programming Language :: Python :: 3.12
14
- Classifier: License :: OSI Approved :: MIT License
15
+ Classifier: License :: OSI Approved :: Apache Software License
15
16
  Classifier: Operating System :: OS Independent
16
17
  Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
17
18
  Classifier: Topic :: Text Processing :: Markup
@@ -73,7 +74,7 @@ Prebuilt images are on **[zev94/radiant-llm](https://hub.docker.com/r/zev94/radi
73
74
 
74
75
  | Tag | Description |
75
76
  |-----|-------------|
76
- | `visual-parser-1.0` | Pinned release |
77
+ | `visual-parser-2.0` | Pinned release (v2.0) |
77
78
  | `visual-parser-latest` | Latest visual-parser build |
78
79
 
79
80
  ### 1) Install Docker
@@ -81,7 +82,7 @@ Prebuilt images are on **[zev94/radiant-llm](https://hub.docker.com/r/zev94/radi
81
82
 
82
83
  ### 2) Pull the image
83
84
  ```bash
84
- docker pull zev94/radiant-llm:visual-parser-1.0
85
+ docker pull zev94/radiant-llm:visual-parser-2.0
85
86
  ```
86
87
 
87
88
  ### 3) Run (input + output on the same mounted folder)
@@ -89,7 +90,7 @@ Windows PowerShell:
89
90
  ```powershell
90
91
  docker run --rm --env-file .env `
91
92
  -v "C:\path\to\pdfs:/data" `
92
- zev94/radiant-llm:visual-parser-1.0 `
93
+ zev94/radiant-llm:visual-parser-2.0 `
93
94
  --input-dir /data --output-dir /data
94
95
  ```
95
96
 
@@ -97,7 +98,7 @@ Linux / WSL:
97
98
  ```bash
98
99
  docker run --rm --env-file .env \
99
100
  -v "/path/to/pdfs:/data" \
100
- zev94/radiant-llm:visual-parser-1.0 \
101
+ zev94/radiant-llm:visual-parser-2.0 \
101
102
  --input-dir /data --output-dir /data
102
103
  ```
103
104
 
@@ -107,7 +108,7 @@ Windows PowerShell:
107
108
  docker run --rm --env-file .env `
108
109
  -v "C:\path\to\pdfs:/data" `
109
110
  -v "C:\path\to\out:/out" `
110
- zev94/radiant-llm:visual-parser-1.0 `
111
+ zev94/radiant-llm:visual-parser-2.0 `
111
112
  --input-dir /data --output-dir /out
112
113
  ```
113
114
 
@@ -124,7 +125,7 @@ Default vision model is **GPT-5.5** when using `--vision-provider gpt`. Override
124
125
 
125
126
  ```powershell
126
127
  docker run --rm --env-file .env -v "C:\path\to\pdfs:/data" `
127
- zev94/radiant-llm:visual-parser-1.0 `
128
+ zev94/radiant-llm:visual-parser-2.0 `
128
129
  --input-dir /data --output-dir /data --vision-model gpt-5.4
129
130
  ```
130
131
 
@@ -140,7 +141,7 @@ python visual-parser.py --input-dir "C:\path\to\pdfs"
140
141
  After pulling the image, run:
141
142
 
142
143
  ```bash
143
- docker run --rm zev94/radiant-llm:visual-parser-1.0 --help
144
+ docker run --rm zev94/radiant-llm:visual-parser-2.0 --help
144
145
  ```
145
146
 
146
147
  For copy-paste **Docker** examples (vision presets, text modes, workers, rebuild), see [`docker-usage-examples.md`](docker-usage-examples.md).
@@ -32,7 +32,7 @@ Prebuilt images are on **[zev94/radiant-llm](https://hub.docker.com/r/zev94/radi
32
32
 
33
33
  | Tag | Description |
34
34
  |-----|-------------|
35
- | `visual-parser-1.0` | Pinned release |
35
+ | `visual-parser-2.0` | Pinned release (v2.0) |
36
36
  | `visual-parser-latest` | Latest visual-parser build |
37
37
 
38
38
  ### 1) Install Docker
@@ -40,7 +40,7 @@ Prebuilt images are on **[zev94/radiant-llm](https://hub.docker.com/r/zev94/radi
40
40
 
41
41
  ### 2) Pull the image
42
42
  ```bash
43
- docker pull zev94/radiant-llm:visual-parser-1.0
43
+ docker pull zev94/radiant-llm:visual-parser-2.0
44
44
  ```
45
45
 
46
46
  ### 3) Run (input + output on the same mounted folder)
@@ -48,7 +48,7 @@ Windows PowerShell:
48
48
  ```powershell
49
49
  docker run --rm --env-file .env `
50
50
  -v "C:\path\to\pdfs:/data" `
51
- zev94/radiant-llm:visual-parser-1.0 `
51
+ zev94/radiant-llm:visual-parser-2.0 `
52
52
  --input-dir /data --output-dir /data
53
53
  ```
54
54
 
@@ -56,7 +56,7 @@ Linux / WSL:
56
56
  ```bash
57
57
  docker run --rm --env-file .env \
58
58
  -v "/path/to/pdfs:/data" \
59
- zev94/radiant-llm:visual-parser-1.0 \
59
+ zev94/radiant-llm:visual-parser-2.0 \
60
60
  --input-dir /data --output-dir /data
61
61
  ```
62
62
 
@@ -66,7 +66,7 @@ Windows PowerShell:
66
66
  docker run --rm --env-file .env `
67
67
  -v "C:\path\to\pdfs:/data" `
68
68
  -v "C:\path\to\out:/out" `
69
- zev94/radiant-llm:visual-parser-1.0 `
69
+ zev94/radiant-llm:visual-parser-2.0 `
70
70
  --input-dir /data --output-dir /out
71
71
  ```
72
72
 
@@ -83,7 +83,7 @@ Default vision model is **GPT-5.5** when using `--vision-provider gpt`. Override
83
83
 
84
84
  ```powershell
85
85
  docker run --rm --env-file .env -v "C:\path\to\pdfs:/data" `
86
- zev94/radiant-llm:visual-parser-1.0 `
86
+ zev94/radiant-llm:visual-parser-2.0 `
87
87
  --input-dir /data --output-dir /data --vision-model gpt-5.4
88
88
  ```
89
89
 
@@ -99,7 +99,7 @@ python visual-parser.py --input-dir "C:\path\to\pdfs"
99
99
  After pulling the image, run:
100
100
 
101
101
  ```bash
102
- docker run --rm zev94/radiant-llm:visual-parser-1.0 --help
102
+ docker run --rm zev94/radiant-llm:visual-parser-2.0 --help
103
103
  ```
104
104
 
105
105
  For copy-paste **Docker** examples (vision presets, text modes, workers, rebuild), see [`docker-usage-examples.md`](docker-usage-examples.md).
@@ -1,14 +1,17 @@
1
1
  [build-system]
2
- requires = ["setuptools>=61", "wheel"]
2
+ requires = ["setuptools>=61", "wheel"]
3
3
  build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
- name = "visual-parser"
7
- version = "1.0.2"
8
- description = "Standalone Visual-RAG PDF Parser — text extraction + Vision-LLM figure descriptions → JSONL"
9
- readme = "README.md"
6
+ name = "visual-parser"
7
+ version = "2.0.1"
8
+ description = "Standalone Visual-RAG PDF Parser - text extraction and Vision-LLM figure descriptions to JSONL"
9
+ readme = "README.md"
10
10
  requires-python = ">=3.10"
11
- license = { text = "MIT" }
11
+ license = { text = "Apache-2.0" }
12
+ authors = [
13
+ { name = "Zavier N. Ndum", email = "zavier.ndum@tamu.edu" },
14
+ ]
12
15
 
13
16
  keywords = [
14
17
  "pdf", "rag", "nougat", "vision-llm", "ocr",
@@ -20,7 +23,7 @@ classifiers = [
20
23
  "Programming Language :: Python :: 3.10",
21
24
  "Programming Language :: Python :: 3.11",
22
25
  "Programming Language :: Python :: 3.12",
23
- "License :: OSI Approved :: MIT License",
26
+ "License :: OSI Approved :: Apache Software License",
24
27
  "Operating System :: OS Independent",
25
28
  "Topic :: Scientific/Engineering :: Artificial Intelligence",
26
29
  "Topic :: Text Processing :: Markup",
@@ -45,13 +48,13 @@ dependencies = [
45
48
  ]
46
49
 
47
50
  [project.optional-dependencies]
48
- ocr = ["pytesseract==0.3.13"] # requires Tesseract binary installed separately
51
+ ocr = ["pytesseract==0.3.13"]
49
52
  dev = ["pytest", "ruff", "mypy"]
50
53
 
51
54
  [project.urls]
52
- Homepage = "https://github.com/SmartLabNuclear/RADIANT_LLM"
53
- Repository = "https://github.com/SmartLabNuclear/RADIANT_LLM"
54
- "Docker Hub" = "https://hub.docker.com/r/zev94/radiant-llm"
55
+ Homepage = "https://github.com/SmartLabNuclear/RADIANT_LLM"
56
+ Repository = "https://github.com/SmartLabNuclear/RADIANT_LLM"
57
+ "Docker Hub" = "https://hub.docker.com/r/zev94/radiant-llm"
55
58
 
56
59
  [project.scripts]
57
60
  visual-parser = "visual_parser.cli_main:main"
@@ -17,4 +17,4 @@ from visual_parser.config import ParserConfig
17
17
  from visual_parser.pipeline import run_pipeline
18
18
 
19
19
  __all__ = ["ParserConfig", "run_pipeline"]
20
- __version__ = "1.0.2"
20
+ __version__ = "2.0.0"
@@ -173,6 +173,17 @@ def _build_arg_parser() -> argparse.ArgumentParser:
173
173
  "Use after changing prompts, chunking strategy, or switching models."
174
174
  ),
175
175
  )
176
+ misc_group.add_argument(
177
+ "--skip-text",
178
+ action="store_true",
179
+ help=(
180
+ "Skip text extraction (Step 1) and resume only the vision steps "
181
+ "(figure descriptions + metadata). Use when chunking already completed "
182
+ "but the run was interrupted mid-vision (e.g. API credit exhaustion). "
183
+ "PDFs already present in 02_visuals_kb.jsonl / 03_metadata_kb.jsonl "
184
+ "are skipped automatically — no duplicates."
185
+ ),
186
+ )
176
187
  misc_group.add_argument(
177
188
  "--log-level",
178
189
  choices=["DEBUG", "INFO", "WARNING", "ERROR"],
@@ -216,6 +227,7 @@ def main(argv=None) -> int:
216
227
  metadata_pages = args.metadata_pages,
217
228
  max_workers = args.max_workers,
218
229
  rebuild = args.rebuild,
230
+ skip_text = args.skip_text,
219
231
  log_level = args.log_level,
220
232
  )
221
233
 
@@ -168,6 +168,17 @@ def _build_arg_parser() -> argparse.ArgumentParser:
168
168
  "Use after changing prompts, chunking strategy, or switching models."
169
169
  ),
170
170
  )
171
+ misc_group.add_argument(
172
+ "--skip-text",
173
+ action="store_true",
174
+ help=(
175
+ "Skip text extraction (Step 1) and resume only the vision steps "
176
+ "(figure descriptions + metadata). Use when chunking already completed "
177
+ "but the run was interrupted mid-vision (e.g. API credit exhaustion). "
178
+ "PDFs already present in 02_visuals_kb.jsonl / 03_metadata_kb.jsonl "
179
+ "are skipped automatically — no duplicates."
180
+ ),
181
+ )
171
182
  misc_group.add_argument(
172
183
  "--log-level",
173
184
  choices=["DEBUG", "INFO", "WARNING", "ERROR"],
@@ -206,6 +217,7 @@ def main(argv=None) -> int:
206
217
  metadata_pages=args.metadata_pages,
207
218
  max_workers=args.max_workers,
208
219
  rebuild=args.rebuild,
220
+ skip_text=args.skip_text,
209
221
  log_level=args.log_level,
210
222
  )
211
223
 
@@ -121,6 +121,15 @@ class ParserConfig:
121
121
  rebuild: bool = False
122
122
  """If True, reprocess all PDFs even if already recorded in 04_processed_pdfs.txt."""
123
123
 
124
+ skip_text: bool = False
125
+ """
126
+ If True, skip text extraction (Step 1) entirely.
127
+ Use when chunking already completed but the vision steps (figures / metadata)
128
+ failed mid-run (e.g. API credit exhaustion). All PDFs in input_dir are
129
+ re-queued for vision steps; PDFs already present in 02_visuals_kb.jsonl /
130
+ 03_metadata_kb.jsonl are skipped automatically.
131
+ """
132
+
124
133
  log_level: str = "ERROR"
125
134
 
126
135
  # -------------------------------------------------------------------------
@@ -145,6 +154,7 @@ class ParserConfig:
145
154
  metadata_pages = int(os.getenv("VISUAL_PARSER_METADATA_PAGES", "2")),
146
155
  max_workers = int(os.getenv("VISUAL_PARSER_MAX_WORKERS", "4")),
147
156
  rebuild = os.getenv("VISUAL_PARSER_REBUILD", "false").lower() == "true",
157
+ skip_text = os.getenv("VISUAL_PARSER_SKIP_TEXT", "false").lower() == "true",
148
158
  log_level = os.getenv("VISUAL_PARSER_LOG_LEVEL", "ERROR"),
149
159
  )
150
160
 
@@ -8,7 +8,7 @@ function in PDFAnalyser.py.
8
8
  Output
9
9
  ------
10
10
  One record per figure (or per page that contains at least one figure) is
11
- appended to ``02_visuals_kb.jsonl`` in *output_dir*:
11
+ appended to ``02_visuals_kb.jsonl`` in *output_dir*:
12
12
 
13
13
  {
14
14
  "source": "myreport.pdf",
@@ -84,16 +84,16 @@ def describe_figures_for_new_pdfs(
84
84
  vision_detail: str = "low",
85
85
  raster_dpi: int = 200,
86
86
  figure_prompt: str = FIGURE_PROMPT,
87
- reasoning_effort: Optional[str] = "medium",
88
- ) -> None:
87
+ reasoning_effort: Optional[str] = "medium",
88
+ ) -> None:
89
89
  """
90
90
  For each PDF in *new_pdf_paths*, rasterise every page at *raster_dpi* DPI,
91
91
  call the Vision LLM page-by-page, parse the figure descriptions, and
92
- append the results to ``02_visuals_kb.jsonl`` in *output_dir*.
92
+ append the results to ``02_visuals_kb.jsonl`` in *output_dir*.
93
93
 
94
94
  Args:
95
95
  new_pdf_paths: Full paths of PDFs to describe.
96
- output_dir: Directory where ``02_visuals_kb.jsonl`` is written.
96
+ output_dir: Directory where ``02_visuals_kb.jsonl`` is written.
97
97
  vision_provider: ``'gpt'`` or ``'gemini'``.
98
98
  vision_api_key: API key for the chosen provider.
99
99
  vision_model: Vision model name string.
@@ -125,7 +125,7 @@ def describe_figures_for_new_pdfs(
125
125
 
126
126
  if not page_images:
127
127
  logger.info("No pages to describe (all PDFs failed to rasterise).")
128
- return
128
+ return
129
129
 
130
130
  # -----------------------------------------------------------------------
131
131
  # Step 2 – Group page images by PDF name
@@ -135,16 +135,47 @@ def describe_figures_for_new_pdfs(
135
135
  pages_by_pdf[record["pdf"]].append(record)
136
136
 
137
137
  # -----------------------------------------------------------------------
138
- # Step 3 – Call Vision LLM once per page
138
+ # Step 2.5 – Build set of (source, page) pairs already on disk so a
139
+ # mid-run crash can be resumed at page granularity.
139
140
  # -----------------------------------------------------------------------
140
- descriptions_by_page: Dict[tuple, List[str]] = {}
141
+ figures_path = os.path.join(output_dir, "02_visuals_kb.jsonl")
142
+ done_pages: set = set()
143
+ if os.path.exists(figures_path):
144
+ with open(figures_path, encoding="utf-8") as _fh:
145
+ for _line in _fh:
146
+ _line = _line.strip()
147
+ if not _line:
148
+ continue
149
+ try:
150
+ _rec = json.loads(_line)
151
+ _src = _rec.get("source", "")
152
+ _pg = _rec.get("page")
153
+ if _src and _pg is not None:
154
+ done_pages.add((_src, _pg))
155
+ except Exception:
156
+ pass
157
+ if done_pages:
158
+ logger.info(
159
+ "Resuming: %d page(s) already in 02_visuals_kb.jsonl — will skip.",
160
+ len(done_pages),
161
+ )
162
+
163
+ # -----------------------------------------------------------------------
164
+ # Step 3 – Call Vision LLM per page; flush each page to disk immediately.
165
+ # Figure records are independent so per-page atomicity is safe.
166
+ # -----------------------------------------------------------------------
167
+ total_written = 0
141
168
 
142
169
  for pdf_name, image_records in pages_by_pdf.items():
143
- per_pdf_count = 0
170
+ pdf_written = 0
144
171
 
145
172
  for record in image_records:
146
- page_number = record["page"]
147
- image_bytes = record["bytes"]
173
+ page_number = record["page"]
174
+ image_bytes = record["bytes"]
175
+
176
+ if (pdf_name, page_number) in done_pages:
177
+ logger.debug("Skipping %s page %d (already in KB).", pdf_name, page_number)
178
+ continue
148
179
 
149
180
  try:
150
181
  raw_response = call_vision_llm(
@@ -159,8 +190,6 @@ def describe_figures_for_new_pdfs(
159
190
 
160
191
  captions = _parse_llm_response(raw_response, pdf_name, page_number)
161
192
 
162
- # Normalise: the model should return a list, but sometimes
163
- # returns a single dict for single-figure pages.
164
193
  if isinstance(captions, dict):
165
194
  captions = [captions]
166
195
 
@@ -171,15 +200,27 @@ def describe_figures_for_new_pdfs(
171
200
  )
172
201
  continue
173
202
 
174
- for caption in captions:
203
+ document_id = make_document_id(pdf_name)
204
+ page_rows: List[Dict] = []
205
+ for fig_idx, caption in enumerate(captions):
175
206
  if not isinstance(caption, dict):
176
207
  continue
177
208
  description = caption.get("description")
178
209
  if description is None:
179
210
  continue
180
- key = (pdf_name, page_number)
181
- descriptions_by_page.setdefault(key, []).append(description)
182
- per_pdf_count += 1
211
+ page_rows.append({
212
+ "source": pdf_name,
213
+ "page": page_number,
214
+ "document_id": document_id,
215
+ "figure_index": fig_idx,
216
+ "figure_id": f"{document_id}:p{page_number}:f{fig_idx}",
217
+ "description": description,
218
+ })
219
+
220
+ if page_rows:
221
+ append_to_jsonl(figures_path, page_rows)
222
+ pdf_written += len(page_rows)
223
+ total_written += len(page_rows)
183
224
 
184
225
  except Exception as exc:
185
226
  logger.error(
@@ -187,32 +228,12 @@ def describe_figures_for_new_pdfs(
187
228
  pdf_name, page_number, exc,
188
229
  )
189
230
 
190
- logger.info("[FIGURES] %s: %d figure(s) extracted.", pdf_name, per_pdf_count)
191
-
192
- total = sum(len(v) for v in descriptions_by_page.values())
193
- logger.info("Total figures captured: %d across %d PDF(s).", total,
194
- len({k[0] for k in descriptions_by_page}))
231
+ if pdf_written:
232
+ logger.info("[FIGURES] %s: %d figure(s) written.", pdf_name, pdf_written)
233
+ else:
234
+ logger.info("[FIGURES] %s: no new figures (all pages done or none detected).", pdf_name)
195
235
 
196
- # -----------------------------------------------------------------------
197
- # Step 4 – Write figure descriptions to 02_visuals_kb.jsonl
198
- # -----------------------------------------------------------------------
199
- figure_rows: List[Dict] = []
200
-
201
- for (pdf_name, page_number), descriptions in descriptions_by_page.items():
202
- document_id = make_document_id(pdf_name)
203
- for fig_idx, description in enumerate(descriptions):
204
- figure_rows.append({
205
- "source": pdf_name,
206
- "page": page_number,
207
- "document_id": document_id,
208
- "figure_index": fig_idx,
209
- "figure_id": f"{document_id}:p{page_number}:f{fig_idx}",
210
- "description": description,
211
- })
212
-
213
- if figure_rows:
214
- figures_path = os.path.join(output_dir, "02_visuals_kb.jsonl")
215
- append_to_jsonl(figures_path, figure_rows)
216
- print(f"[FIGURES] Wrote {len(figure_rows)} figure record(s) to 02_visuals_kb.jsonl.")
236
+ if total_written:
237
+ print(f"[FIGURES] Wrote {total_written} figure record(s) to 02_visuals_kb.jsonl.")
217
238
  else:
218
- logger.info("No figures detected — 02_visuals_kb.jsonl not updated.")
239
+ logger.info("No new figures written — all pages already processed or none detected.")