doc-redaction 2.2.2__tar.gz → 2.2.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (88) hide show
  1. {doc_redaction-2.2.2/doc_redaction.egg-info → doc_redaction-2.2.3}/PKG-INFO +1 -1
  2. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/app.py +14 -1
  3. {doc_redaction-2.2.2 → doc_redaction-2.2.3/doc_redaction.egg-info}/PKG-INFO +1 -1
  4. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/doc_redaction.egg-info/SOURCES.txt +1 -0
  5. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/pyproject.toml +1 -1
  6. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/config.py +27 -3
  7. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/custom_image_analyser_engine.py +31 -4
  8. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/file_redaction.py +22 -0
  9. doc_redaction-2.2.3/tools/preview_redaction_boxes.py +323 -0
  10. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/simplified_api.py +99 -0
  11. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/MANIFEST.in +0 -0
  12. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/README.md +0 -0
  13. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/README_PYPI.md +0 -0
  14. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/agent_routes.py +0 -0
  15. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/cli_redact.py +0 -0
  16. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/doc_redaction/__init__.py +0 -0
  17. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/doc_redaction/api.py +0 -0
  18. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/doc_redaction/cli_api.py +0 -0
  19. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/doc_redaction/cli_redact.py +0 -0
  20. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/doc_redaction/data_anonymise.py +0 -0
  21. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/doc_redaction/file_conversion.py +0 -0
  22. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/doc_redaction/file_redaction.py +0 -0
  23. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/doc_redaction/find_duplicate_pages.py +0 -0
  24. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/doc_redaction/find_duplicate_tabular.py +0 -0
  25. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/doc_redaction/gradio_app.py +0 -0
  26. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/doc_redaction/helper_functions.py +0 -0
  27. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/doc_redaction/install_deps.py +0 -0
  28. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/doc_redaction/lambda_entrypoint.py +0 -0
  29. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/doc_redaction/redaction_review.py +0 -0
  30. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/doc_redaction/summaries.py +0 -0
  31. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/doc_redaction.egg-info/dependency_links.txt +0 -0
  32. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/doc_redaction.egg-info/entry_points.txt +0 -0
  33. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/doc_redaction.egg-info/requires.txt +0 -0
  34. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/doc_redaction.egg-info/top_level.txt +0 -0
  35. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/favicon.png +0 -0
  36. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/intros/long_intro.txt +0 -0
  37. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/intros/short_intro.txt +0 -0
  38. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/intros/short_intro_responsible.txt +0 -0
  39. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/lambda_entrypoint.py +0 -0
  40. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/load_dynamo_logs.py +0 -0
  41. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/load_s3_logs.py +0 -0
  42. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/mcp_doc_redaction/__init__.py +0 -0
  43. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/mcp_doc_redaction/artifact_bundle.py +0 -0
  44. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/mcp_doc_redaction/gradio_transport.py +0 -0
  45. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/mcp_doc_redaction/schemas.py +0 -0
  46. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/mcp_doc_redaction/server.py +0 -0
  47. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/setup.cfg +0 -0
  48. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/test/test_agent_apply_review_redactions.py +0 -0
  49. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/test/test_annotation_color_parsing.py +0 -0
  50. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/test/test_cli_smoke.py +0 -0
  51. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/test/test_gradio_doc_redact_simple.py +0 -0
  52. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/test/test_gradio_summarise_simple.py +0 -0
  53. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/test/test_gradio_transport_sse.py +0 -0
  54. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/test/test_gradio_upload_staging.py +0 -0
  55. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/test/test_gui_only.py +0 -0
  56. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/test/test_mcp_doc_redaction_bundle.py +0 -0
  57. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/test/test_mcp_doc_redaction_extract_paths.py +0 -0
  58. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/test/test_package_api_smoke.py +0 -0
  59. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/test/test_placeholder_bbox_scaling.py +0 -0
  60. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/test/test_redaction_overlay_export.py +0 -0
  61. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/test/test_redaction_types.py +0 -0
  62. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/test/test_review_ocr_visualisation_export.py +0 -0
  63. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/__init__.py +0 -0
  64. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/apply_hf_zero_gpu_readme_frontmatter.py +0 -0
  65. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/auth.py +0 -0
  66. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/aws_functions.py +0 -0
  67. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/aws_textract.py +0 -0
  68. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/cli_usage_logger.py +0 -0
  69. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/custom_csvlogger.py +0 -0
  70. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/data_anonymise.py +0 -0
  71. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/file_conversion.py +0 -0
  72. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/find_duplicate_pages.py +0 -0
  73. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/find_duplicate_tabular.py +0 -0
  74. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/helper_functions.py +0 -0
  75. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/llm_entity_detection.py +0 -0
  76. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/llm_entity_detection_prompts.py +0 -0
  77. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/llm_funcs.py +0 -0
  78. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/load_spacy_model_custom_recognisers.py +0 -0
  79. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/presidio_analyzer_custom.py +0 -0
  80. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/quickstart.py +0 -0
  81. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/redaction_review.py +0 -0
  82. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/redaction_types.py +0 -0
  83. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/run_vlm.py +0 -0
  84. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/secure_path_utils.py +0 -0
  85. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/secure_regex_utils.py +0 -0
  86. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/summaries.py +0 -0
  87. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/textract_batch_call.py +0 -0
  88. {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/word_segmenter.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: doc_redaction
3
- Version: 2.2.2
3
+ Version: 2.2.3
4
4
  Summary: Redact PDF/image-based documents, Word, or CSV/XLSX files using a Gradio-based GUI interface
5
5
  Author-email: Sean Pedrick-Case <spedrickcase@lambeth.gov.uk>
6
6
  Maintainer-email: Sean Pedrick-Case <spedrickcase@lambeth.gov.uk>
@@ -9432,7 +9432,7 @@ If you are an LLM/agent calling this app programmatically, prefer the **short `g
9432
9432
  ],
9433
9433
  show_progress=True,
9434
9434
  show_progress_on=[summarisation_status],
9435
- api_name="undocumented",
9435
+ api_visibility="undocumented",
9436
9436
  ).success(
9437
9437
  fn=lambda: "summarisation",
9438
9438
  outputs=[task_textbox],
@@ -10105,6 +10105,7 @@ If you are an LLM/agent calling this app programmatically, prefer the **short `g
10105
10105
  from tools.simplified_api import (
10106
10106
  doc_redact_api,
10107
10107
  pdf_summarise_api,
10108
+ preview_boxes_api,
10108
10109
  review_apply_api,
10109
10110
  tabular_redact_api,
10110
10111
  )
@@ -10145,6 +10146,18 @@ If you are an LLM/agent calling this app programmatically, prefer the **short `g
10145
10146
  ),
10146
10147
  )
10147
10148
 
10149
+ gr.api(
10150
+ preview_boxes_api,
10151
+ api_name="preview_boxes",
10152
+ api_description=(
10153
+ "Render proposed redaction boxes from a *_review_file.csv onto the original PDF "
10154
+ "and return a ZIP of preview PNGs. Use this to verify box positions before calling "
10155
+ "/review_apply — no redaction is applied. "
10156
+ "Returns (zip_path, message). For agents with local files, calling "
10157
+ "tools.preview_redaction_boxes.preview_redaction_boxes() directly is faster."
10158
+ ),
10159
+ )
10160
+
10148
10161
  ###
10149
10162
  # APP RUN SETTINGS
10150
10163
  ###
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: doc_redaction
3
- Version: 2.2.2
3
+ Version: 2.2.3
4
4
  Summary: Redact PDF/image-based documents, Word, or CSV/XLSX files using a Gradio-based GUI interface
5
5
  Author-email: Sean Pedrick-Case <spedrickcase@lambeth.gov.uk>
6
6
  Maintainer-email: Sean Pedrick-Case <spedrickcase@lambeth.gov.uk>
@@ -73,6 +73,7 @@ tools/llm_entity_detection_prompts.py
73
73
  tools/llm_funcs.py
74
74
  tools/load_spacy_model_custom_recognisers.py
75
75
  tools/presidio_analyzer_custom.py
76
+ tools/preview_redaction_boxes.py
76
77
  tools/quickstart.py
77
78
  tools/redaction_review.py
78
79
  tools/redaction_types.py
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "doc_redaction"
7
- version = "2.2.2"
7
+ version = "2.2.3"
8
8
  description = "Redact PDF/image-based documents, Word, or CSV/XLSX files using a Gradio-based GUI interface"
9
9
  readme = "README_PYPI.md"
10
10
  authors = [
@@ -587,8 +587,8 @@ LOAD_REDACTION_ANNOTATIONS_FROM_PDF = convert_string_to_boolean(
587
587
  TESSERACT_FOLDER = get_or_create_env_var(
588
588
  "TESSERACT_FOLDER", ""
589
589
  ) # # If installing for Windows, install Tesseract 5.5.0 from here: https://github.com/UB-Mannheim/tesseract/wiki. Then this environment variable should point to the Tesseract folder e.g. tesseract/
590
- if TESSERACT_FOLDER:
591
- TESSERACT_FOLDER = ensure_folder_within_app_directory(TESSERACT_FOLDER)
590
+ if TESSERACT_FOLDER and not os.path.isabs(TESSERACT_FOLDER):
591
+ # TESSERACT_FOLDER = ensure_folder_within_app_directory(TESSERACT_FOLDER)
592
592
  add_folder_to_path(TESSERACT_FOLDER)
593
593
 
594
594
  TESSERACT_DATA_FOLDER = get_or_create_env_var(
@@ -601,7 +601,7 @@ if TESSERACT_DATA_FOLDER and not os.path.isabs(TESSERACT_DATA_FOLDER):
601
601
  POPPLER_FOLDER = get_or_create_env_var(
602
602
  "POPPLER_FOLDER", ""
603
603
  ) # If installing on Windows,install Poppler from here https://github.com/oschwartz10612/poppler-windows. This variable needs to point to the poppler bin folder e.g. poppler/poppler-24.02.0/Library/bin/
604
- if POPPLER_FOLDER:
604
+ if POPPLER_FOLDER and not os.path.isabs(POPPLER_FOLDER):
605
605
  POPPLER_FOLDER = ensure_folder_within_app_directory(POPPLER_FOLDER)
606
606
  add_folder_to_path(POPPLER_FOLDER)
607
607
 
@@ -2157,6 +2157,30 @@ MERGE_SMALL_REDACTIONS = convert_string_to_boolean(
2157
2157
  get_or_create_env_var("MERGE_SMALL_REDACTIONS", "False")
2158
2158
  )
2159
2159
 
2160
+ # When True (and MERGE_SMALL_REDACTIONS), only merge boxes that are on the same visual text line.
2161
+ # Implemented as a required minimum vertical overlap ratio between the two boxes.
2162
+ #
2163
+ # This prevents pathological merges where two separate redactions on different lines get unioned into
2164
+ # one huge rectangle (e.g., if the merge threshold is large relative to the coordinate scale).
2165
+ MERGE_SMALL_REDACTIONS_SAME_LINE_ONLY = convert_string_to_boolean(
2166
+ get_or_create_env_var("MERGE_SMALL_REDACTIONS_SAME_LINE_ONLY", "True")
2167
+ )
2168
+
2169
+ # Minimum vertical overlap ratio (0–1) required for two boxes to be considered "same line" for merging.
2170
+ # Ratio is computed as overlap_height / min(height1, height2).
2171
+ try:
2172
+ _merge_small_redactions_min_y_overlap_ratio = float(
2173
+ get_or_create_env_var(
2174
+ "MERGE_SMALL_REDACTIONS_MIN_Y_OVERLAP_RATIO", "0.6"
2175
+ ).strip()
2176
+ or "0.6"
2177
+ )
2178
+ except ValueError:
2179
+ _merge_small_redactions_min_y_overlap_ratio = 0.6
2180
+ MERGE_SMALL_REDACTIONS_MIN_Y_OVERLAP_RATIO = max(
2181
+ 0.0, min(1.0, _merge_small_redactions_min_y_overlap_ratio)
2182
+ )
2183
+
2160
2184
  # When True, use Polars for review DataFrame coordinate scaling and CSV write in apply_redactions_to_review_df_and_files (faster for large annotation sets).
2161
2185
  USE_POLARS_FOR_REVIEW = convert_string_to_boolean(
2162
2186
  get_or_create_env_var("USE_POLARS_FOR_REVIEW", "True")
@@ -7,6 +7,7 @@ import math
7
7
  import os
8
8
  import re
9
9
  import shutil
10
+ import sys
10
11
  import threading
11
12
  import time
12
13
  from concurrent.futures import ThreadPoolExecutor
@@ -158,6 +159,22 @@ def _guess_tessdata_dir_from_tesseract_exe(tesseract_exe: str | None) -> str | N
158
159
  return None
159
160
 
160
161
 
162
+ def _strip_wrapping_quotes(path: str | None) -> str:
163
+ """
164
+ Some .env setups accidentally include quotes in values, e.g.:
165
+ TESSDATA_PREFIX="tesseract/tessdata"
166
+ If those quotes end up in the environment variable value (including a stray trailing quote),
167
+ Tesseract will literally try to open paths like '"..."/eng.traineddata' and fail.
168
+ """
169
+ if not path:
170
+ return ""
171
+ s = str(path).strip()
172
+ if len(s) >= 2 and ((s[0] == s[-1] == '"') or (s[0] == s[-1] == "'")):
173
+ return s[1:-1].strip()
174
+ # Also handle common "one-sided" cases, e.g. a stray trailing quote.
175
+ return s.strip('"').strip("'").strip()
176
+
177
+
161
178
  def _resolve_tessdata_dir() -> str | None:
162
179
  """
163
180
  Return an absolute tessdata directory if we can find one.
@@ -166,15 +183,16 @@ def _resolve_tessdata_dir() -> str | None:
166
183
  2) tools.config.TESSERACT_DATA_FOLDER if it points to a valid tessdata dir
167
184
  3) Guess based on tesseract executable location (PATH / pytesseract config)
168
185
  """
169
- env_prefix = os.environ.get("TESSDATA_PREFIX", "")
186
+ env_prefix = _strip_wrapping_quotes(os.environ.get("TESSDATA_PREFIX", ""))
170
187
  if _is_probable_tessdata_dir(env_prefix):
171
188
  return os.path.abspath(env_prefix)
172
189
 
173
190
  try:
174
191
  from tools.config import TESSERACT_DATA_FOLDER
175
192
 
176
- if _is_probable_tessdata_dir(TESSERACT_DATA_FOLDER):
177
- return os.path.abspath(TESSERACT_DATA_FOLDER)
193
+ cfg_dir = _strip_wrapping_quotes(TESSERACT_DATA_FOLDER)
194
+ if _is_probable_tessdata_dir(cfg_dir):
195
+ return os.path.abspath(cfg_dir)
178
196
  except Exception:
179
197
  # config import is optional for library use
180
198
  pass
@@ -195,11 +213,20 @@ def _ensure_tessdata_available_in_env(existing_config: str | None) -> str | None
195
213
  if not tessdata_dir:
196
214
  return existing_config
197
215
 
198
- os.environ.setdefault("TESSDATA_PREFIX", tessdata_dir)
216
+ # Overwrite (not setdefault) so we can repair misquoted values already present in env.
217
+ os.environ["TESSDATA_PREFIX"] = tessdata_dir
199
218
 
200
219
  cfg = (existing_config or "").strip()
201
220
  if "--tessdata-dir" in cfg:
202
221
  return cfg
222
+
223
+ # On Windows, pytesseract parses config with shlex(posix=False), which can
224
+ # preserve quote characters in values and make Tesseract treat them as part
225
+ # of the path (e.g. '"C:\\...\\tessdata"/eng.traineddata'). Rely on
226
+ # TESSDATA_PREFIX there instead of injecting --tessdata-dir.
227
+ if sys.platform == "win32":
228
+ return cfg
229
+
203
230
  return (cfg + f' --tessdata-dir "{tessdata_dir}"').strip()
204
231
 
205
232
 
@@ -82,6 +82,8 @@ from tools.config import (
82
82
  MAX_WORKERS,
83
83
  MERGE_BOUNDING_BOXES,
84
84
  MERGE_SMALL_REDACTIONS,
85
+ MERGE_SMALL_REDACTIONS_MIN_Y_OVERLAP_RATIO,
86
+ MERGE_SMALL_REDACTIONS_SAME_LINE_ONLY,
85
87
  NO_REDACTION_PII_OPTION,
86
88
  OCR_FIRST_PASS_MAX_WORKERS,
87
89
  OUTPUT_FOLDER,
@@ -4987,7 +4989,27 @@ def _merge_adjacent_redaction_boxes(
4987
4989
  out["ymax"] = max(b1["ymax"], b2["ymax"])
4988
4990
  return out
4989
4991
 
4992
+ def _y_overlap_ratio(b1: dict, b2: dict) -> float:
4993
+ h1 = max(0.0, float(b1["ymax"]) - float(b1["ymin"]))
4994
+ h2 = max(0.0, float(b2["ymax"]) - float(b2["ymin"]))
4995
+ denom = min(h1, h2)
4996
+ if denom <= 0:
4997
+ return 0.0
4998
+ overlap = max(
4999
+ 0.0,
5000
+ min(float(b1["ymax"]), float(b2["ymax"]))
5001
+ - max(float(b1["ymin"]), float(b2["ymin"])),
5002
+ )
5003
+ return overlap / denom
5004
+
4990
5005
  def _overlap_or_close(b1: dict, b2: dict) -> bool:
5006
+ # Guard rail: do not merge across separate visual lines unless explicitly allowed.
5007
+ # This prevents a single giant union box when two redactions happen to be consecutive
5008
+ # in reading order but are located on very different y positions.
5009
+ if MERGE_SMALL_REDACTIONS_SAME_LINE_ONLY:
5010
+ if _y_overlap_ratio(b1, b2) < MERGE_SMALL_REDACTIONS_MIN_Y_OVERLAP_RATIO:
5011
+ return False
5012
+
4991
5013
  if b1["xmax"] + threshold < b2["xmin"] or b2["xmax"] + threshold < b1["xmin"]:
4992
5014
  return False
4993
5015
  if b1["ymax"] + threshold < b2["ymin"] or b2["ymax"] + threshold < b1["ymin"]:
@@ -0,0 +1,323 @@
1
+ """
2
+ preview_redaction_boxes.py
3
+ ==========================
4
+ Local-first coordinate preview tool for the Document Redaction app.
5
+
6
+ Purpose
7
+ -------
8
+ Render proposed redaction boxes from a ``*_review_file.csv`` onto the
9
+ **original** (un-redacted) PDF pages and save the result as PNG images.
10
+ Because this runs entirely locally with PyMuPDF + Pillow, iteration is
11
+ instantaneous — no server round-trip, no waiting for ``/review_apply``.
12
+
13
+ Primary use-case
14
+ ----------------
15
+ Called by agents or humans **between CSV edits and the API call to
16
+ ``/review_apply``**. Iterate until the preview looks right, *then*
17
+ send to the server. This avoids the expensive cycle of:
18
+
19
+ guess coordinates → apply → download → render → spot the miss → repeat
20
+
21
+ Typical agent workflow
22
+ ----------------------
23
+ 1. Edit ``*_review_file_edited.csv`` (remove FPs, add signatures, etc.).
24
+ 2. Call ``preview_redaction_boxes(pdf_path, csv_path, out_dir)`` locally.
25
+ 3. Inspect the saved PNGs.
26
+ 4. If anything is wrong, adjust the CSV and go to step 2.
27
+ 5. Only when satisfied, call ``/review_apply`` on the server.
28
+
29
+ API endpoint (server-side fallback)
30
+ ------------------------------------
31
+ When the agent does not have a local copy of the original PDF,
32
+ ``preview_boxes_api()`` exposes the same logic as a short ``gr.api``
33
+ endpoint registered as ``/preview_boxes`` in ``app.py``. The caller
34
+ uploads the original PDF and the edited review CSV; the server returns a
35
+ ZIP of preview PNGs.
36
+
37
+ CLI usage
38
+ ---------
39
+ python tools/preview_redaction_boxes.py original.pdf review_file.csv
40
+
41
+ # Optional flags:
42
+ python tools/preview_redaction_boxes.py original.pdf review_file.csv \\
43
+ --out-dir output/preview \\
44
+ --dpi 150 \\
45
+ --max-width 1280 \\
46
+ --grid # draw percentage-grid lines
47
+ --pages 1,3,5 # only render specific pages (1-indexed)
48
+ """
49
+
50
+ from __future__ import annotations
51
+
52
+ import argparse
53
+ import csv
54
+ import zipfile
55
+ from io import BytesIO
56
+ from pathlib import Path
57
+ from typing import Sequence
58
+
59
+ import pymupdf
60
+ from PIL import Image, ImageDraw, ImageFont
61
+
62
+ # ── Colour palette per label type ──────────────────────────────────────────
63
+ _LABEL_COLOURS: dict[str, str] = {
64
+ "PERSON": "#e74c3c", # red
65
+ "SIGNATURE": "#8e44ad", # purple
66
+ "LOCATION": "#2980b9", # blue
67
+ "EMAIL_ADDRESS": "#e67e22", # orange
68
+ "PHONE_NUMBER": "#27ae60", # green
69
+ "CUSTOM": "#f39c12", # amber
70
+ "DATE_TIME": "#16a085", # teal
71
+ "ORG": "#7f8c8d", # grey
72
+ }
73
+ _DEFAULT_COLOUR = "#c0392b"
74
+
75
+ # ── Grid style ─────────────────────────────────────────────────────────────
76
+ _GRID_COLOUR = "#cc0000"
77
+ _GRID_STEP = 5 # percentage intervals
78
+
79
+
80
+ def _label_colour(label: str) -> str:
81
+ for key, colour in _LABEL_COLOURS.items():
82
+ if key in label.upper():
83
+ return colour
84
+ return _DEFAULT_COLOUR
85
+
86
+
87
+ def _load_font(size: int = 11) -> ImageFont.ImageFont:
88
+ """Return a PIL font; fall back to the default if no TTF is available."""
89
+ for name in ("DejaVuSans.ttf", "Arial.ttf", "LiberationSans-Regular.ttf"):
90
+ try:
91
+ return ImageFont.truetype(name, size)
92
+ except OSError:
93
+ pass
94
+ return ImageFont.load_default()
95
+
96
+
97
+ def preview_redaction_boxes(
98
+ pdf_path: str | Path,
99
+ csv_path: str | Path,
100
+ out_dir: str | Path | None = None,
101
+ *,
102
+ dpi: int = 150,
103
+ max_width: int = 1280,
104
+ draw_grid: bool = True,
105
+ pages: Sequence[int] | None = None,
106
+ ) -> list[Path]:
107
+ """
108
+ Render proposed redaction boxes from *csv_path* onto the original PDF
109
+ at *pdf_path* and save one PNG per page to *out_dir*.
110
+
111
+ Parameters
112
+ ----------
113
+ pdf_path:
114
+ Path to the original (un-redacted) PDF.
115
+ csv_path:
116
+ Path to the ``*_review_file.csv`` (original or edited).
117
+ out_dir:
118
+ Directory for output PNGs. Defaults to a ``preview/`` subfolder
119
+ next to the CSV.
120
+ dpi:
121
+ Render resolution. 150 is a good balance of speed vs. detail.
122
+ Use 200-300 for detailed inspection of small text.
123
+ max_width:
124
+ Downscale rendered pages to at most this width (pixels) before
125
+ drawing boxes, to keep file sizes manageable.
126
+ draw_grid:
127
+ If True, overlay horizontal lines at every *_GRID_STEP* percent of
128
+ page height with percentage labels so you can read off normalized
129
+ y-coordinates by eye.
130
+ pages:
131
+ If given, only render these 1-indexed page numbers. Useful when
132
+ you are iterating on a single page and don't want to wait for the
133
+ whole document.
134
+
135
+ Returns
136
+ -------
137
+ list[Path]
138
+ Sorted list of saved PNG paths.
139
+ """
140
+ pdf_path = Path(pdf_path)
141
+ csv_path = Path(csv_path)
142
+
143
+ if out_dir is None:
144
+ out_dir = csv_path.parent / "preview"
145
+ out_dir = Path(out_dir)
146
+ out_dir.mkdir(parents=True, exist_ok=True)
147
+
148
+ # ── Load CSV ────────────────────────────────────────────────────────────
149
+ with csv_path.open(newline="", encoding="utf-8-sig") as fh:
150
+ rows = list(csv.DictReader(fh))
151
+
152
+ rows_by_page: dict[int, list[dict]] = {}
153
+ for row in rows:
154
+ try:
155
+ page_num = int(float(row.get("page", "0") or 0))
156
+ except ValueError:
157
+ continue
158
+ rows_by_page.setdefault(page_num, []).append(row)
159
+
160
+ # ── Render pages ────────────────────────────────────────────────────────
161
+ doc = pymupdf.open(str(pdf_path))
162
+ font = _load_font(11)
163
+ saved: list[Path] = []
164
+
165
+ page_range = range(1, doc.page_count + 1)
166
+ if pages:
167
+ page_range = [p for p in pages if 1 <= p <= doc.page_count]
168
+
169
+ for page_num in page_range:
170
+ pix = doc[page_num - 1].get_pixmap(dpi=dpi)
171
+ render_w, render_h = pix.width, pix.height
172
+
173
+ img = Image.frombytes("RGB", [render_w, render_h], pix.samples)
174
+
175
+ # ── Downscale if needed ──────────────────────────────────────────
176
+ if render_w > max_width:
177
+ scale = max_width / render_w
178
+ img = img.resize((max_width, int(render_h * scale)), Image.LANCZOS)
179
+ draw_w, draw_h = img.size
180
+
181
+ draw = ImageDraw.Draw(img, "RGBA")
182
+
183
+ # ── Percentage grid ──────────────────────────────────────────────
184
+ if draw_grid:
185
+ for pct in range(0, 101, _GRID_STEP):
186
+ y = int(pct / 100 * draw_h)
187
+ draw.line([(0, y), (draw_w, y)], fill=_GRID_COLOUR + "55", width=1)
188
+ draw.text((3, max(0, y - 11)), f"{pct}%", fill=_GRID_COLOUR, font=font)
189
+
190
+ # ── Redaction boxes ──────────────────────────────────────────────
191
+ for row in rows_by_page.get(page_num, []):
192
+ try:
193
+ x0 = float(row["xmin"]) * draw_w
194
+ y0 = float(row["ymin"]) * draw_h
195
+ x1 = float(row["xmax"]) * draw_w
196
+ y1 = float(row["ymax"]) * draw_h
197
+ except (KeyError, ValueError):
198
+ continue
199
+
200
+ label = row.get("label", "CUSTOM")
201
+ colour = _label_colour(label)
202
+ text_snippet = (row.get("text", "") or "")[:30]
203
+
204
+ # Semi-transparent fill
205
+ draw.rectangle(
206
+ [x0, y0, x1, y1], fill=colour + "33", outline=colour, width=2
207
+ )
208
+
209
+ # Label text
210
+ tag = f"{label}: {text_snippet}" if text_snippet else label
211
+ draw.text((x0 + 3, y0 + 2), tag, fill=colour, font=font)
212
+
213
+ # ── Legend (top-right corner) ────────────────────────────────────
214
+ legend_labels = sorted(
215
+ {r.get("label", "CUSTOM") for r in rows_by_page.get(page_num, [])}
216
+ )
217
+ lx, ly = draw_w - 200, 8
218
+ for lbl in legend_labels:
219
+ col = _label_colour(lbl)
220
+ draw.rectangle(
221
+ [lx, ly, lx + 14, ly + 14], fill=col + "cc", outline=col, width=1
222
+ )
223
+ draw.text((lx + 18, ly + 1), lbl, fill=col, font=font)
224
+ ly += 17
225
+
226
+ out_path = out_dir / f"page_{page_num:03d}_preview.png"
227
+ img.save(out_path)
228
+ saved.append(out_path)
229
+
230
+ doc.close()
231
+ print(f"Saved {len(saved)} preview image(s) to: {out_dir}")
232
+ return sorted(saved)
233
+
234
+
235
+ def preview_redaction_boxes_to_zip(
236
+ pdf_path: str | Path,
237
+ csv_path: str | Path,
238
+ *,
239
+ dpi: int = 150,
240
+ max_width: int = 1280,
241
+ draw_grid: bool = True,
242
+ pages: Sequence[int] | None = None,
243
+ ) -> bytes:
244
+ """
245
+ Same as ``preview_redaction_boxes`` but returns a ZIP of PNGs as bytes.
246
+
247
+ Used by the ``preview_boxes_api`` server endpoint so callers receive
248
+ all preview images in a single response without needing a shared
249
+ filesystem.
250
+ """
251
+ import tempfile
252
+
253
+ with tempfile.TemporaryDirectory() as tmp:
254
+ paths = preview_redaction_boxes(
255
+ pdf_path,
256
+ csv_path,
257
+ out_dir=tmp,
258
+ dpi=dpi,
259
+ max_width=max_width,
260
+ draw_grid=draw_grid,
261
+ pages=pages,
262
+ )
263
+ buf = BytesIO()
264
+ with zipfile.ZipFile(buf, "w", zipfile.ZIP_DEFLATED) as zf:
265
+ for p in paths:
266
+ zf.write(p, arcname=Path(p).name)
267
+ return buf.getvalue()
268
+
269
+
270
+ # ── CLI entry-point ─────────────────────────────────────────────────────────
271
+ def _main() -> None:
272
+ parser = argparse.ArgumentParser(
273
+ description="Render proposed redaction boxes from a review CSV onto the original PDF."
274
+ )
275
+ parser.add_argument("pdf", help="Path to the original (un-redacted) PDF")
276
+ parser.add_argument("csv", help="Path to the *_review_file.csv")
277
+ parser.add_argument(
278
+ "--out-dir",
279
+ default=None,
280
+ help="Output directory for PNGs (default: <csv-dir>/preview/)",
281
+ )
282
+ parser.add_argument(
283
+ "--dpi", type=int, default=150, help="Render DPI (default: 150)"
284
+ )
285
+ parser.add_argument(
286
+ "--max-width",
287
+ type=int,
288
+ default=1280,
289
+ help="Max image width in pixels (default: 1280)",
290
+ )
291
+ parser.add_argument(
292
+ "--grid",
293
+ action="store_true",
294
+ default=True,
295
+ help="Draw percentage grid (default: on)",
296
+ )
297
+ parser.add_argument(
298
+ "--no-grid", dest="grid", action="store_false", help="Disable percentage grid"
299
+ )
300
+ parser.add_argument(
301
+ "--pages",
302
+ default=None,
303
+ help="Comma-separated 1-indexed page numbers to render, e.g. 1,3,5 (default: all)",
304
+ )
305
+ args = parser.parse_args()
306
+
307
+ pages = None
308
+ if args.pages:
309
+ pages = [int(p.strip()) for p in args.pages.split(",")]
310
+
311
+ preview_redaction_boxes(
312
+ args.pdf,
313
+ args.csv,
314
+ out_dir=args.out_dir,
315
+ dpi=args.dpi,
316
+ max_width=args.max_width,
317
+ draw_grid=args.grid,
318
+ pages=pages,
319
+ )
320
+
321
+
322
+ if __name__ == "__main__":
323
+ _main()
@@ -1077,6 +1077,104 @@ def tabular_redact_api(
1077
1077
  )
1078
1078
 
1079
1079
 
1080
+ def preview_boxes_api(
1081
+ pdf_file: Any,
1082
+ review_csv_file: Any,
1083
+ dpi: int | None = 150,
1084
+ max_width: int | None = 1280,
1085
+ draw_grid: bool | None = True,
1086
+ pages: str | None = None,
1087
+ ) -> tuple[str, str]:
1088
+ """
1089
+ Render proposed redaction boxes from *review_csv_file* onto the
1090
+ original *pdf_file* and return a ZIP archive of preview PNGs.
1091
+
1092
+ Use this endpoint when you do **not** have a local copy of the
1093
+ original PDF and want to verify box positions without calling
1094
+ ``/review_apply``. For agents that already hold local files,
1095
+ calling ``tools.preview_redaction_boxes.preview_redaction_boxes``
1096
+ directly is faster (no upload/download round-trip).
1097
+
1098
+ Parameters
1099
+ ----------
1100
+ pdf_file:
1101
+ The original (un-redacted) PDF uploaded by the caller.
1102
+ review_csv_file:
1103
+ The ``*_review_file.csv`` (original or edited) uploaded by the
1104
+ caller.
1105
+ dpi:
1106
+ Render resolution (default 150).
1107
+ max_width:
1108
+ Maximum output image width in pixels (default 1280).
1109
+ draw_grid:
1110
+ If True (default), overlay percentage-grid lines so normalized
1111
+ y-coordinates can be read by eye.
1112
+ pages:
1113
+ Optional comma-separated 1-indexed page numbers, e.g. ``"1,3,5"``.
1114
+ If omitted, all pages are rendered.
1115
+
1116
+ Returns
1117
+ -------
1118
+ tuple[str, str]
1119
+ ``(zip_path, message)`` where *zip_path* is a server-side path to
1120
+ a ZIP file of preview PNGs retrievable via
1121
+ ``GET /gradio_api/file=<zip_path>``.
1122
+ """
1123
+ import tempfile
1124
+
1125
+ from tools.preview_redaction_boxes import preview_redaction_boxes
1126
+
1127
+ pdf_path = normalize_gradio_file_to_path(pdf_file)
1128
+ csv_path = normalize_gradio_file_to_path(review_csv_file)
1129
+
1130
+ if not pdf_path or not csv_path:
1131
+ return "", "Error: both pdf_file and review_csv_file are required."
1132
+
1133
+ pdf_path = stage_gradio_upload_if_ephemeral(pdf_path, INPUT_FOLDER)
1134
+ csv_path = stage_gradio_upload_if_ephemeral(csv_path, INPUT_FOLDER)
1135
+
1136
+ page_list: list[int] | None = None
1137
+ if pages:
1138
+ try:
1139
+ page_list = [int(p.strip()) for p in pages.split(",") if p.strip()]
1140
+ except ValueError:
1141
+ return (
1142
+ "",
1143
+ f"Error: 'pages' must be comma-separated integers, got: {pages!r}",
1144
+ )
1145
+
1146
+ with tempfile.TemporaryDirectory() as tmp:
1147
+ out_paths = preview_redaction_boxes(
1148
+ pdf_path,
1149
+ csv_path,
1150
+ out_dir=tmp,
1151
+ dpi=int(dpi or 150),
1152
+ max_width=int(max_width or 1280),
1153
+ draw_grid=bool(draw_grid),
1154
+ pages=page_list,
1155
+ )
1156
+
1157
+ if not out_paths:
1158
+ return (
1159
+ "",
1160
+ "No pages rendered — check that the CSV contains rows with valid page numbers.",
1161
+ )
1162
+
1163
+ out_base = Path(OUTPUT_FOLDER) / f"preview_{Path(pdf_path).stem}"
1164
+ out_base.mkdir(parents=True, exist_ok=True)
1165
+ zip_path = str(out_base / "preview_boxes.zip")
1166
+
1167
+ import zipfile
1168
+
1169
+ with zipfile.ZipFile(zip_path, "w", zipfile.ZIP_DEFLATED) as zf:
1170
+ for p in out_paths:
1171
+ zf.write(p, arcname=Path(p).name)
1172
+
1173
+ n = len(out_paths)
1174
+ msg = f"Preview complete: {n} page(s) rendered. Download the ZIP to inspect box positions."
1175
+ return zip_path, msg
1176
+
1177
+
1080
1178
  def doc_redact_api(
1081
1179
  document_file: Any,
1082
1180
  redact_entities: list[str] | None = None,
@@ -1117,4 +1215,5 @@ __all__ = [
1117
1215
  "run_apply_review_redactions",
1118
1216
  "summarise_document_from_upload_for_gradio_api",
1119
1217
  "pdf_summarise_api",
1218
+ "preview_boxes_api",
1120
1219
  ]
File without changes
File without changes
File without changes
File without changes