doc-redaction 2.2.2__tar.gz → 2.2.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {doc_redaction-2.2.2/doc_redaction.egg-info → doc_redaction-2.2.3}/PKG-INFO +1 -1
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/app.py +14 -1
- {doc_redaction-2.2.2 → doc_redaction-2.2.3/doc_redaction.egg-info}/PKG-INFO +1 -1
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/doc_redaction.egg-info/SOURCES.txt +1 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/pyproject.toml +1 -1
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/config.py +27 -3
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/custom_image_analyser_engine.py +31 -4
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/file_redaction.py +22 -0
- doc_redaction-2.2.3/tools/preview_redaction_boxes.py +323 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/simplified_api.py +99 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/MANIFEST.in +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/README.md +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/README_PYPI.md +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/agent_routes.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/cli_redact.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/doc_redaction/__init__.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/doc_redaction/api.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/doc_redaction/cli_api.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/doc_redaction/cli_redact.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/doc_redaction/data_anonymise.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/doc_redaction/file_conversion.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/doc_redaction/file_redaction.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/doc_redaction/find_duplicate_pages.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/doc_redaction/find_duplicate_tabular.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/doc_redaction/gradio_app.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/doc_redaction/helper_functions.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/doc_redaction/install_deps.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/doc_redaction/lambda_entrypoint.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/doc_redaction/redaction_review.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/doc_redaction/summaries.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/doc_redaction.egg-info/dependency_links.txt +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/doc_redaction.egg-info/entry_points.txt +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/doc_redaction.egg-info/requires.txt +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/doc_redaction.egg-info/top_level.txt +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/favicon.png +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/intros/long_intro.txt +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/intros/short_intro.txt +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/intros/short_intro_responsible.txt +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/lambda_entrypoint.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/load_dynamo_logs.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/load_s3_logs.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/mcp_doc_redaction/__init__.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/mcp_doc_redaction/artifact_bundle.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/mcp_doc_redaction/gradio_transport.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/mcp_doc_redaction/schemas.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/mcp_doc_redaction/server.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/setup.cfg +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/test/test_agent_apply_review_redactions.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/test/test_annotation_color_parsing.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/test/test_cli_smoke.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/test/test_gradio_doc_redact_simple.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/test/test_gradio_summarise_simple.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/test/test_gradio_transport_sse.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/test/test_gradio_upload_staging.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/test/test_gui_only.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/test/test_mcp_doc_redaction_bundle.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/test/test_mcp_doc_redaction_extract_paths.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/test/test_package_api_smoke.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/test/test_placeholder_bbox_scaling.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/test/test_redaction_overlay_export.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/test/test_redaction_types.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/test/test_review_ocr_visualisation_export.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/__init__.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/apply_hf_zero_gpu_readme_frontmatter.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/auth.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/aws_functions.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/aws_textract.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/cli_usage_logger.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/custom_csvlogger.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/data_anonymise.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/file_conversion.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/find_duplicate_pages.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/find_duplicate_tabular.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/helper_functions.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/llm_entity_detection.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/llm_entity_detection_prompts.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/llm_funcs.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/load_spacy_model_custom_recognisers.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/presidio_analyzer_custom.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/quickstart.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/redaction_review.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/redaction_types.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/run_vlm.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/secure_path_utils.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/secure_regex_utils.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/summaries.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/textract_batch_call.py +0 -0
- {doc_redaction-2.2.2 → doc_redaction-2.2.3}/tools/word_segmenter.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: doc_redaction
|
|
3
|
-
Version: 2.2.
|
|
3
|
+
Version: 2.2.3
|
|
4
4
|
Summary: Redact PDF/image-based documents, Word, or CSV/XLSX files using a Gradio-based GUI interface
|
|
5
5
|
Author-email: Sean Pedrick-Case <spedrickcase@lambeth.gov.uk>
|
|
6
6
|
Maintainer-email: Sean Pedrick-Case <spedrickcase@lambeth.gov.uk>
|
|
@@ -9432,7 +9432,7 @@ If you are an LLM/agent calling this app programmatically, prefer the **short `g
|
|
|
9432
9432
|
],
|
|
9433
9433
|
show_progress=True,
|
|
9434
9434
|
show_progress_on=[summarisation_status],
|
|
9435
|
-
|
|
9435
|
+
api_visibility="undocumented",
|
|
9436
9436
|
).success(
|
|
9437
9437
|
fn=lambda: "summarisation",
|
|
9438
9438
|
outputs=[task_textbox],
|
|
@@ -10105,6 +10105,7 @@ If you are an LLM/agent calling this app programmatically, prefer the **short `g
|
|
|
10105
10105
|
from tools.simplified_api import (
|
|
10106
10106
|
doc_redact_api,
|
|
10107
10107
|
pdf_summarise_api,
|
|
10108
|
+
preview_boxes_api,
|
|
10108
10109
|
review_apply_api,
|
|
10109
10110
|
tabular_redact_api,
|
|
10110
10111
|
)
|
|
@@ -10145,6 +10146,18 @@ If you are an LLM/agent calling this app programmatically, prefer the **short `g
|
|
|
10145
10146
|
),
|
|
10146
10147
|
)
|
|
10147
10148
|
|
|
10149
|
+
gr.api(
|
|
10150
|
+
preview_boxes_api,
|
|
10151
|
+
api_name="preview_boxes",
|
|
10152
|
+
api_description=(
|
|
10153
|
+
"Render proposed redaction boxes from a *_review_file.csv onto the original PDF "
|
|
10154
|
+
"and return a ZIP of preview PNGs. Use this to verify box positions before calling "
|
|
10155
|
+
"/review_apply — no redaction is applied. "
|
|
10156
|
+
"Returns (zip_path, message). For agents with local files, calling "
|
|
10157
|
+
"tools.preview_redaction_boxes.preview_redaction_boxes() directly is faster."
|
|
10158
|
+
),
|
|
10159
|
+
)
|
|
10160
|
+
|
|
10148
10161
|
###
|
|
10149
10162
|
# APP RUN SETTINGS
|
|
10150
10163
|
###
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: doc_redaction
|
|
3
|
-
Version: 2.2.
|
|
3
|
+
Version: 2.2.3
|
|
4
4
|
Summary: Redact PDF/image-based documents, Word, or CSV/XLSX files using a Gradio-based GUI interface
|
|
5
5
|
Author-email: Sean Pedrick-Case <spedrickcase@lambeth.gov.uk>
|
|
6
6
|
Maintainer-email: Sean Pedrick-Case <spedrickcase@lambeth.gov.uk>
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "doc_redaction"
|
|
7
|
-
version = "2.2.
|
|
7
|
+
version = "2.2.3"
|
|
8
8
|
description = "Redact PDF/image-based documents, Word, or CSV/XLSX files using a Gradio-based GUI interface"
|
|
9
9
|
readme = "README_PYPI.md"
|
|
10
10
|
authors = [
|
|
@@ -587,8 +587,8 @@ LOAD_REDACTION_ANNOTATIONS_FROM_PDF = convert_string_to_boolean(
|
|
|
587
587
|
TESSERACT_FOLDER = get_or_create_env_var(
|
|
588
588
|
"TESSERACT_FOLDER", ""
|
|
589
589
|
) # # If installing for Windows, install Tesseract 5.5.0 from here: https://github.com/UB-Mannheim/tesseract/wiki. Then this environment variable should point to the Tesseract folder e.g. tesseract/
|
|
590
|
-
if TESSERACT_FOLDER:
|
|
591
|
-
TESSERACT_FOLDER = ensure_folder_within_app_directory(TESSERACT_FOLDER)
|
|
590
|
+
if TESSERACT_FOLDER and not os.path.isabs(TESSERACT_FOLDER):
|
|
591
|
+
# TESSERACT_FOLDER = ensure_folder_within_app_directory(TESSERACT_FOLDER)
|
|
592
592
|
add_folder_to_path(TESSERACT_FOLDER)
|
|
593
593
|
|
|
594
594
|
TESSERACT_DATA_FOLDER = get_or_create_env_var(
|
|
@@ -601,7 +601,7 @@ if TESSERACT_DATA_FOLDER and not os.path.isabs(TESSERACT_DATA_FOLDER):
|
|
|
601
601
|
POPPLER_FOLDER = get_or_create_env_var(
|
|
602
602
|
"POPPLER_FOLDER", ""
|
|
603
603
|
) # If installing on Windows,install Poppler from here https://github.com/oschwartz10612/poppler-windows. This variable needs to point to the poppler bin folder e.g. poppler/poppler-24.02.0/Library/bin/
|
|
604
|
-
if POPPLER_FOLDER:
|
|
604
|
+
if POPPLER_FOLDER and not os.path.isabs(POPPLER_FOLDER):
|
|
605
605
|
POPPLER_FOLDER = ensure_folder_within_app_directory(POPPLER_FOLDER)
|
|
606
606
|
add_folder_to_path(POPPLER_FOLDER)
|
|
607
607
|
|
|
@@ -2157,6 +2157,30 @@ MERGE_SMALL_REDACTIONS = convert_string_to_boolean(
|
|
|
2157
2157
|
get_or_create_env_var("MERGE_SMALL_REDACTIONS", "False")
|
|
2158
2158
|
)
|
|
2159
2159
|
|
|
2160
|
+
# When True (and MERGE_SMALL_REDACTIONS), only merge boxes that are on the same visual text line.
|
|
2161
|
+
# Implemented as a required minimum vertical overlap ratio between the two boxes.
|
|
2162
|
+
#
|
|
2163
|
+
# This prevents pathological merges where two separate redactions on different lines get unioned into
|
|
2164
|
+
# one huge rectangle (e.g., if the merge threshold is large relative to the coordinate scale).
|
|
2165
|
+
MERGE_SMALL_REDACTIONS_SAME_LINE_ONLY = convert_string_to_boolean(
|
|
2166
|
+
get_or_create_env_var("MERGE_SMALL_REDACTIONS_SAME_LINE_ONLY", "True")
|
|
2167
|
+
)
|
|
2168
|
+
|
|
2169
|
+
# Minimum vertical overlap ratio (0–1) required for two boxes to be considered "same line" for merging.
|
|
2170
|
+
# Ratio is computed as overlap_height / min(height1, height2).
|
|
2171
|
+
try:
|
|
2172
|
+
_merge_small_redactions_min_y_overlap_ratio = float(
|
|
2173
|
+
get_or_create_env_var(
|
|
2174
|
+
"MERGE_SMALL_REDACTIONS_MIN_Y_OVERLAP_RATIO", "0.6"
|
|
2175
|
+
).strip()
|
|
2176
|
+
or "0.6"
|
|
2177
|
+
)
|
|
2178
|
+
except ValueError:
|
|
2179
|
+
_merge_small_redactions_min_y_overlap_ratio = 0.6
|
|
2180
|
+
MERGE_SMALL_REDACTIONS_MIN_Y_OVERLAP_RATIO = max(
|
|
2181
|
+
0.0, min(1.0, _merge_small_redactions_min_y_overlap_ratio)
|
|
2182
|
+
)
|
|
2183
|
+
|
|
2160
2184
|
# When True, use Polars for review DataFrame coordinate scaling and CSV write in apply_redactions_to_review_df_and_files (faster for large annotation sets).
|
|
2161
2185
|
USE_POLARS_FOR_REVIEW = convert_string_to_boolean(
|
|
2162
2186
|
get_or_create_env_var("USE_POLARS_FOR_REVIEW", "True")
|
|
@@ -7,6 +7,7 @@ import math
|
|
|
7
7
|
import os
|
|
8
8
|
import re
|
|
9
9
|
import shutil
|
|
10
|
+
import sys
|
|
10
11
|
import threading
|
|
11
12
|
import time
|
|
12
13
|
from concurrent.futures import ThreadPoolExecutor
|
|
@@ -158,6 +159,22 @@ def _guess_tessdata_dir_from_tesseract_exe(tesseract_exe: str | None) -> str | N
|
|
|
158
159
|
return None
|
|
159
160
|
|
|
160
161
|
|
|
162
|
+
def _strip_wrapping_quotes(path: str | None) -> str:
|
|
163
|
+
"""
|
|
164
|
+
Some .env setups accidentally include quotes in values, e.g.:
|
|
165
|
+
TESSDATA_PREFIX="tesseract/tessdata"
|
|
166
|
+
If those quotes end up in the environment variable value (including a stray trailing quote),
|
|
167
|
+
Tesseract will literally try to open paths like '"..."/eng.traineddata' and fail.
|
|
168
|
+
"""
|
|
169
|
+
if not path:
|
|
170
|
+
return ""
|
|
171
|
+
s = str(path).strip()
|
|
172
|
+
if len(s) >= 2 and ((s[0] == s[-1] == '"') or (s[0] == s[-1] == "'")):
|
|
173
|
+
return s[1:-1].strip()
|
|
174
|
+
# Also handle common "one-sided" cases, e.g. a stray trailing quote.
|
|
175
|
+
return s.strip('"').strip("'").strip()
|
|
176
|
+
|
|
177
|
+
|
|
161
178
|
def _resolve_tessdata_dir() -> str | None:
|
|
162
179
|
"""
|
|
163
180
|
Return an absolute tessdata directory if we can find one.
|
|
@@ -166,15 +183,16 @@ def _resolve_tessdata_dir() -> str | None:
|
|
|
166
183
|
2) tools.config.TESSERACT_DATA_FOLDER if it points to a valid tessdata dir
|
|
167
184
|
3) Guess based on tesseract executable location (PATH / pytesseract config)
|
|
168
185
|
"""
|
|
169
|
-
env_prefix = os.environ.get("TESSDATA_PREFIX", "")
|
|
186
|
+
env_prefix = _strip_wrapping_quotes(os.environ.get("TESSDATA_PREFIX", ""))
|
|
170
187
|
if _is_probable_tessdata_dir(env_prefix):
|
|
171
188
|
return os.path.abspath(env_prefix)
|
|
172
189
|
|
|
173
190
|
try:
|
|
174
191
|
from tools.config import TESSERACT_DATA_FOLDER
|
|
175
192
|
|
|
176
|
-
|
|
177
|
-
|
|
193
|
+
cfg_dir = _strip_wrapping_quotes(TESSERACT_DATA_FOLDER)
|
|
194
|
+
if _is_probable_tessdata_dir(cfg_dir):
|
|
195
|
+
return os.path.abspath(cfg_dir)
|
|
178
196
|
except Exception:
|
|
179
197
|
# config import is optional for library use
|
|
180
198
|
pass
|
|
@@ -195,11 +213,20 @@ def _ensure_tessdata_available_in_env(existing_config: str | None) -> str | None
|
|
|
195
213
|
if not tessdata_dir:
|
|
196
214
|
return existing_config
|
|
197
215
|
|
|
198
|
-
|
|
216
|
+
# Overwrite (not setdefault) so we can repair misquoted values already present in env.
|
|
217
|
+
os.environ["TESSDATA_PREFIX"] = tessdata_dir
|
|
199
218
|
|
|
200
219
|
cfg = (existing_config or "").strip()
|
|
201
220
|
if "--tessdata-dir" in cfg:
|
|
202
221
|
return cfg
|
|
222
|
+
|
|
223
|
+
# On Windows, pytesseract parses config with shlex(posix=False), which can
|
|
224
|
+
# preserve quote characters in values and make Tesseract treat them as part
|
|
225
|
+
# of the path (e.g. '"C:\\...\\tessdata"/eng.traineddata'). Rely on
|
|
226
|
+
# TESSDATA_PREFIX there instead of injecting --tessdata-dir.
|
|
227
|
+
if sys.platform == "win32":
|
|
228
|
+
return cfg
|
|
229
|
+
|
|
203
230
|
return (cfg + f' --tessdata-dir "{tessdata_dir}"').strip()
|
|
204
231
|
|
|
205
232
|
|
|
@@ -82,6 +82,8 @@ from tools.config import (
|
|
|
82
82
|
MAX_WORKERS,
|
|
83
83
|
MERGE_BOUNDING_BOXES,
|
|
84
84
|
MERGE_SMALL_REDACTIONS,
|
|
85
|
+
MERGE_SMALL_REDACTIONS_MIN_Y_OVERLAP_RATIO,
|
|
86
|
+
MERGE_SMALL_REDACTIONS_SAME_LINE_ONLY,
|
|
85
87
|
NO_REDACTION_PII_OPTION,
|
|
86
88
|
OCR_FIRST_PASS_MAX_WORKERS,
|
|
87
89
|
OUTPUT_FOLDER,
|
|
@@ -4987,7 +4989,27 @@ def _merge_adjacent_redaction_boxes(
|
|
|
4987
4989
|
out["ymax"] = max(b1["ymax"], b2["ymax"])
|
|
4988
4990
|
return out
|
|
4989
4991
|
|
|
4992
|
+
def _y_overlap_ratio(b1: dict, b2: dict) -> float:
|
|
4993
|
+
h1 = max(0.0, float(b1["ymax"]) - float(b1["ymin"]))
|
|
4994
|
+
h2 = max(0.0, float(b2["ymax"]) - float(b2["ymin"]))
|
|
4995
|
+
denom = min(h1, h2)
|
|
4996
|
+
if denom <= 0:
|
|
4997
|
+
return 0.0
|
|
4998
|
+
overlap = max(
|
|
4999
|
+
0.0,
|
|
5000
|
+
min(float(b1["ymax"]), float(b2["ymax"]))
|
|
5001
|
+
- max(float(b1["ymin"]), float(b2["ymin"])),
|
|
5002
|
+
)
|
|
5003
|
+
return overlap / denom
|
|
5004
|
+
|
|
4990
5005
|
def _overlap_or_close(b1: dict, b2: dict) -> bool:
|
|
5006
|
+
# Guard rail: do not merge across separate visual lines unless explicitly allowed.
|
|
5007
|
+
# This prevents a single giant union box when two redactions happen to be consecutive
|
|
5008
|
+
# in reading order but are located on very different y positions.
|
|
5009
|
+
if MERGE_SMALL_REDACTIONS_SAME_LINE_ONLY:
|
|
5010
|
+
if _y_overlap_ratio(b1, b2) < MERGE_SMALL_REDACTIONS_MIN_Y_OVERLAP_RATIO:
|
|
5011
|
+
return False
|
|
5012
|
+
|
|
4991
5013
|
if b1["xmax"] + threshold < b2["xmin"] or b2["xmax"] + threshold < b1["xmin"]:
|
|
4992
5014
|
return False
|
|
4993
5015
|
if b1["ymax"] + threshold < b2["ymin"] or b2["ymax"] + threshold < b1["ymin"]:
|
|
@@ -0,0 +1,323 @@
|
|
|
1
|
+
"""
|
|
2
|
+
preview_redaction_boxes.py
|
|
3
|
+
==========================
|
|
4
|
+
Local-first coordinate preview tool for the Document Redaction app.
|
|
5
|
+
|
|
6
|
+
Purpose
|
|
7
|
+
-------
|
|
8
|
+
Render proposed redaction boxes from a ``*_review_file.csv`` onto the
|
|
9
|
+
**original** (un-redacted) PDF pages and save the result as PNG images.
|
|
10
|
+
Because this runs entirely locally with PyMuPDF + Pillow, iteration is
|
|
11
|
+
instantaneous — no server round-trip, no waiting for ``/review_apply``.
|
|
12
|
+
|
|
13
|
+
Primary use-case
|
|
14
|
+
----------------
|
|
15
|
+
Called by agents or humans **between CSV edits and the API call to
|
|
16
|
+
``/review_apply``**. Iterate until the preview looks right, *then*
|
|
17
|
+
send to the server. This avoids the expensive cycle of:
|
|
18
|
+
|
|
19
|
+
guess coordinates → apply → download → render → spot the miss → repeat
|
|
20
|
+
|
|
21
|
+
Typical agent workflow
|
|
22
|
+
----------------------
|
|
23
|
+
1. Edit ``*_review_file_edited.csv`` (remove FPs, add signatures, etc.).
|
|
24
|
+
2. Call ``preview_redaction_boxes(pdf_path, csv_path, out_dir)`` locally.
|
|
25
|
+
3. Inspect the saved PNGs.
|
|
26
|
+
4. If anything is wrong, adjust the CSV and go to step 2.
|
|
27
|
+
5. Only when satisfied, call ``/review_apply`` on the server.
|
|
28
|
+
|
|
29
|
+
API endpoint (server-side fallback)
|
|
30
|
+
------------------------------------
|
|
31
|
+
When the agent does not have a local copy of the original PDF,
|
|
32
|
+
``preview_boxes_api()`` exposes the same logic as a short ``gr.api``
|
|
33
|
+
endpoint registered as ``/preview_boxes`` in ``app.py``. The caller
|
|
34
|
+
uploads the original PDF and the edited review CSV; the server returns a
|
|
35
|
+
ZIP of preview PNGs.
|
|
36
|
+
|
|
37
|
+
CLI usage
|
|
38
|
+
---------
|
|
39
|
+
python tools/preview_redaction_boxes.py original.pdf review_file.csv
|
|
40
|
+
|
|
41
|
+
# Optional flags:
|
|
42
|
+
python tools/preview_redaction_boxes.py original.pdf review_file.csv \\
|
|
43
|
+
--out-dir output/preview \\
|
|
44
|
+
--dpi 150 \\
|
|
45
|
+
--max-width 1280 \\
|
|
46
|
+
--grid # draw percentage-grid lines
|
|
47
|
+
--pages 1,3,5 # only render specific pages (1-indexed)
|
|
48
|
+
"""
|
|
49
|
+
|
|
50
|
+
from __future__ import annotations
|
|
51
|
+
|
|
52
|
+
import argparse
|
|
53
|
+
import csv
|
|
54
|
+
import zipfile
|
|
55
|
+
from io import BytesIO
|
|
56
|
+
from pathlib import Path
|
|
57
|
+
from typing import Sequence
|
|
58
|
+
|
|
59
|
+
import pymupdf
|
|
60
|
+
from PIL import Image, ImageDraw, ImageFont
|
|
61
|
+
|
|
62
|
+
# ── Colour palette per label type ──────────────────────────────────────────
|
|
63
|
+
_LABEL_COLOURS: dict[str, str] = {
|
|
64
|
+
"PERSON": "#e74c3c", # red
|
|
65
|
+
"SIGNATURE": "#8e44ad", # purple
|
|
66
|
+
"LOCATION": "#2980b9", # blue
|
|
67
|
+
"EMAIL_ADDRESS": "#e67e22", # orange
|
|
68
|
+
"PHONE_NUMBER": "#27ae60", # green
|
|
69
|
+
"CUSTOM": "#f39c12", # amber
|
|
70
|
+
"DATE_TIME": "#16a085", # teal
|
|
71
|
+
"ORG": "#7f8c8d", # grey
|
|
72
|
+
}
|
|
73
|
+
_DEFAULT_COLOUR = "#c0392b"
|
|
74
|
+
|
|
75
|
+
# ── Grid style ─────────────────────────────────────────────────────────────
|
|
76
|
+
_GRID_COLOUR = "#cc0000"
|
|
77
|
+
_GRID_STEP = 5 # percentage intervals
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def _label_colour(label: str) -> str:
|
|
81
|
+
for key, colour in _LABEL_COLOURS.items():
|
|
82
|
+
if key in label.upper():
|
|
83
|
+
return colour
|
|
84
|
+
return _DEFAULT_COLOUR
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def _load_font(size: int = 11) -> ImageFont.ImageFont:
|
|
88
|
+
"""Return a PIL font; fall back to the default if no TTF is available."""
|
|
89
|
+
for name in ("DejaVuSans.ttf", "Arial.ttf", "LiberationSans-Regular.ttf"):
|
|
90
|
+
try:
|
|
91
|
+
return ImageFont.truetype(name, size)
|
|
92
|
+
except OSError:
|
|
93
|
+
pass
|
|
94
|
+
return ImageFont.load_default()
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def preview_redaction_boxes(
|
|
98
|
+
pdf_path: str | Path,
|
|
99
|
+
csv_path: str | Path,
|
|
100
|
+
out_dir: str | Path | None = None,
|
|
101
|
+
*,
|
|
102
|
+
dpi: int = 150,
|
|
103
|
+
max_width: int = 1280,
|
|
104
|
+
draw_grid: bool = True,
|
|
105
|
+
pages: Sequence[int] | None = None,
|
|
106
|
+
) -> list[Path]:
|
|
107
|
+
"""
|
|
108
|
+
Render proposed redaction boxes from *csv_path* onto the original PDF
|
|
109
|
+
at *pdf_path* and save one PNG per page to *out_dir*.
|
|
110
|
+
|
|
111
|
+
Parameters
|
|
112
|
+
----------
|
|
113
|
+
pdf_path:
|
|
114
|
+
Path to the original (un-redacted) PDF.
|
|
115
|
+
csv_path:
|
|
116
|
+
Path to the ``*_review_file.csv`` (original or edited).
|
|
117
|
+
out_dir:
|
|
118
|
+
Directory for output PNGs. Defaults to a ``preview/`` subfolder
|
|
119
|
+
next to the CSV.
|
|
120
|
+
dpi:
|
|
121
|
+
Render resolution. 150 is a good balance of speed vs. detail.
|
|
122
|
+
Use 200-300 for detailed inspection of small text.
|
|
123
|
+
max_width:
|
|
124
|
+
Downscale rendered pages to at most this width (pixels) before
|
|
125
|
+
drawing boxes, to keep file sizes manageable.
|
|
126
|
+
draw_grid:
|
|
127
|
+
If True, overlay horizontal lines at every *_GRID_STEP* percent of
|
|
128
|
+
page height with percentage labels so you can read off normalized
|
|
129
|
+
y-coordinates by eye.
|
|
130
|
+
pages:
|
|
131
|
+
If given, only render these 1-indexed page numbers. Useful when
|
|
132
|
+
you are iterating on a single page and don't want to wait for the
|
|
133
|
+
whole document.
|
|
134
|
+
|
|
135
|
+
Returns
|
|
136
|
+
-------
|
|
137
|
+
list[Path]
|
|
138
|
+
Sorted list of saved PNG paths.
|
|
139
|
+
"""
|
|
140
|
+
pdf_path = Path(pdf_path)
|
|
141
|
+
csv_path = Path(csv_path)
|
|
142
|
+
|
|
143
|
+
if out_dir is None:
|
|
144
|
+
out_dir = csv_path.parent / "preview"
|
|
145
|
+
out_dir = Path(out_dir)
|
|
146
|
+
out_dir.mkdir(parents=True, exist_ok=True)
|
|
147
|
+
|
|
148
|
+
# ── Load CSV ────────────────────────────────────────────────────────────
|
|
149
|
+
with csv_path.open(newline="", encoding="utf-8-sig") as fh:
|
|
150
|
+
rows = list(csv.DictReader(fh))
|
|
151
|
+
|
|
152
|
+
rows_by_page: dict[int, list[dict]] = {}
|
|
153
|
+
for row in rows:
|
|
154
|
+
try:
|
|
155
|
+
page_num = int(float(row.get("page", "0") or 0))
|
|
156
|
+
except ValueError:
|
|
157
|
+
continue
|
|
158
|
+
rows_by_page.setdefault(page_num, []).append(row)
|
|
159
|
+
|
|
160
|
+
# ── Render pages ────────────────────────────────────────────────────────
|
|
161
|
+
doc = pymupdf.open(str(pdf_path))
|
|
162
|
+
font = _load_font(11)
|
|
163
|
+
saved: list[Path] = []
|
|
164
|
+
|
|
165
|
+
page_range = range(1, doc.page_count + 1)
|
|
166
|
+
if pages:
|
|
167
|
+
page_range = [p for p in pages if 1 <= p <= doc.page_count]
|
|
168
|
+
|
|
169
|
+
for page_num in page_range:
|
|
170
|
+
pix = doc[page_num - 1].get_pixmap(dpi=dpi)
|
|
171
|
+
render_w, render_h = pix.width, pix.height
|
|
172
|
+
|
|
173
|
+
img = Image.frombytes("RGB", [render_w, render_h], pix.samples)
|
|
174
|
+
|
|
175
|
+
# ── Downscale if needed ──────────────────────────────────────────
|
|
176
|
+
if render_w > max_width:
|
|
177
|
+
scale = max_width / render_w
|
|
178
|
+
img = img.resize((max_width, int(render_h * scale)), Image.LANCZOS)
|
|
179
|
+
draw_w, draw_h = img.size
|
|
180
|
+
|
|
181
|
+
draw = ImageDraw.Draw(img, "RGBA")
|
|
182
|
+
|
|
183
|
+
# ── Percentage grid ──────────────────────────────────────────────
|
|
184
|
+
if draw_grid:
|
|
185
|
+
for pct in range(0, 101, _GRID_STEP):
|
|
186
|
+
y = int(pct / 100 * draw_h)
|
|
187
|
+
draw.line([(0, y), (draw_w, y)], fill=_GRID_COLOUR + "55", width=1)
|
|
188
|
+
draw.text((3, max(0, y - 11)), f"{pct}%", fill=_GRID_COLOUR, font=font)
|
|
189
|
+
|
|
190
|
+
# ── Redaction boxes ──────────────────────────────────────────────
|
|
191
|
+
for row in rows_by_page.get(page_num, []):
|
|
192
|
+
try:
|
|
193
|
+
x0 = float(row["xmin"]) * draw_w
|
|
194
|
+
y0 = float(row["ymin"]) * draw_h
|
|
195
|
+
x1 = float(row["xmax"]) * draw_w
|
|
196
|
+
y1 = float(row["ymax"]) * draw_h
|
|
197
|
+
except (KeyError, ValueError):
|
|
198
|
+
continue
|
|
199
|
+
|
|
200
|
+
label = row.get("label", "CUSTOM")
|
|
201
|
+
colour = _label_colour(label)
|
|
202
|
+
text_snippet = (row.get("text", "") or "")[:30]
|
|
203
|
+
|
|
204
|
+
# Semi-transparent fill
|
|
205
|
+
draw.rectangle(
|
|
206
|
+
[x0, y0, x1, y1], fill=colour + "33", outline=colour, width=2
|
|
207
|
+
)
|
|
208
|
+
|
|
209
|
+
# Label text
|
|
210
|
+
tag = f"{label}: {text_snippet}" if text_snippet else label
|
|
211
|
+
draw.text((x0 + 3, y0 + 2), tag, fill=colour, font=font)
|
|
212
|
+
|
|
213
|
+
# ── Legend (top-right corner) ────────────────────────────────────
|
|
214
|
+
legend_labels = sorted(
|
|
215
|
+
{r.get("label", "CUSTOM") for r in rows_by_page.get(page_num, [])}
|
|
216
|
+
)
|
|
217
|
+
lx, ly = draw_w - 200, 8
|
|
218
|
+
for lbl in legend_labels:
|
|
219
|
+
col = _label_colour(lbl)
|
|
220
|
+
draw.rectangle(
|
|
221
|
+
[lx, ly, lx + 14, ly + 14], fill=col + "cc", outline=col, width=1
|
|
222
|
+
)
|
|
223
|
+
draw.text((lx + 18, ly + 1), lbl, fill=col, font=font)
|
|
224
|
+
ly += 17
|
|
225
|
+
|
|
226
|
+
out_path = out_dir / f"page_{page_num:03d}_preview.png"
|
|
227
|
+
img.save(out_path)
|
|
228
|
+
saved.append(out_path)
|
|
229
|
+
|
|
230
|
+
doc.close()
|
|
231
|
+
print(f"Saved {len(saved)} preview image(s) to: {out_dir}")
|
|
232
|
+
return sorted(saved)
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
def preview_redaction_boxes_to_zip(
|
|
236
|
+
pdf_path: str | Path,
|
|
237
|
+
csv_path: str | Path,
|
|
238
|
+
*,
|
|
239
|
+
dpi: int = 150,
|
|
240
|
+
max_width: int = 1280,
|
|
241
|
+
draw_grid: bool = True,
|
|
242
|
+
pages: Sequence[int] | None = None,
|
|
243
|
+
) -> bytes:
|
|
244
|
+
"""
|
|
245
|
+
Same as ``preview_redaction_boxes`` but returns a ZIP of PNGs as bytes.
|
|
246
|
+
|
|
247
|
+
Used by the ``preview_boxes_api`` server endpoint so callers receive
|
|
248
|
+
all preview images in a single response without needing a shared
|
|
249
|
+
filesystem.
|
|
250
|
+
"""
|
|
251
|
+
import tempfile
|
|
252
|
+
|
|
253
|
+
with tempfile.TemporaryDirectory() as tmp:
|
|
254
|
+
paths = preview_redaction_boxes(
|
|
255
|
+
pdf_path,
|
|
256
|
+
csv_path,
|
|
257
|
+
out_dir=tmp,
|
|
258
|
+
dpi=dpi,
|
|
259
|
+
max_width=max_width,
|
|
260
|
+
draw_grid=draw_grid,
|
|
261
|
+
pages=pages,
|
|
262
|
+
)
|
|
263
|
+
buf = BytesIO()
|
|
264
|
+
with zipfile.ZipFile(buf, "w", zipfile.ZIP_DEFLATED) as zf:
|
|
265
|
+
for p in paths:
|
|
266
|
+
zf.write(p, arcname=Path(p).name)
|
|
267
|
+
return buf.getvalue()
|
|
268
|
+
|
|
269
|
+
|
|
270
|
+
# ── CLI entry-point ─────────────────────────────────────────────────────────
|
|
271
|
+
def _main() -> None:
|
|
272
|
+
parser = argparse.ArgumentParser(
|
|
273
|
+
description="Render proposed redaction boxes from a review CSV onto the original PDF."
|
|
274
|
+
)
|
|
275
|
+
parser.add_argument("pdf", help="Path to the original (un-redacted) PDF")
|
|
276
|
+
parser.add_argument("csv", help="Path to the *_review_file.csv")
|
|
277
|
+
parser.add_argument(
|
|
278
|
+
"--out-dir",
|
|
279
|
+
default=None,
|
|
280
|
+
help="Output directory for PNGs (default: <csv-dir>/preview/)",
|
|
281
|
+
)
|
|
282
|
+
parser.add_argument(
|
|
283
|
+
"--dpi", type=int, default=150, help="Render DPI (default: 150)"
|
|
284
|
+
)
|
|
285
|
+
parser.add_argument(
|
|
286
|
+
"--max-width",
|
|
287
|
+
type=int,
|
|
288
|
+
default=1280,
|
|
289
|
+
help="Max image width in pixels (default: 1280)",
|
|
290
|
+
)
|
|
291
|
+
parser.add_argument(
|
|
292
|
+
"--grid",
|
|
293
|
+
action="store_true",
|
|
294
|
+
default=True,
|
|
295
|
+
help="Draw percentage grid (default: on)",
|
|
296
|
+
)
|
|
297
|
+
parser.add_argument(
|
|
298
|
+
"--no-grid", dest="grid", action="store_false", help="Disable percentage grid"
|
|
299
|
+
)
|
|
300
|
+
parser.add_argument(
|
|
301
|
+
"--pages",
|
|
302
|
+
default=None,
|
|
303
|
+
help="Comma-separated 1-indexed page numbers to render, e.g. 1,3,5 (default: all)",
|
|
304
|
+
)
|
|
305
|
+
args = parser.parse_args()
|
|
306
|
+
|
|
307
|
+
pages = None
|
|
308
|
+
if args.pages:
|
|
309
|
+
pages = [int(p.strip()) for p in args.pages.split(",")]
|
|
310
|
+
|
|
311
|
+
preview_redaction_boxes(
|
|
312
|
+
args.pdf,
|
|
313
|
+
args.csv,
|
|
314
|
+
out_dir=args.out_dir,
|
|
315
|
+
dpi=args.dpi,
|
|
316
|
+
max_width=args.max_width,
|
|
317
|
+
draw_grid=args.grid,
|
|
318
|
+
pages=pages,
|
|
319
|
+
)
|
|
320
|
+
|
|
321
|
+
|
|
322
|
+
if __name__ == "__main__":
|
|
323
|
+
_main()
|
|
@@ -1077,6 +1077,104 @@ def tabular_redact_api(
|
|
|
1077
1077
|
)
|
|
1078
1078
|
|
|
1079
1079
|
|
|
1080
|
+
def preview_boxes_api(
|
|
1081
|
+
pdf_file: Any,
|
|
1082
|
+
review_csv_file: Any,
|
|
1083
|
+
dpi: int | None = 150,
|
|
1084
|
+
max_width: int | None = 1280,
|
|
1085
|
+
draw_grid: bool | None = True,
|
|
1086
|
+
pages: str | None = None,
|
|
1087
|
+
) -> tuple[str, str]:
|
|
1088
|
+
"""
|
|
1089
|
+
Render proposed redaction boxes from *review_csv_file* onto the
|
|
1090
|
+
original *pdf_file* and return a ZIP archive of preview PNGs.
|
|
1091
|
+
|
|
1092
|
+
Use this endpoint when you do **not** have a local copy of the
|
|
1093
|
+
original PDF and want to verify box positions without calling
|
|
1094
|
+
``/review_apply``. For agents that already hold local files,
|
|
1095
|
+
calling ``tools.preview_redaction_boxes.preview_redaction_boxes``
|
|
1096
|
+
directly is faster (no upload/download round-trip).
|
|
1097
|
+
|
|
1098
|
+
Parameters
|
|
1099
|
+
----------
|
|
1100
|
+
pdf_file:
|
|
1101
|
+
The original (un-redacted) PDF uploaded by the caller.
|
|
1102
|
+
review_csv_file:
|
|
1103
|
+
The ``*_review_file.csv`` (original or edited) uploaded by the
|
|
1104
|
+
caller.
|
|
1105
|
+
dpi:
|
|
1106
|
+
Render resolution (default 150).
|
|
1107
|
+
max_width:
|
|
1108
|
+
Maximum output image width in pixels (default 1280).
|
|
1109
|
+
draw_grid:
|
|
1110
|
+
If True (default), overlay percentage-grid lines so normalized
|
|
1111
|
+
y-coordinates can be read by eye.
|
|
1112
|
+
pages:
|
|
1113
|
+
Optional comma-separated 1-indexed page numbers, e.g. ``"1,3,5"``.
|
|
1114
|
+
If omitted, all pages are rendered.
|
|
1115
|
+
|
|
1116
|
+
Returns
|
|
1117
|
+
-------
|
|
1118
|
+
tuple[str, str]
|
|
1119
|
+
``(zip_path, message)`` where *zip_path* is a server-side path to
|
|
1120
|
+
a ZIP file of preview PNGs retrievable via
|
|
1121
|
+
``GET /gradio_api/file=<zip_path>``.
|
|
1122
|
+
"""
|
|
1123
|
+
import tempfile
|
|
1124
|
+
|
|
1125
|
+
from tools.preview_redaction_boxes import preview_redaction_boxes
|
|
1126
|
+
|
|
1127
|
+
pdf_path = normalize_gradio_file_to_path(pdf_file)
|
|
1128
|
+
csv_path = normalize_gradio_file_to_path(review_csv_file)
|
|
1129
|
+
|
|
1130
|
+
if not pdf_path or not csv_path:
|
|
1131
|
+
return "", "Error: both pdf_file and review_csv_file are required."
|
|
1132
|
+
|
|
1133
|
+
pdf_path = stage_gradio_upload_if_ephemeral(pdf_path, INPUT_FOLDER)
|
|
1134
|
+
csv_path = stage_gradio_upload_if_ephemeral(csv_path, INPUT_FOLDER)
|
|
1135
|
+
|
|
1136
|
+
page_list: list[int] | None = None
|
|
1137
|
+
if pages:
|
|
1138
|
+
try:
|
|
1139
|
+
page_list = [int(p.strip()) for p in pages.split(",") if p.strip()]
|
|
1140
|
+
except ValueError:
|
|
1141
|
+
return (
|
|
1142
|
+
"",
|
|
1143
|
+
f"Error: 'pages' must be comma-separated integers, got: {pages!r}",
|
|
1144
|
+
)
|
|
1145
|
+
|
|
1146
|
+
with tempfile.TemporaryDirectory() as tmp:
|
|
1147
|
+
out_paths = preview_redaction_boxes(
|
|
1148
|
+
pdf_path,
|
|
1149
|
+
csv_path,
|
|
1150
|
+
out_dir=tmp,
|
|
1151
|
+
dpi=int(dpi or 150),
|
|
1152
|
+
max_width=int(max_width or 1280),
|
|
1153
|
+
draw_grid=bool(draw_grid),
|
|
1154
|
+
pages=page_list,
|
|
1155
|
+
)
|
|
1156
|
+
|
|
1157
|
+
if not out_paths:
|
|
1158
|
+
return (
|
|
1159
|
+
"",
|
|
1160
|
+
"No pages rendered — check that the CSV contains rows with valid page numbers.",
|
|
1161
|
+
)
|
|
1162
|
+
|
|
1163
|
+
out_base = Path(OUTPUT_FOLDER) / f"preview_{Path(pdf_path).stem}"
|
|
1164
|
+
out_base.mkdir(parents=True, exist_ok=True)
|
|
1165
|
+
zip_path = str(out_base / "preview_boxes.zip")
|
|
1166
|
+
|
|
1167
|
+
import zipfile
|
|
1168
|
+
|
|
1169
|
+
with zipfile.ZipFile(zip_path, "w", zipfile.ZIP_DEFLATED) as zf:
|
|
1170
|
+
for p in out_paths:
|
|
1171
|
+
zf.write(p, arcname=Path(p).name)
|
|
1172
|
+
|
|
1173
|
+
n = len(out_paths)
|
|
1174
|
+
msg = f"Preview complete: {n} page(s) rendered. Download the ZIP to inspect box positions."
|
|
1175
|
+
return zip_path, msg
|
|
1176
|
+
|
|
1177
|
+
|
|
1080
1178
|
def doc_redact_api(
|
|
1081
1179
|
document_file: Any,
|
|
1082
1180
|
redact_entities: list[str] | None = None,
|
|
@@ -1117,4 +1215,5 @@ __all__ = [
|
|
|
1117
1215
|
"run_apply_review_redactions",
|
|
1118
1216
|
"summarise_document_from_upload_for_gradio_api",
|
|
1119
1217
|
"pdf_summarise_api",
|
|
1218
|
+
"preview_boxes_api",
|
|
1120
1219
|
]
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|