open-redactor 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- open_redactor/__init__.py +4 -0
- open_redactor/analytics.py +146 -0
- open_redactor/cli.py +274 -0
- open_redactor/codes.py +80 -0
- open_redactor/image_pipeline.py +117 -0
- open_redactor/local_gsam.py +215 -0
- open_redactor/mask_decode.py +90 -0
- open_redactor/masks.py +280 -0
- open_redactor/pii.py +167 -0
- open_redactor/pipeline.py +661 -0
- open_redactor/presets.py +54 -0
- open_redactor/replace.py +157 -0
- open_redactor/sam_client.py +468 -0
- open_redactor/webui.py +65 -0
- open_redactor-0.2.0.dist-info/METADATA +216 -0
- open_redactor-0.2.0.dist-info/RECORD +20 -0
- open_redactor-0.2.0.dist-info/WHEEL +5 -0
- open_redactor-0.2.0.dist-info/entry_points.txt +2 -0
- open_redactor-0.2.0.dist-info/licenses/LICENSE +201 -0
- open_redactor-0.2.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,146 @@
|
|
|
1
|
+
"""Analytics and severity for redaction runs and shadow audits.
|
|
2
|
+
|
|
3
|
+
Severity reflects how identifying an element is on its own.
|
|
4
|
+
Critical items can identify a person or account alone. High items
|
|
5
|
+
identify with a little context. Medium items help locate or profile.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import json
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
from typing import Dict, List
|
|
13
|
+
|
|
14
|
+
SEVERITY_WEIGHTS = {"critical": 10, "high": 6, "medium": 3, "low": 1}
|
|
15
|
+
|
|
16
|
+
SEVERITY_MAP = {
|
|
17
|
+
"passport": "critical",
|
|
18
|
+
"credit card": "critical",
|
|
19
|
+
"credit card number": "critical",
|
|
20
|
+
"ssn": "critical",
|
|
21
|
+
"driver license": "critical",
|
|
22
|
+
"id card": "critical",
|
|
23
|
+
"face": "high",
|
|
24
|
+
"person": "high",
|
|
25
|
+
"email": "high",
|
|
26
|
+
"phone number": "high",
|
|
27
|
+
"house number": "high",
|
|
28
|
+
"code": "high",
|
|
29
|
+
"license plate": "medium",
|
|
30
|
+
"screen": "medium",
|
|
31
|
+
"name badge": "medium",
|
|
32
|
+
"street sign": "medium",
|
|
33
|
+
"document": "medium",
|
|
34
|
+
"mailbox": "medium",
|
|
35
|
+
"lanyard": "medium",
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def severity_for(name: str) -> str:
|
|
40
|
+
low = name.lower()
|
|
41
|
+
for key, level in SEVERITY_MAP.items():
|
|
42
|
+
if key in low:
|
|
43
|
+
return level
|
|
44
|
+
return "medium"
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def kind_from_track_key(track_key: str) -> str:
|
|
48
|
+
"""Track keys look like phrase:id, pii:kind:n, or code:code:n."""
|
|
49
|
+
parts = track_key.split(":")
|
|
50
|
+
if parts[0] == "pii" and len(parts) >= 2:
|
|
51
|
+
return parts[1]
|
|
52
|
+
if parts[0] == "code":
|
|
53
|
+
return "code"
|
|
54
|
+
return parts[0]
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def build_summary(
|
|
58
|
+
tracks: Dict[str, Dict[int, object]],
|
|
59
|
+
total_frames: int,
|
|
60
|
+
audio_mode: str = "keep",
|
|
61
|
+
shadow: bool = False,
|
|
62
|
+
) -> dict:
|
|
63
|
+
elements: List[dict] = []
|
|
64
|
+
frames_with_hits: set[int] = set()
|
|
65
|
+
counts = {"critical": 0, "high": 0, "medium": 0, "low": 0}
|
|
66
|
+
for key in sorted(tracks.keys()):
|
|
67
|
+
frame_map = tracks[key]
|
|
68
|
+
if not frame_map:
|
|
69
|
+
continue
|
|
70
|
+
kind = kind_from_track_key(key)
|
|
71
|
+
level = severity_for(kind)
|
|
72
|
+
counts[level] += 1
|
|
73
|
+
detected = sorted(frame_map.keys())
|
|
74
|
+
for f in detected:
|
|
75
|
+
frames_with_hits.add(f)
|
|
76
|
+
elements.append(
|
|
77
|
+
{
|
|
78
|
+
"track": key,
|
|
79
|
+
"kind": kind,
|
|
80
|
+
"severity": level,
|
|
81
|
+
"severity_weight": SEVERITY_WEIGHTS[level],
|
|
82
|
+
"frames_detected": len(detected),
|
|
83
|
+
"first_frame": detected[0],
|
|
84
|
+
"last_frame": detected[-1],
|
|
85
|
+
"share_of_clip": round(len(detected) / total_frames, 3) if total_frames else 0.0,
|
|
86
|
+
}
|
|
87
|
+
)
|
|
88
|
+
elements.sort(key=lambda e: (-e["severity_weight"], -e["frames_detected"], e["track"]))
|
|
89
|
+
risk = sum(e["severity_weight"] * max(0.2, e["share_of_clip"]) for e in elements)
|
|
90
|
+
if audio_mode == "keep":
|
|
91
|
+
audio_note = "Audio kept. Original voices remain a separate identifying channel."
|
|
92
|
+
elif audio_mode == "mute":
|
|
93
|
+
audio_note = "Audio muted. No voice channel remains."
|
|
94
|
+
else:
|
|
95
|
+
audio_note = "Audio pitch shifted. Voices are disguised."
|
|
96
|
+
return {
|
|
97
|
+
"mode": "shadow audit, nothing was redacted or rendered" if shadow else "redaction",
|
|
98
|
+
"total_frames": total_frames,
|
|
99
|
+
"frames_with_sensitive_elements": len(frames_with_hits),
|
|
100
|
+
"share_of_frames_affected": round(len(frames_with_hits) / total_frames, 3) if total_frames else 0.0,
|
|
101
|
+
"elements_found": len(elements),
|
|
102
|
+
"severity_counts": counts,
|
|
103
|
+
"exposure_score": round(risk, 1),
|
|
104
|
+
"audio_mode": audio_mode,
|
|
105
|
+
"audio_note": audio_note,
|
|
106
|
+
"elements": elements,
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def summary_text(summary: dict) -> str:
|
|
111
|
+
lines = []
|
|
112
|
+
lines.append("Open Redactor analytics")
|
|
113
|
+
lines.append(f"Mode: {summary['mode']}")
|
|
114
|
+
lines.append(
|
|
115
|
+
f"Frames affected: {summary['frames_with_sensitive_elements']} of {summary['total_frames']} "
|
|
116
|
+
f"({summary['share_of_frames_affected']:.0%})"
|
|
117
|
+
)
|
|
118
|
+
c = summary["severity_counts"]
|
|
119
|
+
lines.append(
|
|
120
|
+
f"Elements: {summary['elements_found']} total, {c['critical']} critical, {c['high']} high, {c['medium']} medium"
|
|
121
|
+
)
|
|
122
|
+
lines.append(f"Exposure score before action: {summary['exposure_score']}")
|
|
123
|
+
lines.append(f"Audio: {summary['audio_note']}")
|
|
124
|
+
if summary["elements"]:
|
|
125
|
+
lines.append("")
|
|
126
|
+
lines.append("Most sensitive first:")
|
|
127
|
+
for e in summary["elements"]:
|
|
128
|
+
lines.append(
|
|
129
|
+
f"- [{e['severity'].upper()}] {e['track']}: {e['frames_detected']} frames, "
|
|
130
|
+
f"frames {e['first_frame']} to {e['last_frame']}"
|
|
131
|
+
)
|
|
132
|
+
else:
|
|
133
|
+
lines.append("No sensitive elements were detected.")
|
|
134
|
+
return "\n".join(lines) + "\n"
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
def write_summary_files(summary: dict, base_output: Path, shadow: bool = False) -> Dict[str, Path]:
|
|
138
|
+
json_path = base_output.with_suffix(".summary.json")
|
|
139
|
+
json_path.parent.mkdir(parents=True, exist_ok=True)
|
|
140
|
+
json_path.write_text(json.dumps(summary, indent=2))
|
|
141
|
+
paths = {"json": json_path}
|
|
142
|
+
if shadow:
|
|
143
|
+
audit_path = base_output.with_suffix(".audit.txt")
|
|
144
|
+
audit_path.write_text(summary_text(summary))
|
|
145
|
+
paths["audit"] = audit_path
|
|
146
|
+
return paths
|
open_redactor/cli.py
ADDED
|
@@ -0,0 +1,274 @@
|
|
|
1
|
+
"""CLI for Open Redactor."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
import sys
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
from typing import List, Optional
|
|
9
|
+
|
|
10
|
+
from .pipeline import run_pipeline
|
|
11
|
+
from .presets import PRESETS, apply_preset
|
|
12
|
+
from .sam_client import resolve_targets
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
16
|
+
p = argparse.ArgumentParser(prog="open-redactor", description="Make video share-safe with SAM 3.1 prompt-driven redaction")
|
|
17
|
+
p.add_argument("input", nargs="?", default=None, help="Input video file, MP4, MOV, MKV, WebM, AVI, or M4V, or a directory in batch mode")
|
|
18
|
+
p.add_argument("--output", default=None, help="Output path. Defaults to input name plus a redacted suffix")
|
|
19
|
+
p.add_argument("--target", action="append", default=None, help="Add one phrase. Can be repeated")
|
|
20
|
+
p.add_argument("--targets-default", action="store_true", help="Start from the default set of person, face, license plate, and screen")
|
|
21
|
+
p.add_argument("--add-target", action="append", default=None, help="Add one phrase on top of the defaults. Can be repeated")
|
|
22
|
+
p.add_argument("--mode", choices=["blur", "pixelate", "replace"], default="blur", help="Redaction mode: blur, pixelate, or replace with generated stand-ins")
|
|
23
|
+
p.add_argument("--strength", type=int, default=21, help="Blur radius or pixel block size")
|
|
24
|
+
p.add_argument("--mask-margin", type=int, default=10, help="Pad every mask by this many pixels")
|
|
25
|
+
p.add_argument("--carry-frames", type=int, default=4, help="Hold a track for this many frames after it vanishes")
|
|
26
|
+
p.add_argument("--contact-sheet", action="store_true", help="Write a PNG grid of redacted sample frames")
|
|
27
|
+
p.add_argument("--preview", action="store_true", help="Render only a 3 second sample plus contact sheet and coverage report")
|
|
28
|
+
p.add_argument("--report", action="store_true", help="Write a coverage report next to the output. On by default")
|
|
29
|
+
p.add_argument("--no-report", action="store_true", help="Skip the coverage report")
|
|
30
|
+
p.add_argument("--local", action="store_true", help="Run open weights on device and send nothing to the API")
|
|
31
|
+
p.add_argument("--api-key-env", default="MODEL_API_KEY", help="Name the environment variable that holds the API key")
|
|
32
|
+
p.add_argument("--preset", choices=sorted(PRESETS.keys()), default=None, help="Use a preset: family, street, screen-share, or documents")
|
|
33
|
+
p.add_argument("--no-cache", action="store_true", help="Do not reuse cached SAM results")
|
|
34
|
+
p.add_argument("--pii-text", action="store_true", help="Also OCR sampled frames for card numbers, SSNs, phones, and emails and blur them")
|
|
35
|
+
p.add_argument("--codes", action="store_true", help="Also detect and cover QR codes and barcodes in sampled frames")
|
|
36
|
+
p.add_argument("--sensitive", action="store_true", help="Turn on --pii-text and --codes together")
|
|
37
|
+
p.add_argument("--shadow", action="store_true", help="Audit only: report what would be removed with severity, render nothing")
|
|
38
|
+
p.add_argument("--audio", choices=["keep", "mute", "pitch"], default="keep", help="Audio redaction: keep the original track, mute it, or pitch shift voices")
|
|
39
|
+
p.add_argument("--pitch-factor", type=float, default=0.8, help="Pitch multiplier for --audio pitch. Below 1 deepens, above 1 raises")
|
|
40
|
+
p.add_argument("--provider", choices=["sam", "grounding-sam"], default=None, help="Model provider: SAM via API or hosted, or Grounding SAM locally")
|
|
41
|
+
p.add_argument("--backend", choices=["api", "hosted", "local"], default=None, help="Where SAM runs: Meta Model API, your own hosted endpoint, or local open weights")
|
|
42
|
+
p.add_argument("--endpoint", default=None, help="Responses API endpoint for the hosted backend, or set OPEN_REDACTOR_ENDPOINT")
|
|
43
|
+
p.add_argument("--ui", action="store_true", help="Launch the drag and drop local web page instead of processing a file")
|
|
44
|
+
p.add_argument("--batch", action="store_true", help="Treat the input as a directory and process each MP4 inside")
|
|
45
|
+
return p
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def default_output_path(input_path: Path) -> Path:
|
|
49
|
+
return unique_output_path(input_path.with_name(f"{input_path.stem}.redacted.mp4"))
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def unique_output_path(path: Path) -> Path:
|
|
53
|
+
"""Return a path that does not overwrite an existing file."""
|
|
54
|
+
if not path.exists():
|
|
55
|
+
return path
|
|
56
|
+
for i in range(1, 1000):
|
|
57
|
+
candidate = path.with_name(f"{path.stem}-{i}{path.suffix}")
|
|
58
|
+
if not candidate.exists():
|
|
59
|
+
return candidate
|
|
60
|
+
return path
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def process_one(
|
|
64
|
+
input_path: Path,
|
|
65
|
+
output_path: Optional[Path],
|
|
66
|
+
targets: List[str],
|
|
67
|
+
mode: str,
|
|
68
|
+
strength: int,
|
|
69
|
+
mask_margin: int,
|
|
70
|
+
carry_frames: int,
|
|
71
|
+
contact_sheet: bool,
|
|
72
|
+
local: bool,
|
|
73
|
+
api_key_env: str,
|
|
74
|
+
preview: bool = False,
|
|
75
|
+
report: bool = True,
|
|
76
|
+
use_cache: bool = True,
|
|
77
|
+
backend: str | None = None,
|
|
78
|
+
endpoint: str | None = None,
|
|
79
|
+
provider: str | None = None,
|
|
80
|
+
pii_text: bool = False,
|
|
81
|
+
codes: bool = False,
|
|
82
|
+
audio_mode: str = "keep",
|
|
83
|
+
pitch_factor: float = 0.8,
|
|
84
|
+
shadow: bool = False,
|
|
85
|
+
) -> int:
|
|
86
|
+
out = output_path if output_path else default_output_path(input_path)
|
|
87
|
+
if output_path is not None:
|
|
88
|
+
out = unique_output_path(output_path)
|
|
89
|
+
if out.resolve() == input_path.resolve():
|
|
90
|
+
out = unique_output_path(input_path.with_name(f'{input_path.stem}.redacted.mp4'))
|
|
91
|
+
elif out != output_path:
|
|
92
|
+
print(f'Output exists, writing to {out} instead so nothing is overwritten.')
|
|
93
|
+
try:
|
|
94
|
+
run_pipeline(
|
|
95
|
+
input_path=input_path,
|
|
96
|
+
output_path=out,
|
|
97
|
+
targets=targets,
|
|
98
|
+
mode=mode,
|
|
99
|
+
strength=strength,
|
|
100
|
+
mask_margin=mask_margin,
|
|
101
|
+
carry_frames=carry_frames,
|
|
102
|
+
contact_sheet=contact_sheet,
|
|
103
|
+
local=local,
|
|
104
|
+
api_key_env=api_key_env,
|
|
105
|
+
preview=preview,
|
|
106
|
+
report=report,
|
|
107
|
+
use_cache=use_cache,
|
|
108
|
+
backend=backend,
|
|
109
|
+
endpoint=endpoint,
|
|
110
|
+
provider=provider,
|
|
111
|
+
pii_text=pii_text,
|
|
112
|
+
codes=codes,
|
|
113
|
+
audio_mode=audio_mode,
|
|
114
|
+
pitch_factor=pitch_factor,
|
|
115
|
+
shadow=shadow,
|
|
116
|
+
)
|
|
117
|
+
return 0
|
|
118
|
+
except FileNotFoundError as exc:
|
|
119
|
+
print(f"Error: {exc}", file=sys.stderr)
|
|
120
|
+
return 2
|
|
121
|
+
except ValueError as exc:
|
|
122
|
+
print(f"Error: {exc}", file=sys.stderr)
|
|
123
|
+
return 2
|
|
124
|
+
except Exception as exc:
|
|
125
|
+
print(f"Error: {exc}", file=sys.stderr)
|
|
126
|
+
return 1
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def main(argv: Optional[List[str]] = None) -> int:
|
|
130
|
+
parser = build_parser()
|
|
131
|
+
args = parser.parse_args(argv)
|
|
132
|
+
|
|
133
|
+
if args.ui:
|
|
134
|
+
from .webui import serve_ui
|
|
135
|
+
print("Open http://127.0.0.1:8765 in your browser. Files stay on this machine.")
|
|
136
|
+
serve_ui()
|
|
137
|
+
return 0
|
|
138
|
+
|
|
139
|
+
if not args.input:
|
|
140
|
+
parser.error("input is required unless you pass --ui")
|
|
141
|
+
|
|
142
|
+
# Presets fill only values the user did not set explicitly
|
|
143
|
+
if args.preset:
|
|
144
|
+
preset = apply_preset(args.preset)
|
|
145
|
+
if not args.target and not args.add_target and not args.targets_default:
|
|
146
|
+
args.target = list(preset["targets"])
|
|
147
|
+
if args.mode == "blur" and preset["mode"] != "blur":
|
|
148
|
+
args.mode = preset["mode"]
|
|
149
|
+
# Strength, margin, and carry use CLI defaults, so only apply preset
|
|
150
|
+
# values when the user left them at the default sentinels
|
|
151
|
+
if args.strength == 21:
|
|
152
|
+
args.strength = int(preset["strength"])
|
|
153
|
+
if args.mask_margin == 10:
|
|
154
|
+
args.mask_margin = int(preset["mask_margin"])
|
|
155
|
+
if args.carry_frames == 4:
|
|
156
|
+
args.carry_frames = int(preset["carry_frames"])
|
|
157
|
+
print(f"Preset: {args.preset} - {preset['description']}")
|
|
158
|
+
|
|
159
|
+
targets = resolve_targets(
|
|
160
|
+
targets=args.target,
|
|
161
|
+
add_targets=args.add_target,
|
|
162
|
+
use_defaults=args.targets_default,
|
|
163
|
+
)
|
|
164
|
+
|
|
165
|
+
input_path = Path(args.input)
|
|
166
|
+
|
|
167
|
+
if args.batch:
|
|
168
|
+
if not input_path.is_dir():
|
|
169
|
+
print(f"Error: batch mode needs a directory: {input_path}", file=sys.stderr)
|
|
170
|
+
return 2
|
|
171
|
+
from .pipeline import SUPPORTED_INPUT_SUFFIXES
|
|
172
|
+
mp4s = sorted(f for f in input_path.iterdir() if f.is_file() and f.suffix.lower() in SUPPORTED_INPUT_SUFFIXES)
|
|
173
|
+
if not mp4s:
|
|
174
|
+
print(f"No supported video files found in {input_path}")
|
|
175
|
+
return 0
|
|
176
|
+
exit_code = 0
|
|
177
|
+
for mp4 in mp4s:
|
|
178
|
+
out = None
|
|
179
|
+
if args.output:
|
|
180
|
+
# In batch mode --output is treated as an output directory
|
|
181
|
+
out_dir = Path(args.output)
|
|
182
|
+
out = out_dir / default_output_path(mp4).name
|
|
183
|
+
code = process_one(
|
|
184
|
+
input_path=mp4,
|
|
185
|
+
output_path=out,
|
|
186
|
+
targets=targets,
|
|
187
|
+
mode=args.mode,
|
|
188
|
+
strength=args.strength,
|
|
189
|
+
mask_margin=args.mask_margin,
|
|
190
|
+
carry_frames=args.carry_frames,
|
|
191
|
+
contact_sheet=args.contact_sheet,
|
|
192
|
+
local=args.local,
|
|
193
|
+
api_key_env=args.api_key_env,
|
|
194
|
+
preview=args.preview,
|
|
195
|
+
report=not args.no_report,
|
|
196
|
+
use_cache=not args.no_cache,
|
|
197
|
+
backend=args.backend,
|
|
198
|
+
endpoint=args.endpoint,
|
|
199
|
+
provider=args.provider,
|
|
200
|
+
pii_text=args.pii_text or args.sensitive,
|
|
201
|
+
codes=args.codes or args.sensitive,
|
|
202
|
+
audio_mode=args.audio,
|
|
203
|
+
pitch_factor=args.pitch_factor,
|
|
204
|
+
shadow=args.shadow,
|
|
205
|
+
)
|
|
206
|
+
if code != 0:
|
|
207
|
+
exit_code = code
|
|
208
|
+
return exit_code
|
|
209
|
+
|
|
210
|
+
# Photo mode: image inputs go to the image pipeline, output keeps its format
|
|
211
|
+
if input_path.suffix.lower() in {".jpg", ".jpeg", ".png", ".webp", ".bmp"}:
|
|
212
|
+
from .image_pipeline import run_image_pipeline
|
|
213
|
+
out = Path(args.output) if args.output else unique_output_path(
|
|
214
|
+
input_path.with_name(f"{input_path.stem}.redacted{input_path.suffix}")
|
|
215
|
+
)
|
|
216
|
+
if args.output:
|
|
217
|
+
out = unique_output_path(out)
|
|
218
|
+
try:
|
|
219
|
+
run_image_pipeline(
|
|
220
|
+
input_path=input_path,
|
|
221
|
+
output_path=out,
|
|
222
|
+
targets=targets,
|
|
223
|
+
mode=args.mode,
|
|
224
|
+
strength=args.strength,
|
|
225
|
+
mask_margin=args.mask_margin,
|
|
226
|
+
local=args.local,
|
|
227
|
+
api_key_env=args.api_key_env,
|
|
228
|
+
backend=args.backend,
|
|
229
|
+
endpoint=args.endpoint,
|
|
230
|
+
provider=args.provider,
|
|
231
|
+
pii_text=args.pii_text or args.sensitive,
|
|
232
|
+
codes=args.codes or args.sensitive,
|
|
233
|
+
)
|
|
234
|
+
return 0
|
|
235
|
+
except Exception as exc:
|
|
236
|
+
print(f"Error: {exc}", file=sys.stderr)
|
|
237
|
+
return 1
|
|
238
|
+
|
|
239
|
+
# Single file mode
|
|
240
|
+
if not input_path.exists():
|
|
241
|
+
print(f"Error: Input not found: {input_path}", file=sys.stderr)
|
|
242
|
+
return 2
|
|
243
|
+
if input_path.is_dir():
|
|
244
|
+
print("Error: Input is a directory. Use --batch to process a folder.", file=sys.stderr)
|
|
245
|
+
return 2
|
|
246
|
+
|
|
247
|
+
output_path = Path(args.output) if args.output else None
|
|
248
|
+
return process_one(
|
|
249
|
+
input_path=input_path,
|
|
250
|
+
output_path=output_path,
|
|
251
|
+
targets=targets,
|
|
252
|
+
mode=args.mode,
|
|
253
|
+
strength=args.strength,
|
|
254
|
+
mask_margin=args.mask_margin,
|
|
255
|
+
carry_frames=args.carry_frames,
|
|
256
|
+
contact_sheet=args.contact_sheet,
|
|
257
|
+
local=args.local,
|
|
258
|
+
api_key_env=args.api_key_env,
|
|
259
|
+
preview=args.preview,
|
|
260
|
+
report=not args.no_report,
|
|
261
|
+
use_cache=not args.no_cache,
|
|
262
|
+
backend=args.backend,
|
|
263
|
+
endpoint=args.endpoint,
|
|
264
|
+
provider=args.provider,
|
|
265
|
+
pii_text=args.pii_text or args.sensitive,
|
|
266
|
+
codes=args.codes or args.sensitive,
|
|
267
|
+
audio_mode=args.audio,
|
|
268
|
+
pitch_factor=args.pitch_factor,
|
|
269
|
+
shadow=args.shadow,
|
|
270
|
+
)
|
|
271
|
+
|
|
272
|
+
|
|
273
|
+
if __name__ == "__main__":
|
|
274
|
+
raise SystemExit(main())
|
open_redactor/codes.py
ADDED
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
"""QR code and barcode scanning for sampled frames.
|
|
2
|
+
|
|
3
|
+
Uses OpenCV detectors that ship with opencv-python. QR codes carry URLs,
|
|
4
|
+
Wi-Fi passwords, payment links, and contact cards, so any readable code
|
|
5
|
+
in frame is treated as sensitive and covered. Boxes are held between
|
|
6
|
+
samples the same way text PII boxes are.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from typing import Dict, List, Tuple
|
|
12
|
+
|
|
13
|
+
import numpy as np
|
|
14
|
+
|
|
15
|
+
Box = Tuple[int, int, int, int]
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def boxes_from_points(points) -> List[Box]:
|
|
19
|
+
"""Turn detector corner points into axis aligned boxes."""
|
|
20
|
+
boxes: List[Box] = []
|
|
21
|
+
if points is None:
|
|
22
|
+
return boxes
|
|
23
|
+
arr = np.asarray(points, dtype=float)
|
|
24
|
+
if arr.size == 0:
|
|
25
|
+
return boxes
|
|
26
|
+
# Shapes seen in the wild: (n, 4, 2), (n, 1, 4, 2), or (4, 2) for one code
|
|
27
|
+
arr = arr.reshape(-1, arr.shape[-2], arr.shape[-1]) if arr.ndim >= 3 else arr.reshape(1, -1, 2)
|
|
28
|
+
for poly in arr:
|
|
29
|
+
xs, ys = poly[:, 0], poly[:, 1]
|
|
30
|
+
boxes.append((int(xs.min()), int(ys.min()), int(xs.max()), int(ys.max())))
|
|
31
|
+
return boxes
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def detect_codes_in_frame(frame) -> List[Box]:
|
|
35
|
+
"""Detect QR codes, and barcodes when the OpenCV build provides them."""
|
|
36
|
+
import cv2
|
|
37
|
+
|
|
38
|
+
boxes: List[Box] = []
|
|
39
|
+
try:
|
|
40
|
+
detector = cv2.QRCodeDetector()
|
|
41
|
+
ok, decoded, points, _ = detector.detectAndDecodeMulti(frame)
|
|
42
|
+
if points is not None:
|
|
43
|
+
boxes.extend(boxes_from_points(points))
|
|
44
|
+
elif ok:
|
|
45
|
+
single_ok, single_points = detector.detect(frame)
|
|
46
|
+
if single_ok:
|
|
47
|
+
boxes.extend(boxes_from_points(single_points))
|
|
48
|
+
except Exception:
|
|
49
|
+
pass
|
|
50
|
+
# Barcode detector lives in opencv-contrib builds only
|
|
51
|
+
try:
|
|
52
|
+
barcode_detector = cv2.barcode.BarcodeDetector() # type: ignore[attr-defined]
|
|
53
|
+
ok, decoded_info, decoded_type, points = barcode_detector.detectAndDecodeWithType(frame)
|
|
54
|
+
if points is not None:
|
|
55
|
+
boxes.extend(boxes_from_points(points))
|
|
56
|
+
except Exception:
|
|
57
|
+
pass
|
|
58
|
+
return boxes
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
class CodeScanner:
|
|
62
|
+
"""Sample frames, detect codes, and hold each box until the next sample."""
|
|
63
|
+
|
|
64
|
+
def __init__(self, sample_every: int = 10) -> None:
|
|
65
|
+
self.sample_every = sample_every
|
|
66
|
+
|
|
67
|
+
def scan_frames(self, frames) -> Dict[str, Dict[int, List[Box]]]:
|
|
68
|
+
result: Dict[str, Dict[int, List[Box]]] = {}
|
|
69
|
+
total = len(frames)
|
|
70
|
+
count = 0
|
|
71
|
+
for idx in range(0, total, max(1, self.sample_every)):
|
|
72
|
+
boxes = detect_codes_in_frame(frames[idx])
|
|
73
|
+
for box in boxes:
|
|
74
|
+
count += 1
|
|
75
|
+
key = f"code:{count}"
|
|
76
|
+
hold_until = min(total, idx + max(1, self.sample_every))
|
|
77
|
+
for f in range(idx, hold_until):
|
|
78
|
+
result.setdefault(key, {}).setdefault(f, []).append(box)
|
|
79
|
+
print(f"Code: QR or barcode covered near frame {idx}")
|
|
80
|
+
return result
|
|
@@ -0,0 +1,117 @@
|
|
|
1
|
+
"""Photo pipeline. One image in, redacted image out, same layers as video."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from typing import Dict, List
|
|
7
|
+
|
|
8
|
+
import cv2
|
|
9
|
+
import numpy as np
|
|
10
|
+
|
|
11
|
+
from .masks import build_per_frame_masks
|
|
12
|
+
from .pipeline import apply_redaction_to_frame
|
|
13
|
+
from .sam_client import LocalSamStub, SamApiClient, SegmentationResult
|
|
14
|
+
|
|
15
|
+
SUPPORTED_IMAGE_SUFFIXES = {".jpg", ".jpeg", ".png", ".webp", ".bmp"}
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def run_image_pipeline(
|
|
19
|
+
input_path: Path,
|
|
20
|
+
output_path: Path,
|
|
21
|
+
targets: List[str],
|
|
22
|
+
mode: str = "blur",
|
|
23
|
+
strength: int = 21,
|
|
24
|
+
mask_margin: int = 10,
|
|
25
|
+
local: bool = False,
|
|
26
|
+
api_key_env: str = "MODEL_API_KEY",
|
|
27
|
+
backend: str | None = None,
|
|
28
|
+
endpoint: str | None = None,
|
|
29
|
+
provider: str | None = None,
|
|
30
|
+
pii_text: bool = False,
|
|
31
|
+
codes: bool = False,
|
|
32
|
+
) -> Dict[str, object]:
|
|
33
|
+
frame = cv2.imread(str(input_path))
|
|
34
|
+
if frame is None:
|
|
35
|
+
raise ValueError(f"Could not read image: {input_path}")
|
|
36
|
+
height, width = frame.shape[:2]
|
|
37
|
+
shape = (height, width)
|
|
38
|
+
print(f"Input photo: {input_path} {width}x{height}")
|
|
39
|
+
print(f"Targets: {', '.join(targets)}")
|
|
40
|
+
|
|
41
|
+
resolved_backend = backend or ("local" if local else "api")
|
|
42
|
+
resolved_provider = provider or ("grounding-sam" if resolved_backend == "local" else "sam")
|
|
43
|
+
client: object
|
|
44
|
+
if resolved_backend == "local":
|
|
45
|
+
if resolved_provider == "grounding-sam":
|
|
46
|
+
from .local_gsam import LocalGroundingSamClient
|
|
47
|
+
|
|
48
|
+
client = LocalGroundingSamClient()
|
|
49
|
+
else:
|
|
50
|
+
client = LocalSamStub()
|
|
51
|
+
else:
|
|
52
|
+
client = SamApiClient.from_env(api_key_env, endpoint=endpoint)
|
|
53
|
+
if resolved_backend == "hosted":
|
|
54
|
+
print(f"Hosted backend endpoint: {client.endpoint}") # type: ignore[attr-defined]
|
|
55
|
+
|
|
56
|
+
all_tracks: Dict[str, Dict[int, np.ndarray]] = {}
|
|
57
|
+
total_objects = 0
|
|
58
|
+
for phrase in targets:
|
|
59
|
+
try:
|
|
60
|
+
result: SegmentationResult = client.segment_image(input_path, phrase, shape=shape) # type: ignore[attr-defined]
|
|
61
|
+
except Exception as exc:
|
|
62
|
+
print(f"Warning: SAM failed for phrase '{phrase}': {exc}")
|
|
63
|
+
result = SegmentationResult(phrase=phrase)
|
|
64
|
+
total_objects += len(result.objects)
|
|
65
|
+
for track_id, frame_map in result.tracks.items():
|
|
66
|
+
merged: Dict[int, np.ndarray] = {}
|
|
67
|
+
for _fidx, m in frame_map.items():
|
|
68
|
+
merged[0] = m
|
|
69
|
+
if merged:
|
|
70
|
+
all_tracks[f"{phrase}:{track_id}"] = merged
|
|
71
|
+
|
|
72
|
+
frames = [frame]
|
|
73
|
+
if pii_text:
|
|
74
|
+
from .pii import PiiScanner
|
|
75
|
+
|
|
76
|
+
for key, frame_boxes in PiiScanner(sample_every=1).scan_frames(frames).items():
|
|
77
|
+
for fidx, boxes in frame_boxes.items():
|
|
78
|
+
m = np.zeros(shape, dtype=bool)
|
|
79
|
+
for x1, y1, x2, y2 in boxes:
|
|
80
|
+
m[max(0, y1) : min(height, y2), max(0, x1) : min(width, x2)] = True
|
|
81
|
+
if np.any(m):
|
|
82
|
+
all_tracks[f"pii:{key}"] = {0: m}
|
|
83
|
+
total_objects += 1
|
|
84
|
+
if codes:
|
|
85
|
+
from .codes import CodeScanner
|
|
86
|
+
|
|
87
|
+
for key, frame_boxes in CodeScanner(sample_every=1).scan_frames(frames).items():
|
|
88
|
+
for fidx, boxes in frame_boxes.items():
|
|
89
|
+
m = np.zeros(shape, dtype=bool)
|
|
90
|
+
for x1, y1, x2, y2 in boxes:
|
|
91
|
+
m[max(0, y1) : min(height, y2), max(0, x1) : min(width, x2)] = True
|
|
92
|
+
if np.any(m):
|
|
93
|
+
all_tracks[f"code:{key}"] = {0: m}
|
|
94
|
+
total_objects += 1
|
|
95
|
+
|
|
96
|
+
if total_objects == 0:
|
|
97
|
+
print("Nothing matched. Writing a clean copy.")
|
|
98
|
+
masks = build_per_frame_masks(
|
|
99
|
+
tracks=all_tracks, total_frames=1, shape=shape, margin=mask_margin, carry_frames=0, smooth_radius=1
|
|
100
|
+
)
|
|
101
|
+
if mode == "replace":
|
|
102
|
+
from .replace import apply_replacement
|
|
103
|
+
redacted = frame
|
|
104
|
+
for track_key, track_map in all_tracks.items():
|
|
105
|
+
single = build_per_frame_masks(
|
|
106
|
+
tracks={track_key: track_map}, total_frames=1, shape=shape,
|
|
107
|
+
margin=mask_margin, carry_frames=0, smooth_radius=1,
|
|
108
|
+
)
|
|
109
|
+
if np.any(single[0]):
|
|
110
|
+
redacted = apply_replacement(redacted, single[0], track_key)
|
|
111
|
+
else:
|
|
112
|
+
redacted = apply_redaction_to_frame(frame, masks[0], mode=mode if mode in ("blur", "pixelate") else "blur", strength=strength)
|
|
113
|
+
output_path.parent.mkdir(parents=True, exist_ok=True)
|
|
114
|
+
if not cv2.imwrite(str(output_path), redacted):
|
|
115
|
+
raise RuntimeError(f"Could not write output image: {output_path}")
|
|
116
|
+
print(f"Wrote: {output_path}")
|
|
117
|
+
return {"input": str(input_path), "output": str(output_path), "objects": total_objects, "photo": True}
|