pdf-metaclean 1.0.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,91 @@
1
+ Metadata-Version: 2.4
2
+ Name: pdf-metaclean
3
+ Version: 1.0.1
4
+ Summary: Strip hidden metadata (author, producer, XMP) from PDFs before sharing — byte-level, no re-encoding.
5
+ Author: Danilo Fortunato
6
+ License-Expression: MIT
7
+ Project-URL: Homepage, https://danyblitz-bit.github.io/pdf-metaclean/
8
+ Project-URL: Repository, https://github.com/danyblitz-bit/pdf-metaclean
9
+ Project-URL: Bug Tracker, https://github.com/danyblitz-bit/pdf-metaclean/issues
10
+ Keywords: pdf,metadata,privacy,xmp,strip,redact,gdpr,cli
11
+ Classifier: Programming Language :: Python :: 3
12
+ Classifier: Environment :: Console
13
+ Classifier: Topic :: Security
14
+ Classifier: Topic :: Text Processing
15
+ Classifier: Intended Audience :: Developers
16
+ Classifier: Intended Audience :: End Users/Desktop
17
+ Requires-Python: >=3.9
18
+ Description-Content-Type: text/markdown
19
+ Requires-Dist: pypdf>=6.15.0
20
+
21
+ # PDF MetaClean
22
+
23
+ Remove sensitive metadata from PDF files before you share them. One command, three modes.
24
+
25
+ ## Why
26
+
27
+ A PDF exported from your business software can silently carry your **name, company, software, and even editing timestamps** — freelancers leak their identity this way all the time. PDF MetaClean strips it all.
28
+
29
+ ## Install & Run
30
+
31
+ ```bash
32
+ pip install -r requirements.txt
33
+
34
+ # Inspect what's hidden in a PDF
35
+ python -m tools.pdf-metaclean report.pdf
36
+
37
+ # See the full metadata list
38
+ python -m tools.pdf-metaclean report.pdf --verbose
39
+
40
+ # Write a sanitized copy
41
+ python -m tools.pdf-metaclean report.pdf --clean
42
+ # -> report_clean.pdf
43
+
44
+ # Overwrite the original
45
+ python -m tools.pdf-metaclean report.pdf --in-place
46
+
47
+ # Verify the tool works on your machine
48
+ python -m tools.pdf-metaclean --self-check
49
+ ```
50
+
51
+ ## Options
52
+
53
+ | Flag | Description |
54
+ |------|-------------|
55
+ | `--clean` | Write cleaned copy as `<name>_clean.pdf` |
56
+ | `--in-place` | Overwrite the original file after cleaning |
57
+ | `--verbose` | List all metadata found before cleaning |
58
+ | `--self-check` | Run a built-in correctness test |
59
+
60
+ ## Why the byte-level strip
61
+
62
+ pypdf silently re-injects `/Producer` on every write. PDF MetaClean strips the PDF `/Info` trailer entry at the byte level after writing, so **no metadata can survive** — verified by the built-in self-check.
63
+
64
+ ## Support the Project
65
+
66
+ PDF MetaClean is free and open source. Like it?
67
+
68
+ - [Get your site audited — €49](https://danyblitz.gumroad.com/l/jsuyla) — I run the audit and send you the fixes
69
+ - [Buy me a coffee](https://danyblitz.gumroad.com/l/hrvpiu) — one-time support
70
+
71
+ ## Report a bug
72
+
73
+ Found a bug or something weird? Run:
74
+
75
+ ```bash
76
+ python -m tools.pdf-metaclean --report
77
+ ```
78
+
79
+ This opens a pre-filled email. Send it and I'll get notified automatically.
80
+
81
+ You can also email **danyblitz@googlemail.com** directly. Use the subject format:
82
+
83
+ ```
84
+ [TOOL-REPORT] pdf-metaclean <what happened>
85
+ ```
86
+
87
+ Attach the PDF or log output if you have one.
88
+
89
+ ## License
90
+
91
+ MIT
@@ -0,0 +1,71 @@
1
+ # PDF MetaClean
2
+
3
+ Remove sensitive metadata from PDF files before you share them. One command, three modes.
4
+
5
+ ## Why
6
+
7
+ A PDF exported from your business software can silently carry your **name, company, software, and even editing timestamps** — freelancers leak their identity this way all the time. PDF MetaClean strips it all.
8
+
9
+ ## Install & Run
10
+
11
+ ```bash
12
+ pip install -r requirements.txt
13
+
14
+ # Inspect what's hidden in a PDF
15
+ python -m tools.pdf-metaclean report.pdf
16
+
17
+ # See the full metadata list
18
+ python -m tools.pdf-metaclean report.pdf --verbose
19
+
20
+ # Write a sanitized copy
21
+ python -m tools.pdf-metaclean report.pdf --clean
22
+ # -> report_clean.pdf
23
+
24
+ # Overwrite the original
25
+ python -m tools.pdf-metaclean report.pdf --in-place
26
+
27
+ # Verify the tool works on your machine
28
+ python -m tools.pdf-metaclean --self-check
29
+ ```
30
+
31
+ ## Options
32
+
33
+ | Flag | Description |
34
+ |------|-------------|
35
+ | `--clean` | Write cleaned copy as `<name>_clean.pdf` |
36
+ | `--in-place` | Overwrite the original file after cleaning |
37
+ | `--verbose` | List all metadata found before cleaning |
38
+ | `--self-check` | Run a built-in correctness test |
39
+
40
+ ## Why the byte-level strip
41
+
42
+ pypdf silently re-injects `/Producer` on every write. PDF MetaClean strips the PDF `/Info` trailer entry at the byte level after writing, so **no metadata can survive** — verified by the built-in self-check.
43
+
44
+ ## Support the Project
45
+
46
+ PDF MetaClean is free and open source. Like it?
47
+
48
+ - [Get your site audited — €49](https://danyblitz.gumroad.com/l/jsuyla) — I run the audit and send you the fixes
49
+ - [Buy me a coffee](https://danyblitz.gumroad.com/l/hrvpiu) — one-time support
50
+
51
+ ## Report a bug
52
+
53
+ Found a bug or something weird? Run:
54
+
55
+ ```bash
56
+ python -m tools.pdf-metaclean --report
57
+ ```
58
+
59
+ This opens a pre-filled email. Send it and I'll get notified automatically.
60
+
61
+ You can also email **danyblitz@googlemail.com** directly. Use the subject format:
62
+
63
+ ```
64
+ [TOOL-REPORT] pdf-metaclean <what happened>
65
+ ```
66
+
67
+ Attach the PDF or log output if you have one.
68
+
69
+ ## License
70
+
71
+ MIT
@@ -0,0 +1,2 @@
1
+ """PDF MetaClean — strip sensitive metadata from PDF files."""
2
+ __version__ = "1.0.1"
@@ -0,0 +1,151 @@
1
+ """PDF MetaClean — strip sensitive metadata before sharing PDFs."""
2
+ import io
3
+ import re
4
+ import sys
5
+ import argparse
6
+ from pathlib import Path
7
+
8
+ from pypdf import PdfReader, PdfWriter
9
+
10
+
11
+ def audit(path: Path) -> dict:
12
+ reader = PdfReader(str(path))
13
+ meta = reader.metadata
14
+ return {
15
+ "pages": len(reader.pages),
16
+ "metadata": {k: str(v) for k, v in meta.items()} if meta else {},
17
+ "has_metadata": bool(meta),
18
+ }
19
+
20
+
21
+ def clean(path: Path, out: Path) -> dict:
22
+ original = path.stat().st_size
23
+ reader = PdfReader(str(path))
24
+ writer = PdfWriter()
25
+ for page in reader.pages:
26
+ writer.add_page(page)
27
+
28
+ buf = io.BytesIO()
29
+ writer.write(buf)
30
+ raw = buf.getvalue().decode("latin-1")
31
+ # pypdf injects /Producer; strip the /Info reference and dict from the trailer outright.
32
+ # ponytail: flat-key Info dicts only; nested binary streams untouched. Upgrade to qpdf if exotic trailers appear.
33
+ raw = re.sub(r"/Info \d+ \d+ R", "", raw)
34
+ raw = re.sub(r"/Info <<[^>]*>>", "", raw)
35
+ out.write_bytes(raw.encode("latin-1"))
36
+
37
+ result = out.stat().st_size
38
+ return {"cleaned": True, "input_bytes": original, "output_bytes": result, "saved_bytes": original - result}
39
+
40
+
41
+ def fmt(v: int) -> str:
42
+ return f"{v} B" if v < 1024 else f"{v/1024:.1f} KB"
43
+
44
+
45
+ def self_check() -> int:
46
+ import tempfile
47
+
48
+ from pypdf import PdfWriter
49
+
50
+ buf = io.BytesIO()
51
+ w = PdfWriter()
52
+ w.add_blank_page(width=595, height=842)
53
+ w.add_metadata({"/Author": "Sensitive Name", "/Title": "secret"})
54
+ w.write(buf)
55
+
56
+ with tempfile.TemporaryDirectory() as d:
57
+ src = Path(d) / "in.pdf"
58
+ out = Path(d) / "out.pdf"
59
+ src.write_bytes(buf.getvalue())
60
+ res = clean(src, out)
61
+ assert res["cleaned"]
62
+ assert "/Author" not in out.read_bytes().decode("latin-1"), "Author metadata survived"
63
+ check = audit(out)
64
+ assert not check["has_metadata"], f"Metadata remained: {check['metadata']}"
65
+ assert check["pages"] == 1, "Page count changed"
66
+
67
+ # in-place: out is path, so the input size must be read before writing
68
+ ip = Path(d) / "inplace.pdf"
69
+ ip.write_bytes(buf.getvalue())
70
+ res2 = clean(ip, ip)
71
+ assert res2["saved_bytes"] > 0, f"in-place reported no savings: {res2}"
72
+ print("self-check OK: metadata stripped, page preserved")
73
+ return 0
74
+
75
+
76
+ def report_to_email(message: str):
77
+ import webbrowser
78
+ from urllib.parse import quote
79
+
80
+ subject = "[TOOL-REPORT] pdf-metaclean bug or issue"
81
+ webbrowser.open(f"mailto:danyblitz@googlemail.com?subject={quote(subject)}&body={quote(message)}")
82
+
83
+
84
+ def main():
85
+ parser = argparse.ArgumentParser(
86
+ prog="pdf-metaclean",
87
+ description="Remove sensitive metadata from PDF files. Print-audit a file or clean it in place.",
88
+ )
89
+ parser.add_argument("file", nargs="*", help="PDF file(s) to audit or clean")
90
+ parser.add_argument("--self-check", action="store_true", help="Run internal correctness check")
91
+ output_mode = parser.add_mutually_exclusive_group()
92
+ output_mode.add_argument("--clean", action="store_true", help="Write a cleaned copy as <name>_clean.pdf")
93
+ output_mode.add_argument("--in-place", action="store_true", help="Overwrite the original after cleaning")
94
+ parser.add_argument("--verbose", action="store_true", help="Show full metadata before cleaning")
95
+ parser.add_argument("--report", action="store_true", help="Open a pre-filled email to report an issue")
96
+ args = parser.parse_args()
97
+
98
+ if args.self_check:
99
+ sys.exit(self_check())
100
+
101
+ if not args.file:
102
+ parser.error("at least one PDF file is required (or use --self-check)")
103
+
104
+ summary_lines = []
105
+ for file_arg in args.file:
106
+ p = Path(file_arg)
107
+ if not p.exists():
108
+ print(f"[SKIP] {p}: not found")
109
+ summary_lines.append(f"[SKIP] {p}: not found")
110
+ continue
111
+ if p.suffix.lower() != ".pdf":
112
+ print(f"[SKIP] {p}: not a PDF")
113
+ summary_lines.append(f"[SKIP] {p}: not a PDF")
114
+ continue
115
+
116
+ info = audit(p)
117
+ print(f"\n=== {p.name} ===")
118
+ print(f" Pages: {info['pages']} | Size: {fmt(p.stat().st_size)}")
119
+ summary_lines.append(f"=== {p.name} === Pages: {info['pages']} Size: {fmt(p.stat().st_size)}")
120
+
121
+ if not info["has_metadata"]:
122
+ print(" Metadata: none found", " (already clean)" if args.clean else "")
123
+ if args.clean and not args.in_place:
124
+ src = p.read_bytes()
125
+ out = p.with_name(p.stem + "_clean.pdf")
126
+ out.write_bytes(src)
127
+ print(f" Cleaned copy written to {out.name} (no metadata to remove, size unchanged)")
128
+ continue
129
+
130
+ if args.verbose:
131
+ print(" Metadata found:")
132
+ for k, v in info["metadata"].items():
133
+ print(f" {k}: {v[:80]}")
134
+ summary_lines.append(f" {k}: {v[:80]}")
135
+
136
+ if args.clean or args.in_place:
137
+ out_path = p if args.in_place else p.with_name(p.stem + "_clean.pdf")
138
+ res = clean(p, out_path)
139
+ print(f" Cleaned -> {out_path.name} ({fmt(res['output_bytes'])}, {fmt(res['saved_bytes'])} saved)")
140
+ else:
141
+ fields = ", ".join(info["metadata"].keys()) or "(none printable)"
142
+ print(f" Metadata present: {fields}")
143
+ print(" Use --clean to write a sanitized copy, or --in-place to overwrite.")
144
+
145
+ if args.report:
146
+ report_to_email("\n".join(summary_lines))
147
+ print("\n Email draft opened — send it and I'll get notified automatically.")
148
+
149
+
150
+ if __name__ == "__main__":
151
+ main()
@@ -0,0 +1,91 @@
1
+ Metadata-Version: 2.4
2
+ Name: pdf-metaclean
3
+ Version: 1.0.1
4
+ Summary: Strip hidden metadata (author, producer, XMP) from PDFs before sharing — byte-level, no re-encoding.
5
+ Author: Danilo Fortunato
6
+ License-Expression: MIT
7
+ Project-URL: Homepage, https://danyblitz-bit.github.io/pdf-metaclean/
8
+ Project-URL: Repository, https://github.com/danyblitz-bit/pdf-metaclean
9
+ Project-URL: Bug Tracker, https://github.com/danyblitz-bit/pdf-metaclean/issues
10
+ Keywords: pdf,metadata,privacy,xmp,strip,redact,gdpr,cli
11
+ Classifier: Programming Language :: Python :: 3
12
+ Classifier: Environment :: Console
13
+ Classifier: Topic :: Security
14
+ Classifier: Topic :: Text Processing
15
+ Classifier: Intended Audience :: Developers
16
+ Classifier: Intended Audience :: End Users/Desktop
17
+ Requires-Python: >=3.9
18
+ Description-Content-Type: text/markdown
19
+ Requires-Dist: pypdf>=6.15.0
20
+
21
+ # PDF MetaClean
22
+
23
+ Remove sensitive metadata from PDF files before you share them. One command, three modes.
24
+
25
+ ## Why
26
+
27
+ A PDF exported from your business software can silently carry your **name, company, software, and even editing timestamps** — freelancers leak their identity this way all the time. PDF MetaClean strips it all.
28
+
29
+ ## Install & Run
30
+
31
+ ```bash
32
+ pip install -r requirements.txt
33
+
34
+ # Inspect what's hidden in a PDF
35
+ python -m tools.pdf-metaclean report.pdf
36
+
37
+ # See the full metadata list
38
+ python -m tools.pdf-metaclean report.pdf --verbose
39
+
40
+ # Write a sanitized copy
41
+ python -m tools.pdf-metaclean report.pdf --clean
42
+ # -> report_clean.pdf
43
+
44
+ # Overwrite the original
45
+ python -m tools.pdf-metaclean report.pdf --in-place
46
+
47
+ # Verify the tool works on your machine
48
+ python -m tools.pdf-metaclean --self-check
49
+ ```
50
+
51
+ ## Options
52
+
53
+ | Flag | Description |
54
+ |------|-------------|
55
+ | `--clean` | Write cleaned copy as `<name>_clean.pdf` |
56
+ | `--in-place` | Overwrite the original file after cleaning |
57
+ | `--verbose` | List all metadata found before cleaning |
58
+ | `--self-check` | Run a built-in correctness test |
59
+
60
+ ## Why the byte-level strip
61
+
62
+ pypdf silently re-injects `/Producer` on every write. PDF MetaClean strips the PDF `/Info` trailer entry at the byte level after writing, so **no metadata can survive** — verified by the built-in self-check.
63
+
64
+ ## Support the Project
65
+
66
+ PDF MetaClean is free and open source. Like it?
67
+
68
+ - [Get your site audited — €49](https://danyblitz.gumroad.com/l/jsuyla) — I run the audit and send you the fixes
69
+ - [Buy me a coffee](https://danyblitz.gumroad.com/l/hrvpiu) — one-time support
70
+
71
+ ## Report a bug
72
+
73
+ Found a bug or something weird? Run:
74
+
75
+ ```bash
76
+ python -m tools.pdf-metaclean --report
77
+ ```
78
+
79
+ This opens a pre-filled email. Send it and I'll get notified automatically.
80
+
81
+ You can also email **danyblitz@googlemail.com** directly. Use the subject format:
82
+
83
+ ```
84
+ [TOOL-REPORT] pdf-metaclean <what happened>
85
+ ```
86
+
87
+ Attach the PDF or log output if you have one.
88
+
89
+ ## License
90
+
91
+ MIT
@@ -0,0 +1,12 @@
1
+ README.md
2
+ __init__.py
3
+ __main__.py
4
+ pyproject.toml
5
+ ./__init__.py
6
+ ./__main__.py
7
+ pdf_metaclean.egg-info/PKG-INFO
8
+ pdf_metaclean.egg-info/SOURCES.txt
9
+ pdf_metaclean.egg-info/dependency_links.txt
10
+ pdf_metaclean.egg-info/entry_points.txt
11
+ pdf_metaclean.egg-info/requires.txt
12
+ pdf_metaclean.egg-info/top_level.txt
@@ -0,0 +1,2 @@
1
+ [console_scripts]
2
+ pdfmetaclean = pdfmetaclean.__main__:main
@@ -0,0 +1 @@
1
+ pypdf>=6.15.0
@@ -0,0 +1 @@
1
+ pdfmetaclean
@@ -0,0 +1,45 @@
1
+ [build-system]
2
+ requires = ["setuptools>=68"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "pdf-metaclean"
7
+ version = "1.0.1"
8
+ description = "Strip hidden metadata (author, producer, XMP) from PDFs before sharing — byte-level, no re-encoding."
9
+ readme = "README.md"
10
+ requires-python = ">=3.9"
11
+ license = "MIT"
12
+ authors = [{ name = "Danilo Fortunato" }]
13
+ keywords = [
14
+ "pdf",
15
+ "metadata",
16
+ "privacy",
17
+ "xmp",
18
+ "strip",
19
+ "redact",
20
+ "gdpr",
21
+ "cli",
22
+ ]
23
+ classifiers = [
24
+ "Programming Language :: Python :: 3",
25
+ "Environment :: Console",
26
+ "Topic :: Security",
27
+ "Topic :: Text Processing",
28
+ "Intended Audience :: Developers",
29
+ "Intended Audience :: End Users/Desktop",
30
+ ]
31
+ dependencies = ["pypdf>=6.15.0"]
32
+
33
+ [project.urls]
34
+ Homepage = "https://danyblitz-bit.github.io/pdf-metaclean/"
35
+ Repository = "https://github.com/danyblitz-bit/pdf-metaclean"
36
+ "Bug Tracker" = "https://github.com/danyblitz-bit/pdf-metaclean/issues"
37
+
38
+ [project.scripts]
39
+ pdfmetaclean = "pdfmetaclean.__main__:main"
40
+
41
+ [tool.setuptools]
42
+ packages = ["pdfmetaclean"]
43
+
44
+ [tool.setuptools.package-dir]
45
+ pdfmetaclean = "."
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+