watermark-cleaner 0.2.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- watermark_cleaner-0.2.1/LICENSE +21 -0
- watermark_cleaner-0.2.1/PKG-INFO +194 -0
- watermark_cleaner-0.2.1/README.md +168 -0
- watermark_cleaner-0.2.1/pyproject.toml +45 -0
- watermark_cleaner-0.2.1/python/watermark_cleaner/__init__.py +5 -0
- watermark_cleaner-0.2.1/python/watermark_cleaner/cli.py +109 -0
- watermark_cleaner-0.2.1/python/watermark_cleaner/config.py +57 -0
- watermark_cleaner-0.2.1/python/watermark_cleaner/core.py +68 -0
- watermark_cleaner-0.2.1/python/watermark_cleaner/findings.py +57 -0
- watermark_cleaner-0.2.1/python/watermark_cleaner/layers/__init__.py +0 -0
- watermark_cleaner-0.2.1/python/watermark_cleaner/layers/characters.py +150 -0
- watermark_cleaner-0.2.1/python/watermark_cleaner/layers/homoglyphs.py +38 -0
- watermark_cleaner-0.2.1/python/watermark_cleaner/layers/metadata.py +214 -0
- watermark_cleaner-0.2.1/python/watermark_cleaner/layers/typography.py +56 -0
- watermark_cleaner-0.2.1/python/watermark_cleaner/layers/voice.py +171 -0
- watermark_cleaner-0.2.1/python/watermark_cleaner/report.py +46 -0
- watermark_cleaner-0.2.1/python/watermark_cleaner/rewrite/__init__.py +0 -0
- watermark_cleaner-0.2.1/python/watermark_cleaner/rewrite/deepl.py +81 -0
- watermark_cleaner-0.2.1/python/watermark_cleaner/rules.py +37 -0
- watermark_cleaner-0.2.1/python/watermark_cleaner/rules_data/characters.json +65 -0
- watermark_cleaner-0.2.1/python/watermark_cleaner/rules_data/homoglyphs.json +14 -0
- watermark_cleaner-0.2.1/python/watermark_cleaner/rules_data/phrases.json +98 -0
- watermark_cleaner-0.2.1/python/watermark_cleaner/rules_data/typography.json +39 -0
- watermark_cleaner-0.2.1/python/watermark_cleaner/runner.py +94 -0
- watermark_cleaner-0.2.1/python/watermark_cleaner/safeio.py +46 -0
- watermark_cleaner-0.2.1/setup.cfg +4 -0
- watermark_cleaner-0.2.1/tests/test_core.py +226 -0
- watermark_cleaner-0.2.1/tests/test_metadata.py +192 -0
- watermark_cleaner-0.2.1/tests/test_safeio.py +62 -0
- watermark_cleaner-0.2.1/watermark_cleaner.egg-info/PKG-INFO +194 -0
- watermark_cleaner-0.2.1/watermark_cleaner.egg-info/SOURCES.txt +32 -0
- watermark_cleaner-0.2.1/watermark_cleaner.egg-info/dependency_links.txt +1 -0
- watermark_cleaner-0.2.1/watermark_cleaner.egg-info/entry_points.txt +2 -0
- watermark_cleaner-0.2.1/watermark_cleaner.egg-info/top_level.txt +1 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Christian Strunk
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,194 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: watermark-cleaner
|
|
3
|
+
Version: 0.2.1
|
|
4
|
+
Summary: Remove AI text artifacts, AI phrases and file metadata before you publish. Detects and strips invisible unicode characters, losslessly cleans image metadata.
|
|
5
|
+
Author: Christian Strunk
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/pixelstrunk/watermark-cleaner
|
|
8
|
+
Project-URL: Issues, https://github.com/pixelstrunk/watermark-cleaner/issues
|
|
9
|
+
Project-URL: Changelog, https://github.com/pixelstrunk/watermark-cleaner/blob/main/CHANGELOG.md
|
|
10
|
+
Keywords: ai,watermark,unicode,text-cleaning,metadata,publishing,pre-commit,trojan-source
|
|
11
|
+
Classifier: Programming Language :: Python :: 3
|
|
12
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
17
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
18
|
+
Classifier: Operating System :: OS Independent
|
|
19
|
+
Classifier: Environment :: Console
|
|
20
|
+
Classifier: Intended Audience :: Developers
|
|
21
|
+
Classifier: Topic :: Text Processing :: Filters
|
|
22
|
+
Requires-Python: >=3.9
|
|
23
|
+
Description-Content-Type: text/markdown
|
|
24
|
+
License-File: LICENSE
|
|
25
|
+
Dynamic: license-file
|
|
26
|
+
|
|
27
|
+
# Watermark Cleaner
|
|
28
|
+
|
|
29
|
+
Remove AI text artifacts, hidden unicode characters and file metadata before you publish.
|
|
30
|
+
|
|
31
|
+
[](https://github.com/pixelstrunk/watermark-cleaner/actions/workflows/ci.yml)
|
|
32
|
+
[](https://pypi.org/project/watermark-cleaner/)
|
|
33
|
+
[](https://www.npmjs.com/package/watermark-cleaner)
|
|
34
|
+
[](LICENSE)
|
|
35
|
+
|
|
36
|
+
Watermark Cleaner is a deterministic, offline, zero-dependency tool for content you own. It cleans the mechanical traces that mark text as machine generated, flags the stylistic phrases that read as AI writing, and strips identifying metadata from images without touching a single pixel. It ships as a Python CLI and a Node CLI that share one rule set and are tested to behave identically.
|
|
37
|
+
|
|
38
|
+
## See it
|
|
39
|
+
|
|
40
|
+
The problem is invisible by definition. Escaped, it looks like this:
|
|
41
|
+
|
|
42
|
+
```
|
|
43
|
+
before "Smart quotes “work” — and a hidden watermark"
|
|
44
|
+
after "Smart quotes \"work\", and a hidden watermark"
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
```
|
|
48
|
+
$ watermark-cleaner check post.md
|
|
49
|
+
post.md
|
|
50
|
+
characters fixed removed zero width space (U+200B) x2
|
|
51
|
+
typography fixed straightened smart quotes x2
|
|
52
|
+
voice error ai phrases present (rewrite required) x1
|
|
53
|
+
|
|
54
|
+
1 file checked, 1 blocked
|
|
55
|
+
$ echo $?
|
|
56
|
+
1
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
`check` reports and never changes anything. `fix` cleans in place and writes a `.bak` backup next to every changed file.
|
|
60
|
+
|
|
61
|
+
## Install
|
|
62
|
+
|
|
63
|
+
| Ecosystem | One-shot | Install |
|
|
64
|
+
|---|---|---|
|
|
65
|
+
| Python | `uvx watermark-cleaner check .` or `pipx run watermark-cleaner check .` | `pipx install watermark-cleaner` |
|
|
66
|
+
| Node | `npx watermark-cleaner check .` | `npm install -g watermark-cleaner` |
|
|
67
|
+
| pre-commit | see below | `.pre-commit-config.yaml` snippet |
|
|
68
|
+
| From source | | `pip install .` or `npm install -g ./node` |
|
|
69
|
+
|
|
70
|
+
Both packages install a single command, `watermark-cleaner`, available in every folder, like git.
|
|
71
|
+
|
|
72
|
+
## Use
|
|
73
|
+
|
|
74
|
+
```
|
|
75
|
+
watermark-cleaner check . # report only, changes nothing, exits non zero if blocked
|
|
76
|
+
watermark-cleaner fix . # clean in place, writes a .bak backup per changed file
|
|
77
|
+
watermark-cleaner check post.md
|
|
78
|
+
watermark-cleaner fix ./content --no-voice
|
|
79
|
+
watermark-cleaner fix ./assets --aggressive # also replace homoglyphs and strip variation selectors
|
|
80
|
+
watermark-cleaner check . --json
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
`check` is safe to run over an entire folder to see what is inside without touching anything. Exit codes: `check` returns 1 when a file contains AI phrases or AI sentence shapes, so it works as a publish gate. `fix` returns 0 after cleaning what it can; pass `--strict` to make `fix` return 1 when blocking findings remain that need a human rewrite.
|
|
84
|
+
|
|
85
|
+
Markdown structure is protected: typography and voice rules never touch fenced code blocks, inline code or YAML frontmatter, so code examples in documentation survive cleaning. Invisible characters are still removed inside code, because hidden characters in code are exactly the Trojan Source attack this tool defends against. Disable with `"protect_code": false` if you want full-document typography.
|
|
86
|
+
|
|
87
|
+
## What it removes, and what it deliberately keeps
|
|
88
|
+
|
|
89
|
+
| Category | Codepoints | Action |
|
|
90
|
+
|---|---|---|
|
|
91
|
+
| Zero width space, word joiner, BOM | U+200B, U+2060, U+FEFF | removed |
|
|
92
|
+
| Soft hyphen, mongolian vowel separator, combining grapheme joiner | U+00AD, U+180E, U+034F | removed |
|
|
93
|
+
| Invisible math operators | U+2061 to U+2064 | removed |
|
|
94
|
+
| Unicode tag characters (hidden payloads) | U+E0000 to U+E007F | removed |
|
|
95
|
+
| Exotic spaces (nbsp, thin space, ideographic space and friends) | U+00A0, U+2000 to U+200A, U+202F, U+205F, U+3000 | replaced with a normal space |
|
|
96
|
+
| Smart quotes, ellipsis, bullets, dashes | U+2018 and friends | normalized to ascii |
|
|
97
|
+
| Homoglyphs (cyrillic і in latin text and friends) | per rulebook | warned, replaced only with `--aggressive` |
|
|
98
|
+
|
|
99
|
+
Deliberately preserved, because removing them breaks legitimate text:
|
|
100
|
+
|
|
101
|
+
- Zero width joiner (U+200D) in emoji sequences.
|
|
102
|
+
- Zero width non-joiner (U+200C) when adjacent to a script that requires it (Persian, Arabic, Indic scripts and others). Between latin letters it is a watermark and gets removed.
|
|
103
|
+
- Bidi marks and bidi controls in documents that contain right-to-left text. In pure left-to-right documents they are removed, which is the [Trojan Source](https://trojansource.codes/) defense.
|
|
104
|
+
- Non breaking spaces between digits (`12 000` keeps its formatting) unless you disable `keep_nbsp_in_numbers`.
|
|
105
|
+
- Variation selectors, unless you pass `--aggressive`.
|
|
106
|
+
|
|
107
|
+
## Image metadata, losslessly
|
|
108
|
+
|
|
109
|
+
`watermark-cleaner fix` strips EXIF, XMP, C2PA and comment segments from JPEG, PNG and WebP at the container level. Pixels are never re-encoded, so there is no quality loss and the operation is verifiable with a byte diff. ICC color profiles are kept by default, because removing them visibly shifts colors in the browser; strip them too with `"strip_icc": true` or `--aggressive`. SVG metadata, XMP blocks and XML comments are removed as text. GIF and TIFF are never modified; the tool tells you it cannot strip them losslessly and leaves them alone. Files that fail to parse are left untouched.
|
|
110
|
+
|
|
111
|
+
## Use as a publish gate
|
|
112
|
+
|
|
113
|
+
With the [pre-commit](https://pre-commit.com) framework:
|
|
114
|
+
|
|
115
|
+
```yaml
|
|
116
|
+
repos:
|
|
117
|
+
- repo: https://github.com/pixelstrunk/watermark-cleaner
|
|
118
|
+
rev: v0.2.0
|
|
119
|
+
hooks:
|
|
120
|
+
- id: watermark-cleaner-fix
|
|
121
|
+
```
|
|
122
|
+
|
|
123
|
+
`watermark-cleaner-fix` cleans the mechanical layers on every commit and never blocks on style. Add `id: watermark-cleaner-check` if you also want commits blocked on AI phrases. A plain git hook and a deploy gate script live in [`integrations/`](integrations/).
|
|
124
|
+
|
|
125
|
+
Coding agents can drive the CLI through the skill in [`skills/watermark-cleaner/`](skills/watermark-cleaner/), which enforces an inspect-first workflow.
|
|
126
|
+
|
|
127
|
+
All writes are safe by construction: files are replaced atomically, writes through symlinks are refused, and files larger than `max_file_bytes` (default 256 MiB) are skipped instead of loaded into memory.
|
|
128
|
+
|
|
129
|
+
## The voice layer, honestly
|
|
130
|
+
|
|
131
|
+
Cleaning AI text falls into buckets, and this tool is precise about which bucket each action lives in.
|
|
132
|
+
|
|
133
|
+
| Layer | What | Result |
|
|
134
|
+
|---|---|---|
|
|
135
|
+
| Characters | Invisible and format characters, exotic spaces, homoglyphs, NFC normalization | Fixed, 100% verifiable |
|
|
136
|
+
| Typography | Smart quotes, ellipsis and bullet glyphs, em and en dashes | Fixed, 100% verifiable |
|
|
137
|
+
| File metadata | EXIF, XMP and C2PA in images, metadata and comments in SVG | Stripped losslessly, 100% verifiable |
|
|
138
|
+
| Voice | AI filler phrases and AI sentence shapes from a portable rulebook | Filler removed, shapes flagged and blocked |
|
|
139
|
+
|
|
140
|
+
The voice layer auto-deletes only phrases that are pure filler ("without further ado"). When a deletion opens a sentence, the sentence is repaired: leftover spaces go away and the next word is capitalized. Phrases that carry an object ("let's explore the API") are never cut mid-sentence; they are flagged as errors for a human to rewrite. The phrase rulebook is currently English only.
|
|
141
|
+
|
|
142
|
+
## Non-goals, stated plainly
|
|
143
|
+
|
|
144
|
+
- It does not remove statistical text watermarks (the SynthID style signal that some vendors embed in word choice). No deterministic tool can, and there is no public detector to verify removal. Only substantial rewriting degrades that signal.
|
|
145
|
+
- It does not auto-rewrite AI sentence shapes. Rewriting a sentence needs judgment, so the tool detects and blocks those shapes instead of replacing them and producing nonsense.
|
|
146
|
+
- It does not certify that text will pass any AI detector, and it is not a way to misrepresent authorship.
|
|
147
|
+
|
|
148
|
+
## Configure
|
|
149
|
+
|
|
150
|
+
Drop a `watermark-cleaner.config.json` in a project root. Any key overrides the default.
|
|
151
|
+
|
|
152
|
+
```json
|
|
153
|
+
{
|
|
154
|
+
"voice": true,
|
|
155
|
+
"replace_homoglyphs": false,
|
|
156
|
+
"keep_nbsp_in_numbers": true,
|
|
157
|
+
"protect_code": true,
|
|
158
|
+
"strip_icc": false,
|
|
159
|
+
"dash_policy": { "spaced_replacement": " - ", "unspaced_replacement": "-" },
|
|
160
|
+
"custom_banned_phrases": ["our innovative product", "leverage synergies"],
|
|
161
|
+
"ignore_phrases": ["delve", "not-only-but-also"],
|
|
162
|
+
"text_extensions": [".md", ".mdx", ".txt"],
|
|
163
|
+
"exclude": ["node_modules", ".git", "dist"]
|
|
164
|
+
}
|
|
165
|
+
```
|
|
166
|
+
|
|
167
|
+
By default a spaced em dash ("fast — slow") becomes a comma ("fast, slow") and an unspaced one becomes a hyphen. That is an opinionated policy; `dash_policy` lets you change both replacements per project.
|
|
168
|
+
|
|
169
|
+
`custom_banned_phrases` adds your own blocking phrases on top of the rulebook. `ignore_phrases` switches off any built-in phrase, lexicon word or sentence shape id that does not fit your domain.
|
|
170
|
+
|
|
171
|
+
## Optional DeepL rewrite
|
|
172
|
+
|
|
173
|
+
```
|
|
174
|
+
export DEEPL_API_KEY=your-key
|
|
175
|
+
watermark-cleaner rewrite post.md
|
|
176
|
+
watermark-cleaner rewrite post.md --source-lang DE --pivot-lang EN
|
|
177
|
+
watermark-cleaner rewrite post.md --write
|
|
178
|
+
```
|
|
179
|
+
|
|
180
|
+
The document language is auto-detected by DeepL when `--source-lang` is not given.
|
|
181
|
+
|
|
182
|
+
This back translates the text (source to pivot and back) using a model that is not the one that wrote the text, then runs the deterministic cleaner on the result. It changes word choice, which is what degrades a statistical watermark, but it can shift meaning and it sends your text to DeepL. It is off by default and opt in. Do not use it on confidential content. It cannot certify that any vendor detector will fail.
|
|
183
|
+
|
|
184
|
+
## Rules are one source
|
|
185
|
+
|
|
186
|
+
All character lists and phrase lists live in [`rules/*.json`](rules/). The Python and Node packages read the same files, `scripts/sync-rules.sh` copies them into each package, and CI fails when they drift or when the two CLIs produce different output. Edit the rules once, both tools follow. Contributions to the rulebook are the easiest way to help; see [CONTRIBUTING.md](CONTRIBUTING.md).
|
|
187
|
+
|
|
188
|
+
## Scope and intent
|
|
189
|
+
|
|
190
|
+
This tool is for hygiene, privacy and transparency on content you own: see what is hidden in your text, and publish without machine fingerprints you did not choose to include.
|
|
191
|
+
|
|
192
|
+
## License
|
|
193
|
+
|
|
194
|
+
[MIT](LICENSE)
|
|
@@ -0,0 +1,168 @@
|
|
|
1
|
+
# Watermark Cleaner
|
|
2
|
+
|
|
3
|
+
Remove AI text artifacts, hidden unicode characters and file metadata before you publish.
|
|
4
|
+
|
|
5
|
+
[](https://github.com/pixelstrunk/watermark-cleaner/actions/workflows/ci.yml)
|
|
6
|
+
[](https://pypi.org/project/watermark-cleaner/)
|
|
7
|
+
[](https://www.npmjs.com/package/watermark-cleaner)
|
|
8
|
+
[](LICENSE)
|
|
9
|
+
|
|
10
|
+
Watermark Cleaner is a deterministic, offline, zero-dependency tool for content you own. It cleans the mechanical traces that mark text as machine generated, flags the stylistic phrases that read as AI writing, and strips identifying metadata from images without touching a single pixel. It ships as a Python CLI and a Node CLI that share one rule set and are tested to behave identically.
|
|
11
|
+
|
|
12
|
+
## See it
|
|
13
|
+
|
|
14
|
+
The problem is invisible by definition. Escaped, it looks like this:
|
|
15
|
+
|
|
16
|
+
```
|
|
17
|
+
before "Smart quotes “work” — and a hidden watermark"
|
|
18
|
+
after "Smart quotes \"work\", and a hidden watermark"
|
|
19
|
+
```
|
|
20
|
+
|
|
21
|
+
```
|
|
22
|
+
$ watermark-cleaner check post.md
|
|
23
|
+
post.md
|
|
24
|
+
characters fixed removed zero width space (U+200B) x2
|
|
25
|
+
typography fixed straightened smart quotes x2
|
|
26
|
+
voice error ai phrases present (rewrite required) x1
|
|
27
|
+
|
|
28
|
+
1 file checked, 1 blocked
|
|
29
|
+
$ echo $?
|
|
30
|
+
1
|
|
31
|
+
```
|
|
32
|
+
|
|
33
|
+
`check` reports and never changes anything. `fix` cleans in place and writes a `.bak` backup next to every changed file.
|
|
34
|
+
|
|
35
|
+
## Install
|
|
36
|
+
|
|
37
|
+
| Ecosystem | One-shot | Install |
|
|
38
|
+
|---|---|---|
|
|
39
|
+
| Python | `uvx watermark-cleaner check .` or `pipx run watermark-cleaner check .` | `pipx install watermark-cleaner` |
|
|
40
|
+
| Node | `npx watermark-cleaner check .` | `npm install -g watermark-cleaner` |
|
|
41
|
+
| pre-commit | see below | `.pre-commit-config.yaml` snippet |
|
|
42
|
+
| From source | | `pip install .` or `npm install -g ./node` |
|
|
43
|
+
|
|
44
|
+
Both packages install a single command, `watermark-cleaner`, available in every folder, like git.
|
|
45
|
+
|
|
46
|
+
## Use
|
|
47
|
+
|
|
48
|
+
```
|
|
49
|
+
watermark-cleaner check . # report only, changes nothing, exits non zero if blocked
|
|
50
|
+
watermark-cleaner fix . # clean in place, writes a .bak backup per changed file
|
|
51
|
+
watermark-cleaner check post.md
|
|
52
|
+
watermark-cleaner fix ./content --no-voice
|
|
53
|
+
watermark-cleaner fix ./assets --aggressive # also replace homoglyphs and strip variation selectors
|
|
54
|
+
watermark-cleaner check . --json
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
`check` is safe to run over an entire folder to see what is inside without touching anything. Exit codes: `check` returns 1 when a file contains AI phrases or AI sentence shapes, so it works as a publish gate. `fix` returns 0 after cleaning what it can; pass `--strict` to make `fix` return 1 when blocking findings remain that need a human rewrite.
|
|
58
|
+
|
|
59
|
+
Markdown structure is protected: typography and voice rules never touch fenced code blocks, inline code or YAML frontmatter, so code examples in documentation survive cleaning. Invisible characters are still removed inside code, because hidden characters in code are exactly the Trojan Source attack this tool defends against. Disable with `"protect_code": false` if you want full-document typography.
|
|
60
|
+
|
|
61
|
+
## What it removes, and what it deliberately keeps
|
|
62
|
+
|
|
63
|
+
| Category | Codepoints | Action |
|
|
64
|
+
|---|---|---|
|
|
65
|
+
| Zero width space, word joiner, BOM | U+200B, U+2060, U+FEFF | removed |
|
|
66
|
+
| Soft hyphen, mongolian vowel separator, combining grapheme joiner | U+00AD, U+180E, U+034F | removed |
|
|
67
|
+
| Invisible math operators | U+2061 to U+2064 | removed |
|
|
68
|
+
| Unicode tag characters (hidden payloads) | U+E0000 to U+E007F | removed |
|
|
69
|
+
| Exotic spaces (nbsp, thin space, ideographic space and friends) | U+00A0, U+2000 to U+200A, U+202F, U+205F, U+3000 | replaced with a normal space |
|
|
70
|
+
| Smart quotes, ellipsis, bullets, dashes | U+2018 and friends | normalized to ascii |
|
|
71
|
+
| Homoglyphs (cyrillic і in latin text and friends) | per rulebook | warned, replaced only with `--aggressive` |
|
|
72
|
+
|
|
73
|
+
Deliberately preserved, because removing them breaks legitimate text:
|
|
74
|
+
|
|
75
|
+
- Zero width joiner (U+200D) in emoji sequences.
|
|
76
|
+
- Zero width non-joiner (U+200C) when adjacent to a script that requires it (Persian, Arabic, Indic scripts and others). Between latin letters it is a watermark and gets removed.
|
|
77
|
+
- Bidi marks and bidi controls in documents that contain right-to-left text. In pure left-to-right documents they are removed, which is the [Trojan Source](https://trojansource.codes/) defense.
|
|
78
|
+
- Non breaking spaces between digits (`12 000` keeps its formatting) unless you disable `keep_nbsp_in_numbers`.
|
|
79
|
+
- Variation selectors, unless you pass `--aggressive`.
|
|
80
|
+
|
|
81
|
+
## Image metadata, losslessly
|
|
82
|
+
|
|
83
|
+
`watermark-cleaner fix` strips EXIF, XMP, C2PA and comment segments from JPEG, PNG and WebP at the container level. Pixels are never re-encoded, so there is no quality loss and the operation is verifiable with a byte diff. ICC color profiles are kept by default, because removing them visibly shifts colors in the browser; strip them too with `"strip_icc": true` or `--aggressive`. SVG metadata, XMP blocks and XML comments are removed as text. GIF and TIFF are never modified; the tool tells you it cannot strip them losslessly and leaves them alone. Files that fail to parse are left untouched.
|
|
84
|
+
|
|
85
|
+
## Use as a publish gate
|
|
86
|
+
|
|
87
|
+
With the [pre-commit](https://pre-commit.com) framework:
|
|
88
|
+
|
|
89
|
+
```yaml
|
|
90
|
+
repos:
|
|
91
|
+
- repo: https://github.com/pixelstrunk/watermark-cleaner
|
|
92
|
+
rev: v0.2.0
|
|
93
|
+
hooks:
|
|
94
|
+
- id: watermark-cleaner-fix
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
`watermark-cleaner-fix` cleans the mechanical layers on every commit and never blocks on style. Add `id: watermark-cleaner-check` if you also want commits blocked on AI phrases. A plain git hook and a deploy gate script live in [`integrations/`](integrations/).
|
|
98
|
+
|
|
99
|
+
Coding agents can drive the CLI through the skill in [`skills/watermark-cleaner/`](skills/watermark-cleaner/), which enforces an inspect-first workflow.
|
|
100
|
+
|
|
101
|
+
All writes are safe by construction: files are replaced atomically, writes through symlinks are refused, and files larger than `max_file_bytes` (default 256 MiB) are skipped instead of loaded into memory.
|
|
102
|
+
|
|
103
|
+
## The voice layer, honestly
|
|
104
|
+
|
|
105
|
+
Cleaning AI text falls into buckets, and this tool is precise about which bucket each action lives in.
|
|
106
|
+
|
|
107
|
+
| Layer | What | Result |
|
|
108
|
+
|---|---|---|
|
|
109
|
+
| Characters | Invisible and format characters, exotic spaces, homoglyphs, NFC normalization | Fixed, 100% verifiable |
|
|
110
|
+
| Typography | Smart quotes, ellipsis and bullet glyphs, em and en dashes | Fixed, 100% verifiable |
|
|
111
|
+
| File metadata | EXIF, XMP and C2PA in images, metadata and comments in SVG | Stripped losslessly, 100% verifiable |
|
|
112
|
+
| Voice | AI filler phrases and AI sentence shapes from a portable rulebook | Filler removed, shapes flagged and blocked |
|
|
113
|
+
|
|
114
|
+
The voice layer auto-deletes only phrases that are pure filler ("without further ado"). When a deletion opens a sentence, the sentence is repaired: leftover spaces go away and the next word is capitalized. Phrases that carry an object ("let's explore the API") are never cut mid-sentence; they are flagged as errors for a human to rewrite. The phrase rulebook is currently English only.
|
|
115
|
+
|
|
116
|
+
## Non-goals, stated plainly
|
|
117
|
+
|
|
118
|
+
- It does not remove statistical text watermarks (the SynthID style signal that some vendors embed in word choice). No deterministic tool can, and there is no public detector to verify removal. Only substantial rewriting degrades that signal.
|
|
119
|
+
- It does not auto-rewrite AI sentence shapes. Rewriting a sentence needs judgment, so the tool detects and blocks those shapes instead of replacing them and producing nonsense.
|
|
120
|
+
- It does not certify that text will pass any AI detector, and it is not a way to misrepresent authorship.
|
|
121
|
+
|
|
122
|
+
## Configure
|
|
123
|
+
|
|
124
|
+
Drop a `watermark-cleaner.config.json` in a project root. Any key overrides the default.
|
|
125
|
+
|
|
126
|
+
```json
|
|
127
|
+
{
|
|
128
|
+
"voice": true,
|
|
129
|
+
"replace_homoglyphs": false,
|
|
130
|
+
"keep_nbsp_in_numbers": true,
|
|
131
|
+
"protect_code": true,
|
|
132
|
+
"strip_icc": false,
|
|
133
|
+
"dash_policy": { "spaced_replacement": " - ", "unspaced_replacement": "-" },
|
|
134
|
+
"custom_banned_phrases": ["our innovative product", "leverage synergies"],
|
|
135
|
+
"ignore_phrases": ["delve", "not-only-but-also"],
|
|
136
|
+
"text_extensions": [".md", ".mdx", ".txt"],
|
|
137
|
+
"exclude": ["node_modules", ".git", "dist"]
|
|
138
|
+
}
|
|
139
|
+
```
|
|
140
|
+
|
|
141
|
+
By default a spaced em dash ("fast — slow") becomes a comma ("fast, slow") and an unspaced one becomes a hyphen. That is an opinionated policy; `dash_policy` lets you change both replacements per project.
|
|
142
|
+
|
|
143
|
+
`custom_banned_phrases` adds your own blocking phrases on top of the rulebook. `ignore_phrases` switches off any built-in phrase, lexicon word or sentence shape id that does not fit your domain.
|
|
144
|
+
|
|
145
|
+
## Optional DeepL rewrite
|
|
146
|
+
|
|
147
|
+
```
|
|
148
|
+
export DEEPL_API_KEY=your-key
|
|
149
|
+
watermark-cleaner rewrite post.md
|
|
150
|
+
watermark-cleaner rewrite post.md --source-lang DE --pivot-lang EN
|
|
151
|
+
watermark-cleaner rewrite post.md --write
|
|
152
|
+
```
|
|
153
|
+
|
|
154
|
+
The document language is auto-detected by DeepL when `--source-lang` is not given.
|
|
155
|
+
|
|
156
|
+
This back translates the text (source to pivot and back) using a model that is not the one that wrote the text, then runs the deterministic cleaner on the result. It changes word choice, which is what degrades a statistical watermark, but it can shift meaning and it sends your text to DeepL. It is off by default and opt in. Do not use it on confidential content. It cannot certify that any vendor detector will fail.
|
|
157
|
+
|
|
158
|
+
## Rules are one source
|
|
159
|
+
|
|
160
|
+
All character lists and phrase lists live in [`rules/*.json`](rules/). The Python and Node packages read the same files, `scripts/sync-rules.sh` copies them into each package, and CI fails when they drift or when the two CLIs produce different output. Edit the rules once, both tools follow. Contributions to the rulebook are the easiest way to help; see [CONTRIBUTING.md](CONTRIBUTING.md).
|
|
161
|
+
|
|
162
|
+
## Scope and intent
|
|
163
|
+
|
|
164
|
+
This tool is for hygiene, privacy and transparency on content you own: see what is hidden in your text, and publish without machine fingerprints you did not choose to include.
|
|
165
|
+
|
|
166
|
+
## License
|
|
167
|
+
|
|
168
|
+
[MIT](LICENSE)
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=68"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "watermark-cleaner"
|
|
7
|
+
version = "0.2.1"
|
|
8
|
+
description = "Remove AI text artifacts, AI phrases and file metadata before you publish. Detects and strips invisible unicode characters, losslessly cleans image metadata."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.9"
|
|
11
|
+
license = { text = "MIT" }
|
|
12
|
+
authors = [{ name = "Christian Strunk" }]
|
|
13
|
+
keywords = ["ai", "watermark", "unicode", "text-cleaning", "metadata", "publishing", "pre-commit", "trojan-source"]
|
|
14
|
+
classifiers = [
|
|
15
|
+
"Programming Language :: Python :: 3",
|
|
16
|
+
"Programming Language :: Python :: 3.9",
|
|
17
|
+
"Programming Language :: Python :: 3.10",
|
|
18
|
+
"Programming Language :: Python :: 3.11",
|
|
19
|
+
"Programming Language :: Python :: 3.12",
|
|
20
|
+
"Programming Language :: Python :: 3.13",
|
|
21
|
+
"License :: OSI Approved :: MIT License",
|
|
22
|
+
"Operating System :: OS Independent",
|
|
23
|
+
"Environment :: Console",
|
|
24
|
+
"Intended Audience :: Developers",
|
|
25
|
+
"Topic :: Text Processing :: Filters",
|
|
26
|
+
]
|
|
27
|
+
|
|
28
|
+
[project.scripts]
|
|
29
|
+
watermark-cleaner = "watermark_cleaner.cli:main"
|
|
30
|
+
|
|
31
|
+
[project.urls]
|
|
32
|
+
Homepage = "https://github.com/pixelstrunk/watermark-cleaner"
|
|
33
|
+
Issues = "https://github.com/pixelstrunk/watermark-cleaner/issues"
|
|
34
|
+
Changelog = "https://github.com/pixelstrunk/watermark-cleaner/blob/main/CHANGELOG.md"
|
|
35
|
+
|
|
36
|
+
[tool.setuptools]
|
|
37
|
+
packages = ["watermark_cleaner", "watermark_cleaner.layers", "watermark_cleaner.rewrite"]
|
|
38
|
+
|
|
39
|
+
[tool.setuptools.package-dir]
|
|
40
|
+
watermark_cleaner = "python/watermark_cleaner"
|
|
41
|
+
"watermark_cleaner.layers" = "python/watermark_cleaner/layers"
|
|
42
|
+
"watermark_cleaner.rewrite" = "python/watermark_cleaner/rewrite"
|
|
43
|
+
|
|
44
|
+
[tool.setuptools.package-data]
|
|
45
|
+
watermark_cleaner = ["rules_data/*.json"]
|
|
@@ -0,0 +1,109 @@
|
|
|
1
|
+
import argparse
|
|
2
|
+
import json
|
|
3
|
+
import sys
|
|
4
|
+
|
|
5
|
+
from . import __version__
|
|
6
|
+
from .config import apply_aggressive, load_config
|
|
7
|
+
from .report import render_report, render_summary
|
|
8
|
+
from .runner import run
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def build_parser():
|
|
12
|
+
parser = argparse.ArgumentParser(
|
|
13
|
+
prog="watermark-cleaner",
|
|
14
|
+
description="Watermark Cleaner: remove ai text artifacts, ai phrases and file metadata.",
|
|
15
|
+
)
|
|
16
|
+
parser.add_argument("--version", action="version", version=f"watermark-cleaner {__version__}")
|
|
17
|
+
sub = parser.add_subparsers(dest="command", required=True)
|
|
18
|
+
|
|
19
|
+
for name in ("check", "fix"):
|
|
20
|
+
p = sub.add_parser(name, help=f"{name} files or folders")
|
|
21
|
+
p.add_argument("paths", nargs="*", default=["."], help="files or folders (default: .)")
|
|
22
|
+
p.add_argument("--config", help="path to a watermark-cleaner config file")
|
|
23
|
+
p.add_argument("--json", action="store_true", help="emit json report")
|
|
24
|
+
p.add_argument("--no-voice", action="store_true", help="disable the voice/ai-phrase layer")
|
|
25
|
+
p.add_argument("--aggressive", action="store_true", help="also replace homoglyphs and strip variation selectors")
|
|
26
|
+
p.add_argument("--no-backup", action="store_true", help="do not write .bak backups (use inside git hooks)")
|
|
27
|
+
p.add_argument("--quiet", action="store_true", help="only print the summary line")
|
|
28
|
+
p.add_argument("--strict", action="store_true", help="fix only: exit non-zero when blocking findings remain")
|
|
29
|
+
|
|
30
|
+
rewrite = sub.add_parser("rewrite", help="optional deepl rewrite (opt-in, sends text to deepl)")
|
|
31
|
+
rewrite.add_argument("paths", nargs="+", help="text files to rewrite")
|
|
32
|
+
rewrite.add_argument("--source-lang", default=None, help="document language (default: auto-detect via deepl)")
|
|
33
|
+
rewrite.add_argument("--pivot-lang", default="EN", help="intermediate language for back-translation (default EN)")
|
|
34
|
+
rewrite.add_argument("--write", action="store_true", help="write result back (default prints)")
|
|
35
|
+
rewrite.add_argument("--config", help="path to a watermark-cleaner config file")
|
|
36
|
+
return parser
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _resolve_config(args):
|
|
40
|
+
start = args.paths[0] if getattr(args, "paths", None) else "."
|
|
41
|
+
config = load_config(getattr(args, "config", None), start)
|
|
42
|
+
if getattr(args, "no_voice", False):
|
|
43
|
+
config["voice"] = False
|
|
44
|
+
if getattr(args, "aggressive", False):
|
|
45
|
+
config = apply_aggressive(config)
|
|
46
|
+
if getattr(args, "no_backup", False):
|
|
47
|
+
config["backup"] = False
|
|
48
|
+
return config
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def _run_scan(args, write):
|
|
52
|
+
config = _resolve_config(args)
|
|
53
|
+
paths = args.paths or ["."]
|
|
54
|
+
reports = run(paths, config, write=write)
|
|
55
|
+
|
|
56
|
+
if args.json:
|
|
57
|
+
payload = {
|
|
58
|
+
"mode": "fix" if write else "check",
|
|
59
|
+
"summary": render_summary(reports),
|
|
60
|
+
"reports": [r.to_dict() for r in reports],
|
|
61
|
+
}
|
|
62
|
+
print(json.dumps(payload, ensure_ascii=False, indent=2))
|
|
63
|
+
else:
|
|
64
|
+
if not args.quiet:
|
|
65
|
+
for report in reports:
|
|
66
|
+
if report.findings:
|
|
67
|
+
print(render_report(report))
|
|
68
|
+
print()
|
|
69
|
+
print(render_summary(reports))
|
|
70
|
+
|
|
71
|
+
blocking = any(r.has_blocking for r in reports)
|
|
72
|
+
if blocking and (not write or getattr(args, "strict", False)):
|
|
73
|
+
return 1
|
|
74
|
+
return 0
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _run_rewrite(args):
|
|
78
|
+
from .rewrite.deepl import rewrite_file
|
|
79
|
+
|
|
80
|
+
config = load_config(getattr(args, "config", None), args.paths[0])
|
|
81
|
+
code = 0
|
|
82
|
+
for path in args.paths:
|
|
83
|
+
try:
|
|
84
|
+
result = rewrite_file(path, target_lang=args.source_lang, pivot_lang=args.pivot_lang, write=args.write, config=config)
|
|
85
|
+
if args.write:
|
|
86
|
+
print(f"rewritten {path}")
|
|
87
|
+
else:
|
|
88
|
+
print(result)
|
|
89
|
+
except Exception as error:
|
|
90
|
+
print(f"error {path}: {error}", file=sys.stderr)
|
|
91
|
+
code = 1
|
|
92
|
+
return code
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def main(argv=None):
|
|
96
|
+
parser = build_parser()
|
|
97
|
+
args = parser.parse_args(argv)
|
|
98
|
+
if args.command == "check":
|
|
99
|
+
return _run_scan(args, write=False)
|
|
100
|
+
if args.command == "fix":
|
|
101
|
+
return _run_scan(args, write=True)
|
|
102
|
+
if args.command == "rewrite":
|
|
103
|
+
return _run_rewrite(args)
|
|
104
|
+
parser.print_help()
|
|
105
|
+
return 1
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
if __name__ == "__main__":
|
|
109
|
+
sys.exit(main())
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
import json
|
|
2
|
+
from pathlib import Path
|
|
3
|
+
|
|
4
|
+
DEFAULTS = {
|
|
5
|
+
"strip_variation_selectors": False,
|
|
6
|
+
"keep_nbsp_in_numbers": True,
|
|
7
|
+
"normalize_form": "NFC",
|
|
8
|
+
"straight_quotes": True,
|
|
9
|
+
"fix_punctuation": True,
|
|
10
|
+
"fix_dashes": True,
|
|
11
|
+
"replace_homoglyphs": False,
|
|
12
|
+
"fix_safe_delete_phrases": True,
|
|
13
|
+
"voice": True,
|
|
14
|
+
"protect_code": True,
|
|
15
|
+
"strip_icc": False,
|
|
16
|
+
"custom_banned_phrases": [],
|
|
17
|
+
"ignore_phrases": [],
|
|
18
|
+
"layers": ["characters", "homoglyphs", "typography", "voice"],
|
|
19
|
+
"text_extensions": [".md", ".mdx", ".txt", ".html", ".htm", ".markdown"],
|
|
20
|
+
"image_extensions": [".png", ".jpg", ".jpeg", ".svg", ".webp", ".gif", ".tif", ".tiff"],
|
|
21
|
+
"exclude": ["node_modules", ".git", "dist", "build", ".next", ".venv", "__pycache__"],
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
CONFIG_NAMES = ["watermark-cleaner.config.json", ".watermark-cleanerrc.json", ".watermark-cleanerrc"]
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def load_config(explicit_path=None, start_dir="."):
|
|
28
|
+
config = dict(DEFAULTS)
|
|
29
|
+
path = _find_config_file(explicit_path, start_dir)
|
|
30
|
+
if path is not None:
|
|
31
|
+
with open(path, "r", encoding="utf-8") as handle:
|
|
32
|
+
user = json.load(handle)
|
|
33
|
+
config.update(user)
|
|
34
|
+
return config
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def _find_config_file(explicit_path, start_dir):
|
|
38
|
+
if explicit_path:
|
|
39
|
+
p = Path(explicit_path)
|
|
40
|
+
if not p.is_file():
|
|
41
|
+
raise FileNotFoundError(f"config not found: {explicit_path}")
|
|
42
|
+
return p
|
|
43
|
+
base = Path(start_dir).resolve()
|
|
44
|
+
for directory in [base, *base.parents]:
|
|
45
|
+
for name in CONFIG_NAMES:
|
|
46
|
+
candidate = directory / name
|
|
47
|
+
if candidate.is_file():
|
|
48
|
+
return candidate
|
|
49
|
+
return None
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def apply_aggressive(config):
|
|
53
|
+
config = dict(config)
|
|
54
|
+
config["replace_homoglyphs"] = True
|
|
55
|
+
config["strip_variation_selectors"] = True
|
|
56
|
+
config["strip_icc"] = True
|
|
57
|
+
return config
|