url-normalize 2.2.0__tar.gz → 3.0.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (61) hide show
  1. url_normalize-3.0.1/PKG-INFO +215 -0
  2. url_normalize-3.0.1/README.md +184 -0
  3. {url_normalize-2.2.0 → url_normalize-3.0.1}/pyproject.toml +18 -19
  4. {url_normalize-2.2.0 → url_normalize-3.0.1}/tests/test_cli.py +83 -1
  5. url_normalize-3.0.1/tests/test_deconstruct_url.py +87 -0
  6. url_normalize-3.0.1/tests/test_generic_url_cleanup.py +72 -0
  7. url_normalize-3.0.1/tests/test_normalize_fragment.py +23 -0
  8. url_normalize-3.0.1/tests/test_normalize_host.py +58 -0
  9. url_normalize-3.0.1/tests/test_normalize_path.py +58 -0
  10. url_normalize-3.0.1/tests/test_normalize_port.py +26 -0
  11. url_normalize-3.0.1/tests/test_normalize_query.py +57 -0
  12. url_normalize-3.0.1/tests/test_normalize_query_filters.py +226 -0
  13. url_normalize-3.0.1/tests/test_normalize_scheme.py +18 -0
  14. url_normalize-3.0.1/tests/test_normalize_userinfo.py +44 -0
  15. url_normalize-3.0.1/tests/test_provide_url_scheme.py +140 -0
  16. url_normalize-3.0.1/tests/test_reconstruct_url.py +42 -0
  17. url_normalize-3.0.1/tests/test_url_humanize.py +165 -0
  18. {url_normalize-2.2.0 → url_normalize-3.0.1}/tests/test_url_normalize.py +45 -1
  19. {url_normalize-2.2.0 → url_normalize-3.0.1}/url_normalize/__init__.py +3 -2
  20. {url_normalize-2.2.0 → url_normalize-3.0.1}/url_normalize/cli.py +21 -4
  21. url_normalize-3.0.1/url_normalize/generic_url_cleanup.py +27 -0
  22. {url_normalize-2.2.0 → url_normalize-3.0.1}/url_normalize/normalize_host.py +6 -7
  23. {url_normalize-2.2.0 → url_normalize-3.0.1}/url_normalize/normalize_path.py +2 -2
  24. {url_normalize-2.2.0 → url_normalize-3.0.1}/url_normalize/normalize_query.py +9 -8
  25. {url_normalize-2.2.0 → url_normalize-3.0.1}/url_normalize/normalize_userinfo.py +9 -1
  26. {url_normalize-2.2.0 → url_normalize-3.0.1}/url_normalize/param_allowlist.py +23 -7
  27. {url_normalize-2.2.0 → url_normalize-3.0.1}/url_normalize/provide_url_scheme.py +12 -2
  28. url_normalize-3.0.1/url_normalize/py.typed +0 -0
  29. {url_normalize-2.2.0 → url_normalize-3.0.1}/url_normalize/tools.py +31 -7
  30. url_normalize-3.0.1/url_normalize/url_humanize.py +139 -0
  31. {url_normalize-2.2.0 → url_normalize-3.0.1}/url_normalize/url_normalize.py +8 -4
  32. url_normalize-3.0.1/url_normalize.egg-info/PKG-INFO +215 -0
  33. {url_normalize-2.2.0 → url_normalize-3.0.1}/url_normalize.egg-info/SOURCES.txt +3 -0
  34. {url_normalize-2.2.0 → url_normalize-3.0.1}/url_normalize.egg-info/requires.txt +0 -2
  35. url_normalize-2.2.0/PKG-INFO +0 -150
  36. url_normalize-2.2.0/README.md +0 -124
  37. url_normalize-2.2.0/tests/test_deconstruct_url.py +0 -35
  38. url_normalize-2.2.0/tests/test_generic_url_cleanup.py +0 -21
  39. url_normalize-2.2.0/tests/test_normalize_fragment.py +0 -21
  40. url_normalize-2.2.0/tests/test_normalize_host.py +0 -29
  41. url_normalize-2.2.0/tests/test_normalize_path.py +0 -42
  42. url_normalize-2.2.0/tests/test_normalize_port.py +0 -13
  43. url_normalize-2.2.0/tests/test_normalize_query.py +0 -24
  44. url_normalize-2.2.0/tests/test_normalize_query_filters.py +0 -98
  45. url_normalize-2.2.0/tests/test_normalize_scheme.py +0 -13
  46. url_normalize-2.2.0/tests/test_normalize_userinfo.py +0 -19
  47. url_normalize-2.2.0/tests/test_provide_url_scheme.py +0 -30
  48. url_normalize-2.2.0/tests/test_reconstruct_url.py +0 -41
  49. url_normalize-2.2.0/url_normalize/generic_url_cleanup.py +0 -19
  50. url_normalize-2.2.0/url_normalize.egg-info/PKG-INFO +0 -150
  51. {url_normalize-2.2.0 → url_normalize-3.0.1}/LICENSE +0 -0
  52. {url_normalize-2.2.0 → url_normalize-3.0.1}/setup.cfg +0 -0
  53. {url_normalize-2.2.0 → url_normalize-3.0.1}/tests/test_provide_url_domain.py +0 -0
  54. {url_normalize-2.2.0 → url_normalize-3.0.1}/tests/test_tools.py +0 -0
  55. {url_normalize-2.2.0 → url_normalize-3.0.1}/url_normalize/normalize_fragment.py +0 -0
  56. {url_normalize-2.2.0 → url_normalize-3.0.1}/url_normalize/normalize_port.py +0 -0
  57. {url_normalize-2.2.0 → url_normalize-3.0.1}/url_normalize/normalize_scheme.py +0 -0
  58. {url_normalize-2.2.0 → url_normalize-3.0.1}/url_normalize/provide_url_domain.py +0 -0
  59. {url_normalize-2.2.0 → url_normalize-3.0.1}/url_normalize.egg-info/dependency_links.txt +0 -0
  60. {url_normalize-2.2.0 → url_normalize-3.0.1}/url_normalize.egg-info/entry_points.txt +0 -0
  61. {url_normalize-2.2.0 → url_normalize-3.0.1}/url_normalize.egg-info/top_level.txt +0 -0
@@ -0,0 +1,215 @@
1
+ Metadata-Version: 2.4
2
+ Name: url-normalize
3
+ Version: 3.0.1
4
+ Summary: URL normalization for Python
5
+ Author-email: Nikolay Panov <github@npanov.com>
6
+ License-Expression: MIT
7
+ Project-URL: Homepage, https://github.com/niksite/url-normalize
8
+ Project-URL: Repository, https://github.com/niksite/url-normalize
9
+ Project-URL: Issues, https://github.com/niksite/url-normalize/issues
10
+ Project-URL: Changelog, https://github.com/niksite/url-normalize/blob/master/CHANGELOG.md
11
+ Keywords: url,normalization,normalize,normalizer
12
+ Classifier: Programming Language :: Python :: 3
13
+ Classifier: Programming Language :: Python :: 3 :: Only
14
+ Classifier: Programming Language :: Python :: 3.10
15
+ Classifier: Programming Language :: Python :: 3.11
16
+ Classifier: Programming Language :: Python :: 3.12
17
+ Classifier: Programming Language :: Python :: 3.13
18
+ Classifier: Programming Language :: Python :: 3.14
19
+ Requires-Python: >=3.10
20
+ Description-Content-Type: text/markdown
21
+ License-File: LICENSE
22
+ Requires-Dist: idna>=3.3
23
+ Provides-Extra: dev
24
+ Requires-Dist: mypy; extra == "dev"
25
+ Requires-Dist: pre-commit; extra == "dev"
26
+ Requires-Dist: pytest-cov; extra == "dev"
27
+ Requires-Dist: pytest-socket; extra == "dev"
28
+ Requires-Dist: pytest; extra == "dev"
29
+ Requires-Dist: ruff; extra == "dev"
30
+ Dynamic: license-file
31
+
32
+ # url-normalize
33
+
34
+ [![tests](https://github.com/niksite/url-normalize/actions/workflows/ci.yml/badge.svg)](https://github.com/niksite/url-normalize/actions/workflows/ci.yml)
35
+ [![Coveralls](https://img.shields.io/coveralls/github/niksite/url-normalize/master.svg)](https://coveralls.io/r/niksite/url-normalize)
36
+ [![PyPI](https://img.shields.io/pypi/v/url-normalize.svg)](https://pypi.org/project/url-normalize/)
37
+ [![Python Versions](https://img.shields.io/pypi/pyversions/url-normalize.svg)](https://pypi.org/project/url-normalize/)
38
+ [![License](https://img.shields.io/pypi/l/url-normalize.svg)](https://github.com/niksite/url-normalize/blob/master/LICENSE)
39
+ [![Ruff](https://img.shields.io/endpoint?url=https://raw.githubusercontent.com/astral-sh/ruff/main/assets/badge/v2.json)](https://github.com/astral-sh/ruff)
40
+
41
+ A Python library for standardizing and normalizing URLs. Ideal for database deduplication, caching, web crawling, and anywhere you need to ensure that equivalent URLs resolve to the exact same string.
42
+
43
+ ```python
44
+ from url_normalize import url_normalize
45
+
46
+ # Fixes IDN, lowercases host/scheme, removes default ports, resolves path segments
47
+ url_normalize("HTTP://User:Pass@www.FOO.com:80///foo/../bar/./baz?q=1#frag")
48
+ # -> 'http://User:Pass@www.foo.com/bar/baz?q=1#frag'
49
+ ```
50
+
51
+ ## Features
52
+
53
+ url-normalize provides a robust URI normalization function that handles IDN domains, scheme/host lowercasing, and RFC-compliant path normalization.
54
+
55
+ - **IDN Support**: Full internationalized domain name handling (using IDNA2008 with UTS46).
56
+ - **Humanization**: Convert normalized URLs to a readable display format while preserving round-trip normalization.
57
+ - **RFC Compliance**:
58
+ - Proper percent-encoding (minimal, uppercase hex).
59
+ - Dot-segment removal in paths.
60
+ - Default port and authority handling.
61
+ - UTF-8 NFC normalization.
62
+ - **Configurable Defaults**:
63
+ - Customizable default scheme (https by default).
64
+ - Configurable default domain for absolute paths.
65
+ - **Query Parameter Control**:
66
+ - Parameter filtering with allowlists.
67
+ - Support for domain-specific parameter rules.
68
+ - **Versatile URL Handling**: Handles empty strings, double-slash URLs (//domain.tld), and shebang (#!) URLs.
69
+ - **Developer Friendly**:
70
+ - Python 3.10+ compatibility.
71
+ - 100% statement coverage, enforced by the test suite.
72
+ - Modern type hints and string handling.
73
+
74
+ Inspired by Sam Ruby's [urlnorm.py](http://intertwingly.net/blog/2004/08/04/Urlnorm).
75
+
76
+ ## Installation
77
+
78
+ Install as a library:
79
+
80
+ ```sh
81
+ pip install url-normalize
82
+ ```
83
+
84
+ Or install as a standalone CLI tool using [uv](https://docs.astral.sh/uv/):
85
+
86
+ ```sh
87
+ uv tool install url-normalize
88
+ ```
89
+
90
+ ## Usage
91
+
92
+ ### Python API
93
+
94
+ #### Basic Normalization
95
+
96
+ ```python
97
+ from url_normalize import url_normalize
98
+
99
+ # Basic normalization (uses https by default)
100
+ print(url_normalize("www.foo.com:80/foo"))
101
+ # Output: https://www.foo.com:80/foo
102
+
103
+ # With custom default scheme
104
+ print(url_normalize("www.foo.com/foo", default_scheme="http"))
105
+ # Output: http://www.foo.com/foo
106
+ ```
107
+
108
+ The `charset` argument remains for compatibility. Unicode characters use UTF-8 percent encoding, regardless of this argument.
109
+
110
+ #### Query Parameter Filtering
111
+
112
+ With `filter_params=True`, normalization retains only allowlisted query parameters. The built-in rules cover selected domains.
113
+
114
+ Other domains have an empty default allowlist. Without a custom allowlist, filtering removes all query parameters from those domains.
115
+
116
+ ```python
117
+ # With the built-in Google allowlist
118
+ print(url_normalize("www.google.com/search?q=test&utm_source=test", filter_params=True))
119
+ # Output: https://www.google.com/search?q=test
120
+
121
+ # With custom parameter allowlist as a list
122
+ print(url_normalize(
123
+ "example.com?page=1&id=123&ref=test",
124
+ filter_params=True,
125
+ param_allowlist=["page", "id"]
126
+ ))
127
+ # Output: https://example.com/?page=1&id=123
128
+
129
+ # With domain-specific parameter allowlists
130
+ print(url_normalize(
131
+ "example.com?page=1&id=123&ref=test",
132
+ filter_params=True,
133
+ param_allowlist={"example.com": ["page", "id"]}
134
+ ))
135
+ # Output: https://example.com/?page=1&id=123
136
+ ```
137
+
138
+ #### Default Domain & Scheme
139
+
140
+ Useful for resolving relative URLs found on a specific page.
141
+
142
+ ```python
143
+ # With default domain for absolute paths
144
+ print(url_normalize("/images/logo.png", default_domain="example.com"))
145
+ # Output: https://example.com/images/logo.png
146
+
147
+ # With default domain and custom scheme
148
+ print(url_normalize("/images/logo.png", default_scheme="http", default_domain="example.com"))
149
+ # Output: http://example.com/images/logo.png
150
+ ```
151
+
152
+ #### Humanizing URLs
153
+
154
+ Convert normalized URLs back into a user-friendly format for display, particularly useful for IDN domains and percent-encoded paths.
155
+
156
+ ```python
157
+ from url_normalize import url_humanize
158
+
159
+ # Human-readable display form that still normalizes back to the same URL
160
+ print(url_humanize("https://xn--e1afmkfd.xn--80akhbyknj4f/%D0%A1%D0%BB%D1%83%D0%B6%D0%B5%D0%B1%D0%BD%D0%B0%D1%8F"))
161
+ # Output: https://пример.испытание/Служебная
162
+
163
+ # Humanization accepts the same normalization options
164
+ print(url_humanize("/%D0%A1%D0%BB%D1%83%D0%B6%D0%B5%D0%B1%D0%BD%D0%B0%D1%8F", default_domain="xn--e1afmkfd.xn--80akhbyknj4f"))
165
+ # Output: https://пример.испытание/Служебная
166
+ ```
167
+
168
+ ### Command-line Usage
169
+
170
+ You can also use `url-normalize` directly from the terminal to process URLs.
171
+
172
+ ```bash
173
+ $ url-normalize "www.foo.com:80/foo"
174
+ # Output: https://www.foo.com:80/foo
175
+
176
+ # With custom default scheme
177
+ $ url-normalize -s http "www.foo.com/foo"
178
+ # Output: http://www.foo.com/foo
179
+
180
+ # With query parameter filtering
181
+ $ url-normalize -f "www.google.com/search?q=test&utm_source=test"
182
+ # Output: https://www.google.com/search?q=test
183
+
184
+ # With custom allowlist
185
+ $ url-normalize -f -p page,id "example.com?page=1&id=123&ref=test"
186
+ # Output: https://example.com/?page=1&id=123
187
+
188
+ # With default domain for absolute paths
189
+ $ url-normalize -d example.com "/images/logo.png"
190
+ # Output: https://example.com/images/logo.png
191
+
192
+ # With default domain and custom scheme
193
+ $ url-normalize -d example.com -s http "/images/logo.png"
194
+ # Output: http://example.com/images/logo.png
195
+
196
+ # Human-readable display form
197
+ $ url-normalize -H "https://xn--e1afmkfd.xn--80akhbyknj4f/%D0%A1%D0%BB%D1%83%D0%B6%D0%B5%D0%B1%D0%BD%D0%B0%D1%8F"
198
+ # Output: https://пример.испытание/Служебная
199
+
200
+ # Via uv tool/uvx
201
+ $ uvx url-normalize www.foo.com:80/foo
202
+ # Output: https://www.foo.com:80/foo
203
+ ```
204
+
205
+ ## Documentation
206
+
207
+ For a complete history of changes, see [CHANGELOG.md](CHANGELOG.md).
208
+
209
+ ## Contributing
210
+
211
+ Contributions are welcome! Please feel free to submit a Pull Request.
212
+
213
+ ## License
214
+
215
+ MIT License
@@ -0,0 +1,184 @@
1
+ # url-normalize
2
+
3
+ [![tests](https://github.com/niksite/url-normalize/actions/workflows/ci.yml/badge.svg)](https://github.com/niksite/url-normalize/actions/workflows/ci.yml)
4
+ [![Coveralls](https://img.shields.io/coveralls/github/niksite/url-normalize/master.svg)](https://coveralls.io/r/niksite/url-normalize)
5
+ [![PyPI](https://img.shields.io/pypi/v/url-normalize.svg)](https://pypi.org/project/url-normalize/)
6
+ [![Python Versions](https://img.shields.io/pypi/pyversions/url-normalize.svg)](https://pypi.org/project/url-normalize/)
7
+ [![License](https://img.shields.io/pypi/l/url-normalize.svg)](https://github.com/niksite/url-normalize/blob/master/LICENSE)
8
+ [![Ruff](https://img.shields.io/endpoint?url=https://raw.githubusercontent.com/astral-sh/ruff/main/assets/badge/v2.json)](https://github.com/astral-sh/ruff)
9
+
10
+ A Python library for standardizing and normalizing URLs. Ideal for database deduplication, caching, web crawling, and anywhere you need to ensure that equivalent URLs resolve to the exact same string.
11
+
12
+ ```python
13
+ from url_normalize import url_normalize
14
+
15
+ # Fixes IDN, lowercases host/scheme, removes default ports, resolves path segments
16
+ url_normalize("HTTP://User:Pass@www.FOO.com:80///foo/../bar/./baz?q=1#frag")
17
+ # -> 'http://User:Pass@www.foo.com/bar/baz?q=1#frag'
18
+ ```
19
+
20
+ ## Features
21
+
22
+ url-normalize provides a robust URI normalization function that handles IDN domains, scheme/host lowercasing, and RFC-compliant path normalization.
23
+
24
+ - **IDN Support**: Full internationalized domain name handling (using IDNA2008 with UTS46).
25
+ - **Humanization**: Convert normalized URLs to a readable display format while preserving round-trip normalization.
26
+ - **RFC Compliance**:
27
+ - Proper percent-encoding (minimal, uppercase hex).
28
+ - Dot-segment removal in paths.
29
+ - Default port and authority handling.
30
+ - UTF-8 NFC normalization.
31
+ - **Configurable Defaults**:
32
+ - Customizable default scheme (https by default).
33
+ - Configurable default domain for absolute paths.
34
+ - **Query Parameter Control**:
35
+ - Parameter filtering with allowlists.
36
+ - Support for domain-specific parameter rules.
37
+ - **Versatile URL Handling**: Handles empty strings, double-slash URLs (//domain.tld), and shebang (#!) URLs.
38
+ - **Developer Friendly**:
39
+ - Python 3.10+ compatibility.
40
+ - 100% statement coverage, enforced by the test suite.
41
+ - Modern type hints and string handling.
42
+
43
+ Inspired by Sam Ruby's [urlnorm.py](http://intertwingly.net/blog/2004/08/04/Urlnorm).
44
+
45
+ ## Installation
46
+
47
+ Install as a library:
48
+
49
+ ```sh
50
+ pip install url-normalize
51
+ ```
52
+
53
+ Or install as a standalone CLI tool using [uv](https://docs.astral.sh/uv/):
54
+
55
+ ```sh
56
+ uv tool install url-normalize
57
+ ```
58
+
59
+ ## Usage
60
+
61
+ ### Python API
62
+
63
+ #### Basic Normalization
64
+
65
+ ```python
66
+ from url_normalize import url_normalize
67
+
68
+ # Basic normalization (uses https by default)
69
+ print(url_normalize("www.foo.com:80/foo"))
70
+ # Output: https://www.foo.com:80/foo
71
+
72
+ # With custom default scheme
73
+ print(url_normalize("www.foo.com/foo", default_scheme="http"))
74
+ # Output: http://www.foo.com/foo
75
+ ```
76
+
77
+ The `charset` argument remains for compatibility. Unicode characters use UTF-8 percent encoding, regardless of this argument.
78
+
79
+ #### Query Parameter Filtering
80
+
81
+ With `filter_params=True`, normalization retains only allowlisted query parameters. The built-in rules cover selected domains.
82
+
83
+ Other domains have an empty default allowlist. Without a custom allowlist, filtering removes all query parameters from those domains.
84
+
85
+ ```python
86
+ # With the built-in Google allowlist
87
+ print(url_normalize("www.google.com/search?q=test&utm_source=test", filter_params=True))
88
+ # Output: https://www.google.com/search?q=test
89
+
90
+ # With custom parameter allowlist as a list
91
+ print(url_normalize(
92
+ "example.com?page=1&id=123&ref=test",
93
+ filter_params=True,
94
+ param_allowlist=["page", "id"]
95
+ ))
96
+ # Output: https://example.com/?page=1&id=123
97
+
98
+ # With domain-specific parameter allowlists
99
+ print(url_normalize(
100
+ "example.com?page=1&id=123&ref=test",
101
+ filter_params=True,
102
+ param_allowlist={"example.com": ["page", "id"]}
103
+ ))
104
+ # Output: https://example.com/?page=1&id=123
105
+ ```
106
+
107
+ #### Default Domain & Scheme
108
+
109
+ Useful for resolving relative URLs found on a specific page.
110
+
111
+ ```python
112
+ # With default domain for absolute paths
113
+ print(url_normalize("/images/logo.png", default_domain="example.com"))
114
+ # Output: https://example.com/images/logo.png
115
+
116
+ # With default domain and custom scheme
117
+ print(url_normalize("/images/logo.png", default_scheme="http", default_domain="example.com"))
118
+ # Output: http://example.com/images/logo.png
119
+ ```
120
+
121
+ #### Humanizing URLs
122
+
123
+ Convert normalized URLs back into a user-friendly format for display, particularly useful for IDN domains and percent-encoded paths.
124
+
125
+ ```python
126
+ from url_normalize import url_humanize
127
+
128
+ # Human-readable display form that still normalizes back to the same URL
129
+ print(url_humanize("https://xn--e1afmkfd.xn--80akhbyknj4f/%D0%A1%D0%BB%D1%83%D0%B6%D0%B5%D0%B1%D0%BD%D0%B0%D1%8F"))
130
+ # Output: https://пример.испытание/Служебная
131
+
132
+ # Humanization accepts the same normalization options
133
+ print(url_humanize("/%D0%A1%D0%BB%D1%83%D0%B6%D0%B5%D0%B1%D0%BD%D0%B0%D1%8F", default_domain="xn--e1afmkfd.xn--80akhbyknj4f"))
134
+ # Output: https://пример.испытание/Служебная
135
+ ```
136
+
137
+ ### Command-line Usage
138
+
139
+ You can also use `url-normalize` directly from the terminal to process URLs.
140
+
141
+ ```bash
142
+ $ url-normalize "www.foo.com:80/foo"
143
+ # Output: https://www.foo.com:80/foo
144
+
145
+ # With custom default scheme
146
+ $ url-normalize -s http "www.foo.com/foo"
147
+ # Output: http://www.foo.com/foo
148
+
149
+ # With query parameter filtering
150
+ $ url-normalize -f "www.google.com/search?q=test&utm_source=test"
151
+ # Output: https://www.google.com/search?q=test
152
+
153
+ # With custom allowlist
154
+ $ url-normalize -f -p page,id "example.com?page=1&id=123&ref=test"
155
+ # Output: https://example.com/?page=1&id=123
156
+
157
+ # With default domain for absolute paths
158
+ $ url-normalize -d example.com "/images/logo.png"
159
+ # Output: https://example.com/images/logo.png
160
+
161
+ # With default domain and custom scheme
162
+ $ url-normalize -d example.com -s http "/images/logo.png"
163
+ # Output: http://example.com/images/logo.png
164
+
165
+ # Human-readable display form
166
+ $ url-normalize -H "https://xn--e1afmkfd.xn--80akhbyknj4f/%D0%A1%D0%BB%D1%83%D0%B6%D0%B5%D0%B1%D0%BD%D0%B0%D1%8F"
167
+ # Output: https://пример.испытание/Служебная
168
+
169
+ # Via uv tool/uvx
170
+ $ uvx url-normalize www.foo.com:80/foo
171
+ # Output: https://www.foo.com:80/foo
172
+ ```
173
+
174
+ ## Documentation
175
+
176
+ For a complete history of changes, see [CHANGELOG.md](CHANGELOG.md).
177
+
178
+ ## Contributing
179
+
180
+ Contributions are welcome! Please feel free to submit a Pull Request.
181
+
182
+ ## License
183
+
184
+ MIT License
@@ -1,12 +1,21 @@
1
1
  [project]
2
2
  name = "url-normalize"
3
- version = "2.2.0"
3
+ version = "3.0.1"
4
4
  description = "URL normalization for Python"
5
5
  authors = [{ name = "Nikolay Panov", email = "github@npanov.com" }]
6
- license = { text = "MIT" }
6
+ license = "MIT"
7
7
  readme = "README.md"
8
- requires-python = ">=3.8"
9
- keywords = ["url", "normalization", "normalize"]
8
+ requires-python = ">=3.10"
9
+ keywords = ["url", "normalization", "normalize", "normalizer"]
10
+ classifiers = [
11
+ "Programming Language :: Python :: 3",
12
+ "Programming Language :: Python :: 3 :: Only",
13
+ "Programming Language :: Python :: 3.10",
14
+ "Programming Language :: Python :: 3.11",
15
+ "Programming Language :: Python :: 3.12",
16
+ "Programming Language :: Python :: 3.13",
17
+ "Programming Language :: Python :: 3.14",
18
+ ]
10
19
  dependencies = ["idna>=3.3"]
11
20
 
12
21
  [project.urls]
@@ -19,19 +28,10 @@ Changelog = "https://github.com/niksite/url-normalize/blob/master/CHANGELOG.md"
19
28
  url-normalize = "url_normalize.cli:main"
20
29
 
21
30
  [project.optional-dependencies]
22
- dev = [
23
- "mypy",
24
- "pre-commit",
25
- "pytest-cov",
26
- "pytest-ruff",
27
- "pytest-socket",
28
- "pytest",
29
- "ruff",
30
- "tox",
31
- ]
31
+ dev = ["mypy", "pre-commit", "pytest-cov", "pytest-socket", "pytest", "ruff"]
32
32
 
33
33
  [tool.ruff]
34
- target-version = "py38"
34
+ target-version = "py310"
35
35
  line-length = 88
36
36
  unsafe-fixes = true
37
37
 
@@ -62,20 +62,19 @@ indent-style = "space"
62
62
  [tool.mypy]
63
63
  ignore_missing_imports = true
64
64
  exclude = ["tests"]
65
- python_version = "3.8"
65
+ python_version = "3.10"
66
66
  show_error_codes = true
67
67
 
68
68
  [build-system]
69
- requires = ["setuptools>=42", "wheel"]
69
+ requires = ["setuptools>=77", "wheel"]
70
70
  build-backend = "setuptools.build_meta"
71
71
 
72
72
  [tool.pytest.ini_options]
73
73
  addopts = [
74
- "--cov-fail-under=100",
75
74
  "--cov-report=term-missing:skip-covered",
76
75
  "--cov=url_normalize",
76
+ "--cov-fail-under=100",
77
77
  "--disable-socket",
78
- "--ruff",
79
78
  "-v",
80
79
  ]
81
80
  python_files = ["tests.py", "test_*.py", "*_tests.py"]
@@ -22,10 +22,42 @@ def run_cli(*args: str) -> subprocess.CompletedProcess:
22
22
  """
23
23
  command = [sys.executable, "-m", "url_normalize.cli", *list(args)]
24
24
  return subprocess.run( # noqa: S603
25
- command, capture_output=True, text=True, check=False
25
+ command, capture_output=True, text=True, encoding="utf-8", check=False
26
26
  )
27
27
 
28
28
 
29
+ def test_cli_help_describes_allowlist_filtering() -> None:
30
+ """Describe filtering as an allowlist, not a tracking-parameter blocklist."""
31
+ result = run_cli("--help")
32
+ assert result.returncode == 0
33
+ help_text = " ".join(result.stdout.split())
34
+ assert "Keep only allowlisted query parameters." in help_text
35
+ assert "Unknown domains have an empty default allowlist." in help_text
36
+
37
+
38
+ def test_cli_help_describes_fixed_utf8_encoding() -> None:
39
+ """Describe charset as a compatibility option for Unicode URL input."""
40
+ result = run_cli("--help")
41
+ assert result.returncode == 0
42
+ help_text = " ".join(result.stdout.split())
43
+ assert "Retained for compatibility." in help_text
44
+ assert "Unicode characters use UTF-8 percent encoding." in help_text
45
+ normalized = run_cli("--charset", "iso-8859-1", "https://example.com/é")
46
+ assert normalized.returncode == 0
47
+ assert normalized.stdout.strip() == "https://example.com/%C3%A9"
48
+
49
+
50
+ def test_cli_successful_main(capsys, monkeypatch):
51
+ """Exercise successful CLI output in the measured process."""
52
+ monkeypatch.setattr(
53
+ sys, "argv", ["url-normalize", "http://EXAMPLE.com/./path/../other/"]
54
+ )
55
+ main()
56
+ captured = capsys.readouterr()
57
+ assert captured.out == "http://example.com/other/\n"
58
+ assert not captured.err
59
+
60
+
29
61
  def test_cli_error_handling(capsys, monkeypatch):
30
62
  """Test CLI error handling when URL normalization fails."""
31
63
  with patch("url_normalize.cli.url_normalize") as mock_normalize:
@@ -66,6 +98,56 @@ def test_cli_basic_normalization_short_args() -> None:
66
98
  assert not result.stderr
67
99
 
68
100
 
101
+ def test_cli_humanize() -> None:
102
+ """Test human-readable URL output via CLI."""
103
+ url = (
104
+ "https://xn--e1afmkfd.xn--80akhbyknj4f/"
105
+ "%D0%A1%D0%BB%D1%83%D0%B6%D0%B5%D0%B1%D0%BD%D0%B0%D1%8F"
106
+ )
107
+ expected = "https://пример.испытание/Служебная"
108
+
109
+ result = run_cli("--humanize", url)
110
+
111
+ assert result.returncode == 0
112
+ assert result.stdout.strip() == expected
113
+ assert not result.stderr
114
+
115
+
116
+ def test_cli_humanize_short_arg() -> None:
117
+ """Test human-readable URL output via CLI using short argument."""
118
+ url = (
119
+ "https://xn--e1afmkfd.xn--80akhbyknj4f/"
120
+ "%D0%A1%D0%BB%D1%83%D0%B6%D0%B5%D0%B1%D0%BD%D0%B0%D1%8F"
121
+ )
122
+ expected = "https://пример.испытание/Служебная"
123
+
124
+ result = run_cli("-H", url)
125
+
126
+ assert result.returncode == 0
127
+ assert result.stdout.strip() == expected
128
+ assert not result.stderr
129
+
130
+
131
+ def test_cli_humanize_respects_normalization_options() -> None:
132
+ """Test humanized output keeps the existing CLI normalization options."""
133
+ url = "/%D0%A1%D0%BB%D1%83%D0%B6%D0%B5%D0%B1%D0%BD%D0%B0%D1%8F"
134
+ expected = "https://пример.испытание/Служебная?keep=Ç"
135
+
136
+ result = run_cli(
137
+ "--humanize",
138
+ "--default-domain",
139
+ "xn--e1afmkfd.xn--80akhbyknj4f",
140
+ "--filter-params",
141
+ "--param-allowlist",
142
+ "keep",
143
+ f"{url}?utm_source=ignored&keep=%C3%87",
144
+ )
145
+
146
+ assert result.returncode == 0
147
+ assert result.stdout.strip() == expected
148
+ assert not result.stderr
149
+
150
+
69
151
  def test_cli_default_scheme() -> None:
70
152
  """Test default scheme addition via CLI."""
71
153
  url = "//example.com"
@@ -0,0 +1,87 @@
1
+ """Deconstruct url tests."""
2
+
3
+ import pytest
4
+
5
+ from url_normalize import url_normalize
6
+ from url_normalize.tools import URL, deconstruct_url
7
+
8
+
9
+ @pytest.mark.parametrize(
10
+ ("url", "expected"),
11
+ [
12
+ (
13
+ "http://site.com",
14
+ URL(
15
+ fragment="",
16
+ host="site.com",
17
+ path="",
18
+ port="",
19
+ query="",
20
+ scheme="http",
21
+ userinfo="",
22
+ ),
23
+ ),
24
+ (
25
+ "http://user@www.example.com:8080/path/index.html?param=val#fragment",
26
+ URL(
27
+ fragment="fragment",
28
+ host="www.example.com",
29
+ path="/path/index.html",
30
+ port="8080",
31
+ query="param=val",
32
+ scheme="http",
33
+ userinfo="user@",
34
+ ),
35
+ ),
36
+ ],
37
+ )
38
+ def test_deconstruct_url_result_is_expected(url: str, expected: URL) -> None:
39
+ """Assert we got expected results from the deconstruct_url function."""
40
+ result = deconstruct_url(url)
41
+ assert result == expected, url
42
+
43
+
44
+ @pytest.mark.parametrize(
45
+ ("authority", "userinfo", "host", "port", "normalized_authority"),
46
+ [
47
+ ("[2001:DB8::1]:080", "", "[2001:DB8::1]", "080", "[2001:db8::1]"),
48
+ ("[2001:DB8::1]:0081", "", "[2001:DB8::1]", "0081", "[2001:db8::1]:81"),
49
+ ("[::1]", "", "[::1]", "", "[::1]"),
50
+ ("[::1]:", "", "[::1]", "", "[::1]"),
51
+ ("user:pass@[::1]:080", "user:pass@", "[::1]", "080", "user:pass@[::1]"),
52
+ ("EXAMPLE.com:080", "", "EXAMPLE.com", "080", "example.com"),
53
+ ("127.0.0.1:0081", "", "127.0.0.1", "0081", "127.0.0.1:81"),
54
+ ],
55
+ )
56
+ def test_deconstruct_url_separates_bracketed_host_and_port(
57
+ authority, userinfo, host, port, normalized_authority
58
+ ):
59
+ """Parse the entire bracketed host before considering a port separator."""
60
+ url = f"http://{authority}/path?q=1#fragment"
61
+ assert deconstruct_url(url) == URL(
62
+ "http", userinfo, host, port, "/path", "q=1", "fragment"
63
+ )
64
+ assert url_normalize(url) == f"http://{normalized_authority}/path?q=1#fragment"
65
+
66
+
67
+ @pytest.mark.parametrize("whitespace", [" ", "\u00a0", " \t\r\n"])
68
+ @pytest.mark.parametrize("suffix", ["", "/path ", "?q=x ", "#part "])
69
+ @pytest.mark.parametrize(
70
+ ("authority", "normalized"),
71
+ [
72
+ ("EXAMPLE.com", "example.com"),
73
+ ("example.com:443", "example.com"),
74
+ ("user:pass@example.com:443", "user:pass@example.com"),
75
+ ("[FE80::1%25ethA]:443", "[fe80::1%25ethA]"),
76
+ ],
77
+ )
78
+ def test_deconstruct_url_trims_authority_whitespace_only(
79
+ authority, normalized, whitespace, suffix
80
+ ):
81
+ """Trim the authority without deleting whitespace from other components."""
82
+ value = f"https://{authority}{whitespace}{suffix}"
83
+ assert deconstruct_url(value) == deconstruct_url(f"https://{authority}{suffix}")
84
+ path_suffix = suffix if suffix.startswith("/") else "/" + suffix
85
+ assert (
86
+ url_normalize(value) == f"https://{normalized}{path_suffix.replace(' ', '%20')}"
87
+ )