url-normalize 2.2.0__tar.gz → 3.0.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- url_normalize-3.0.1/PKG-INFO +215 -0
- url_normalize-3.0.1/README.md +184 -0
- {url_normalize-2.2.0 → url_normalize-3.0.1}/pyproject.toml +18 -19
- {url_normalize-2.2.0 → url_normalize-3.0.1}/tests/test_cli.py +83 -1
- url_normalize-3.0.1/tests/test_deconstruct_url.py +87 -0
- url_normalize-3.0.1/tests/test_generic_url_cleanup.py +72 -0
- url_normalize-3.0.1/tests/test_normalize_fragment.py +23 -0
- url_normalize-3.0.1/tests/test_normalize_host.py +58 -0
- url_normalize-3.0.1/tests/test_normalize_path.py +58 -0
- url_normalize-3.0.1/tests/test_normalize_port.py +26 -0
- url_normalize-3.0.1/tests/test_normalize_query.py +57 -0
- url_normalize-3.0.1/tests/test_normalize_query_filters.py +226 -0
- url_normalize-3.0.1/tests/test_normalize_scheme.py +18 -0
- url_normalize-3.0.1/tests/test_normalize_userinfo.py +44 -0
- url_normalize-3.0.1/tests/test_provide_url_scheme.py +140 -0
- url_normalize-3.0.1/tests/test_reconstruct_url.py +42 -0
- url_normalize-3.0.1/tests/test_url_humanize.py +165 -0
- {url_normalize-2.2.0 → url_normalize-3.0.1}/tests/test_url_normalize.py +45 -1
- {url_normalize-2.2.0 → url_normalize-3.0.1}/url_normalize/__init__.py +3 -2
- {url_normalize-2.2.0 → url_normalize-3.0.1}/url_normalize/cli.py +21 -4
- url_normalize-3.0.1/url_normalize/generic_url_cleanup.py +27 -0
- {url_normalize-2.2.0 → url_normalize-3.0.1}/url_normalize/normalize_host.py +6 -7
- {url_normalize-2.2.0 → url_normalize-3.0.1}/url_normalize/normalize_path.py +2 -2
- {url_normalize-2.2.0 → url_normalize-3.0.1}/url_normalize/normalize_query.py +9 -8
- {url_normalize-2.2.0 → url_normalize-3.0.1}/url_normalize/normalize_userinfo.py +9 -1
- {url_normalize-2.2.0 → url_normalize-3.0.1}/url_normalize/param_allowlist.py +23 -7
- {url_normalize-2.2.0 → url_normalize-3.0.1}/url_normalize/provide_url_scheme.py +12 -2
- url_normalize-3.0.1/url_normalize/py.typed +0 -0
- {url_normalize-2.2.0 → url_normalize-3.0.1}/url_normalize/tools.py +31 -7
- url_normalize-3.0.1/url_normalize/url_humanize.py +139 -0
- {url_normalize-2.2.0 → url_normalize-3.0.1}/url_normalize/url_normalize.py +8 -4
- url_normalize-3.0.1/url_normalize.egg-info/PKG-INFO +215 -0
- {url_normalize-2.2.0 → url_normalize-3.0.1}/url_normalize.egg-info/SOURCES.txt +3 -0
- {url_normalize-2.2.0 → url_normalize-3.0.1}/url_normalize.egg-info/requires.txt +0 -2
- url_normalize-2.2.0/PKG-INFO +0 -150
- url_normalize-2.2.0/README.md +0 -124
- url_normalize-2.2.0/tests/test_deconstruct_url.py +0 -35
- url_normalize-2.2.0/tests/test_generic_url_cleanup.py +0 -21
- url_normalize-2.2.0/tests/test_normalize_fragment.py +0 -21
- url_normalize-2.2.0/tests/test_normalize_host.py +0 -29
- url_normalize-2.2.0/tests/test_normalize_path.py +0 -42
- url_normalize-2.2.0/tests/test_normalize_port.py +0 -13
- url_normalize-2.2.0/tests/test_normalize_query.py +0 -24
- url_normalize-2.2.0/tests/test_normalize_query_filters.py +0 -98
- url_normalize-2.2.0/tests/test_normalize_scheme.py +0 -13
- url_normalize-2.2.0/tests/test_normalize_userinfo.py +0 -19
- url_normalize-2.2.0/tests/test_provide_url_scheme.py +0 -30
- url_normalize-2.2.0/tests/test_reconstruct_url.py +0 -41
- url_normalize-2.2.0/url_normalize/generic_url_cleanup.py +0 -19
- url_normalize-2.2.0/url_normalize.egg-info/PKG-INFO +0 -150
- {url_normalize-2.2.0 → url_normalize-3.0.1}/LICENSE +0 -0
- {url_normalize-2.2.0 → url_normalize-3.0.1}/setup.cfg +0 -0
- {url_normalize-2.2.0 → url_normalize-3.0.1}/tests/test_provide_url_domain.py +0 -0
- {url_normalize-2.2.0 → url_normalize-3.0.1}/tests/test_tools.py +0 -0
- {url_normalize-2.2.0 → url_normalize-3.0.1}/url_normalize/normalize_fragment.py +0 -0
- {url_normalize-2.2.0 → url_normalize-3.0.1}/url_normalize/normalize_port.py +0 -0
- {url_normalize-2.2.0 → url_normalize-3.0.1}/url_normalize/normalize_scheme.py +0 -0
- {url_normalize-2.2.0 → url_normalize-3.0.1}/url_normalize/provide_url_domain.py +0 -0
- {url_normalize-2.2.0 → url_normalize-3.0.1}/url_normalize.egg-info/dependency_links.txt +0 -0
- {url_normalize-2.2.0 → url_normalize-3.0.1}/url_normalize.egg-info/entry_points.txt +0 -0
- {url_normalize-2.2.0 → url_normalize-3.0.1}/url_normalize.egg-info/top_level.txt +0 -0
|
@@ -0,0 +1,215 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: url-normalize
|
|
3
|
+
Version: 3.0.1
|
|
4
|
+
Summary: URL normalization for Python
|
|
5
|
+
Author-email: Nikolay Panov <github@npanov.com>
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/niksite/url-normalize
|
|
8
|
+
Project-URL: Repository, https://github.com/niksite/url-normalize
|
|
9
|
+
Project-URL: Issues, https://github.com/niksite/url-normalize/issues
|
|
10
|
+
Project-URL: Changelog, https://github.com/niksite/url-normalize/blob/master/CHANGELOG.md
|
|
11
|
+
Keywords: url,normalization,normalize,normalizer
|
|
12
|
+
Classifier: Programming Language :: Python :: 3
|
|
13
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
19
|
+
Requires-Python: >=3.10
|
|
20
|
+
Description-Content-Type: text/markdown
|
|
21
|
+
License-File: LICENSE
|
|
22
|
+
Requires-Dist: idna>=3.3
|
|
23
|
+
Provides-Extra: dev
|
|
24
|
+
Requires-Dist: mypy; extra == "dev"
|
|
25
|
+
Requires-Dist: pre-commit; extra == "dev"
|
|
26
|
+
Requires-Dist: pytest-cov; extra == "dev"
|
|
27
|
+
Requires-Dist: pytest-socket; extra == "dev"
|
|
28
|
+
Requires-Dist: pytest; extra == "dev"
|
|
29
|
+
Requires-Dist: ruff; extra == "dev"
|
|
30
|
+
Dynamic: license-file
|
|
31
|
+
|
|
32
|
+
# url-normalize
|
|
33
|
+
|
|
34
|
+
[](https://github.com/niksite/url-normalize/actions/workflows/ci.yml)
|
|
35
|
+
[](https://coveralls.io/r/niksite/url-normalize)
|
|
36
|
+
[](https://pypi.org/project/url-normalize/)
|
|
37
|
+
[](https://pypi.org/project/url-normalize/)
|
|
38
|
+
[](https://github.com/niksite/url-normalize/blob/master/LICENSE)
|
|
39
|
+
[](https://github.com/astral-sh/ruff)
|
|
40
|
+
|
|
41
|
+
A Python library for standardizing and normalizing URLs. Ideal for database deduplication, caching, web crawling, and anywhere you need to ensure that equivalent URLs resolve to the exact same string.
|
|
42
|
+
|
|
43
|
+
```python
|
|
44
|
+
from url_normalize import url_normalize
|
|
45
|
+
|
|
46
|
+
# Fixes IDN, lowercases host/scheme, removes default ports, resolves path segments
|
|
47
|
+
url_normalize("HTTP://User:Pass@www.FOO.com:80///foo/../bar/./baz?q=1#frag")
|
|
48
|
+
# -> 'http://User:Pass@www.foo.com/bar/baz?q=1#frag'
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
## Features
|
|
52
|
+
|
|
53
|
+
url-normalize provides a robust URI normalization function that handles IDN domains, scheme/host lowercasing, and RFC-compliant path normalization.
|
|
54
|
+
|
|
55
|
+
- **IDN Support**: Full internationalized domain name handling (using IDNA2008 with UTS46).
|
|
56
|
+
- **Humanization**: Convert normalized URLs to a readable display format while preserving round-trip normalization.
|
|
57
|
+
- **RFC Compliance**:
|
|
58
|
+
- Proper percent-encoding (minimal, uppercase hex).
|
|
59
|
+
- Dot-segment removal in paths.
|
|
60
|
+
- Default port and authority handling.
|
|
61
|
+
- UTF-8 NFC normalization.
|
|
62
|
+
- **Configurable Defaults**:
|
|
63
|
+
- Customizable default scheme (https by default).
|
|
64
|
+
- Configurable default domain for absolute paths.
|
|
65
|
+
- **Query Parameter Control**:
|
|
66
|
+
- Parameter filtering with allowlists.
|
|
67
|
+
- Support for domain-specific parameter rules.
|
|
68
|
+
- **Versatile URL Handling**: Handles empty strings, double-slash URLs (//domain.tld), and shebang (#!) URLs.
|
|
69
|
+
- **Developer Friendly**:
|
|
70
|
+
- Python 3.10+ compatibility.
|
|
71
|
+
- 100% statement coverage, enforced by the test suite.
|
|
72
|
+
- Modern type hints and string handling.
|
|
73
|
+
|
|
74
|
+
Inspired by Sam Ruby's [urlnorm.py](http://intertwingly.net/blog/2004/08/04/Urlnorm).
|
|
75
|
+
|
|
76
|
+
## Installation
|
|
77
|
+
|
|
78
|
+
Install as a library:
|
|
79
|
+
|
|
80
|
+
```sh
|
|
81
|
+
pip install url-normalize
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
Or install as a standalone CLI tool using [uv](https://docs.astral.sh/uv/):
|
|
85
|
+
|
|
86
|
+
```sh
|
|
87
|
+
uv tool install url-normalize
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
## Usage
|
|
91
|
+
|
|
92
|
+
### Python API
|
|
93
|
+
|
|
94
|
+
#### Basic Normalization
|
|
95
|
+
|
|
96
|
+
```python
|
|
97
|
+
from url_normalize import url_normalize
|
|
98
|
+
|
|
99
|
+
# Basic normalization (uses https by default)
|
|
100
|
+
print(url_normalize("www.foo.com:80/foo"))
|
|
101
|
+
# Output: https://www.foo.com:80/foo
|
|
102
|
+
|
|
103
|
+
# With custom default scheme
|
|
104
|
+
print(url_normalize("www.foo.com/foo", default_scheme="http"))
|
|
105
|
+
# Output: http://www.foo.com/foo
|
|
106
|
+
```
|
|
107
|
+
|
|
108
|
+
The `charset` argument remains for compatibility. Unicode characters use UTF-8 percent encoding, regardless of this argument.
|
|
109
|
+
|
|
110
|
+
#### Query Parameter Filtering
|
|
111
|
+
|
|
112
|
+
With `filter_params=True`, normalization retains only allowlisted query parameters. The built-in rules cover selected domains.
|
|
113
|
+
|
|
114
|
+
Other domains have an empty default allowlist. Without a custom allowlist, filtering removes all query parameters from those domains.
|
|
115
|
+
|
|
116
|
+
```python
|
|
117
|
+
# With the built-in Google allowlist
|
|
118
|
+
print(url_normalize("www.google.com/search?q=test&utm_source=test", filter_params=True))
|
|
119
|
+
# Output: https://www.google.com/search?q=test
|
|
120
|
+
|
|
121
|
+
# With custom parameter allowlist as a list
|
|
122
|
+
print(url_normalize(
|
|
123
|
+
"example.com?page=1&id=123&ref=test",
|
|
124
|
+
filter_params=True,
|
|
125
|
+
param_allowlist=["page", "id"]
|
|
126
|
+
))
|
|
127
|
+
# Output: https://example.com/?page=1&id=123
|
|
128
|
+
|
|
129
|
+
# With domain-specific parameter allowlists
|
|
130
|
+
print(url_normalize(
|
|
131
|
+
"example.com?page=1&id=123&ref=test",
|
|
132
|
+
filter_params=True,
|
|
133
|
+
param_allowlist={"example.com": ["page", "id"]}
|
|
134
|
+
))
|
|
135
|
+
# Output: https://example.com/?page=1&id=123
|
|
136
|
+
```
|
|
137
|
+
|
|
138
|
+
#### Default Domain & Scheme
|
|
139
|
+
|
|
140
|
+
Useful for resolving relative URLs found on a specific page.
|
|
141
|
+
|
|
142
|
+
```python
|
|
143
|
+
# With default domain for absolute paths
|
|
144
|
+
print(url_normalize("/images/logo.png", default_domain="example.com"))
|
|
145
|
+
# Output: https://example.com/images/logo.png
|
|
146
|
+
|
|
147
|
+
# With default domain and custom scheme
|
|
148
|
+
print(url_normalize("/images/logo.png", default_scheme="http", default_domain="example.com"))
|
|
149
|
+
# Output: http://example.com/images/logo.png
|
|
150
|
+
```
|
|
151
|
+
|
|
152
|
+
#### Humanizing URLs
|
|
153
|
+
|
|
154
|
+
Convert normalized URLs back into a user-friendly format for display, particularly useful for IDN domains and percent-encoded paths.
|
|
155
|
+
|
|
156
|
+
```python
|
|
157
|
+
from url_normalize import url_humanize
|
|
158
|
+
|
|
159
|
+
# Human-readable display form that still normalizes back to the same URL
|
|
160
|
+
print(url_humanize("https://xn--e1afmkfd.xn--80akhbyknj4f/%D0%A1%D0%BB%D1%83%D0%B6%D0%B5%D0%B1%D0%BD%D0%B0%D1%8F"))
|
|
161
|
+
# Output: https://пример.испытание/Служебная
|
|
162
|
+
|
|
163
|
+
# Humanization accepts the same normalization options
|
|
164
|
+
print(url_humanize("/%D0%A1%D0%BB%D1%83%D0%B6%D0%B5%D0%B1%D0%BD%D0%B0%D1%8F", default_domain="xn--e1afmkfd.xn--80akhbyknj4f"))
|
|
165
|
+
# Output: https://пример.испытание/Служебная
|
|
166
|
+
```
|
|
167
|
+
|
|
168
|
+
### Command-line Usage
|
|
169
|
+
|
|
170
|
+
You can also use `url-normalize` directly from the terminal to process URLs.
|
|
171
|
+
|
|
172
|
+
```bash
|
|
173
|
+
$ url-normalize "www.foo.com:80/foo"
|
|
174
|
+
# Output: https://www.foo.com:80/foo
|
|
175
|
+
|
|
176
|
+
# With custom default scheme
|
|
177
|
+
$ url-normalize -s http "www.foo.com/foo"
|
|
178
|
+
# Output: http://www.foo.com/foo
|
|
179
|
+
|
|
180
|
+
# With query parameter filtering
|
|
181
|
+
$ url-normalize -f "www.google.com/search?q=test&utm_source=test"
|
|
182
|
+
# Output: https://www.google.com/search?q=test
|
|
183
|
+
|
|
184
|
+
# With custom allowlist
|
|
185
|
+
$ url-normalize -f -p page,id "example.com?page=1&id=123&ref=test"
|
|
186
|
+
# Output: https://example.com/?page=1&id=123
|
|
187
|
+
|
|
188
|
+
# With default domain for absolute paths
|
|
189
|
+
$ url-normalize -d example.com "/images/logo.png"
|
|
190
|
+
# Output: https://example.com/images/logo.png
|
|
191
|
+
|
|
192
|
+
# With default domain and custom scheme
|
|
193
|
+
$ url-normalize -d example.com -s http "/images/logo.png"
|
|
194
|
+
# Output: http://example.com/images/logo.png
|
|
195
|
+
|
|
196
|
+
# Human-readable display form
|
|
197
|
+
$ url-normalize -H "https://xn--e1afmkfd.xn--80akhbyknj4f/%D0%A1%D0%BB%D1%83%D0%B6%D0%B5%D0%B1%D0%BD%D0%B0%D1%8F"
|
|
198
|
+
# Output: https://пример.испытание/Служебная
|
|
199
|
+
|
|
200
|
+
# Via uv tool/uvx
|
|
201
|
+
$ uvx url-normalize www.foo.com:80/foo
|
|
202
|
+
# Output: https://www.foo.com:80/foo
|
|
203
|
+
```
|
|
204
|
+
|
|
205
|
+
## Documentation
|
|
206
|
+
|
|
207
|
+
For a complete history of changes, see [CHANGELOG.md](CHANGELOG.md).
|
|
208
|
+
|
|
209
|
+
## Contributing
|
|
210
|
+
|
|
211
|
+
Contributions are welcome! Please feel free to submit a Pull Request.
|
|
212
|
+
|
|
213
|
+
## License
|
|
214
|
+
|
|
215
|
+
MIT License
|
|
@@ -0,0 +1,184 @@
|
|
|
1
|
+
# url-normalize
|
|
2
|
+
|
|
3
|
+
[](https://github.com/niksite/url-normalize/actions/workflows/ci.yml)
|
|
4
|
+
[](https://coveralls.io/r/niksite/url-normalize)
|
|
5
|
+
[](https://pypi.org/project/url-normalize/)
|
|
6
|
+
[](https://pypi.org/project/url-normalize/)
|
|
7
|
+
[](https://github.com/niksite/url-normalize/blob/master/LICENSE)
|
|
8
|
+
[](https://github.com/astral-sh/ruff)
|
|
9
|
+
|
|
10
|
+
A Python library for standardizing and normalizing URLs. Ideal for database deduplication, caching, web crawling, and anywhere you need to ensure that equivalent URLs resolve to the exact same string.
|
|
11
|
+
|
|
12
|
+
```python
|
|
13
|
+
from url_normalize import url_normalize
|
|
14
|
+
|
|
15
|
+
# Fixes IDN, lowercases host/scheme, removes default ports, resolves path segments
|
|
16
|
+
url_normalize("HTTP://User:Pass@www.FOO.com:80///foo/../bar/./baz?q=1#frag")
|
|
17
|
+
# -> 'http://User:Pass@www.foo.com/bar/baz?q=1#frag'
|
|
18
|
+
```
|
|
19
|
+
|
|
20
|
+
## Features
|
|
21
|
+
|
|
22
|
+
url-normalize provides a robust URI normalization function that handles IDN domains, scheme/host lowercasing, and RFC-compliant path normalization.
|
|
23
|
+
|
|
24
|
+
- **IDN Support**: Full internationalized domain name handling (using IDNA2008 with UTS46).
|
|
25
|
+
- **Humanization**: Convert normalized URLs to a readable display format while preserving round-trip normalization.
|
|
26
|
+
- **RFC Compliance**:
|
|
27
|
+
- Proper percent-encoding (minimal, uppercase hex).
|
|
28
|
+
- Dot-segment removal in paths.
|
|
29
|
+
- Default port and authority handling.
|
|
30
|
+
- UTF-8 NFC normalization.
|
|
31
|
+
- **Configurable Defaults**:
|
|
32
|
+
- Customizable default scheme (https by default).
|
|
33
|
+
- Configurable default domain for absolute paths.
|
|
34
|
+
- **Query Parameter Control**:
|
|
35
|
+
- Parameter filtering with allowlists.
|
|
36
|
+
- Support for domain-specific parameter rules.
|
|
37
|
+
- **Versatile URL Handling**: Handles empty strings, double-slash URLs (//domain.tld), and shebang (#!) URLs.
|
|
38
|
+
- **Developer Friendly**:
|
|
39
|
+
- Python 3.10+ compatibility.
|
|
40
|
+
- 100% statement coverage, enforced by the test suite.
|
|
41
|
+
- Modern type hints and string handling.
|
|
42
|
+
|
|
43
|
+
Inspired by Sam Ruby's [urlnorm.py](http://intertwingly.net/blog/2004/08/04/Urlnorm).
|
|
44
|
+
|
|
45
|
+
## Installation
|
|
46
|
+
|
|
47
|
+
Install as a library:
|
|
48
|
+
|
|
49
|
+
```sh
|
|
50
|
+
pip install url-normalize
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
Or install as a standalone CLI tool using [uv](https://docs.astral.sh/uv/):
|
|
54
|
+
|
|
55
|
+
```sh
|
|
56
|
+
uv tool install url-normalize
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
## Usage
|
|
60
|
+
|
|
61
|
+
### Python API
|
|
62
|
+
|
|
63
|
+
#### Basic Normalization
|
|
64
|
+
|
|
65
|
+
```python
|
|
66
|
+
from url_normalize import url_normalize
|
|
67
|
+
|
|
68
|
+
# Basic normalization (uses https by default)
|
|
69
|
+
print(url_normalize("www.foo.com:80/foo"))
|
|
70
|
+
# Output: https://www.foo.com:80/foo
|
|
71
|
+
|
|
72
|
+
# With custom default scheme
|
|
73
|
+
print(url_normalize("www.foo.com/foo", default_scheme="http"))
|
|
74
|
+
# Output: http://www.foo.com/foo
|
|
75
|
+
```
|
|
76
|
+
|
|
77
|
+
The `charset` argument remains for compatibility. Unicode characters use UTF-8 percent encoding, regardless of this argument.
|
|
78
|
+
|
|
79
|
+
#### Query Parameter Filtering
|
|
80
|
+
|
|
81
|
+
With `filter_params=True`, normalization retains only allowlisted query parameters. The built-in rules cover selected domains.
|
|
82
|
+
|
|
83
|
+
Other domains have an empty default allowlist. Without a custom allowlist, filtering removes all query parameters from those domains.
|
|
84
|
+
|
|
85
|
+
```python
|
|
86
|
+
# With the built-in Google allowlist
|
|
87
|
+
print(url_normalize("www.google.com/search?q=test&utm_source=test", filter_params=True))
|
|
88
|
+
# Output: https://www.google.com/search?q=test
|
|
89
|
+
|
|
90
|
+
# With custom parameter allowlist as a list
|
|
91
|
+
print(url_normalize(
|
|
92
|
+
"example.com?page=1&id=123&ref=test",
|
|
93
|
+
filter_params=True,
|
|
94
|
+
param_allowlist=["page", "id"]
|
|
95
|
+
))
|
|
96
|
+
# Output: https://example.com/?page=1&id=123
|
|
97
|
+
|
|
98
|
+
# With domain-specific parameter allowlists
|
|
99
|
+
print(url_normalize(
|
|
100
|
+
"example.com?page=1&id=123&ref=test",
|
|
101
|
+
filter_params=True,
|
|
102
|
+
param_allowlist={"example.com": ["page", "id"]}
|
|
103
|
+
))
|
|
104
|
+
# Output: https://example.com/?page=1&id=123
|
|
105
|
+
```
|
|
106
|
+
|
|
107
|
+
#### Default Domain & Scheme
|
|
108
|
+
|
|
109
|
+
Useful for resolving relative URLs found on a specific page.
|
|
110
|
+
|
|
111
|
+
```python
|
|
112
|
+
# With default domain for absolute paths
|
|
113
|
+
print(url_normalize("/images/logo.png", default_domain="example.com"))
|
|
114
|
+
# Output: https://example.com/images/logo.png
|
|
115
|
+
|
|
116
|
+
# With default domain and custom scheme
|
|
117
|
+
print(url_normalize("/images/logo.png", default_scheme="http", default_domain="example.com"))
|
|
118
|
+
# Output: http://example.com/images/logo.png
|
|
119
|
+
```
|
|
120
|
+
|
|
121
|
+
#### Humanizing URLs
|
|
122
|
+
|
|
123
|
+
Convert normalized URLs back into a user-friendly format for display, particularly useful for IDN domains and percent-encoded paths.
|
|
124
|
+
|
|
125
|
+
```python
|
|
126
|
+
from url_normalize import url_humanize
|
|
127
|
+
|
|
128
|
+
# Human-readable display form that still normalizes back to the same URL
|
|
129
|
+
print(url_humanize("https://xn--e1afmkfd.xn--80akhbyknj4f/%D0%A1%D0%BB%D1%83%D0%B6%D0%B5%D0%B1%D0%BD%D0%B0%D1%8F"))
|
|
130
|
+
# Output: https://пример.испытание/Служебная
|
|
131
|
+
|
|
132
|
+
# Humanization accepts the same normalization options
|
|
133
|
+
print(url_humanize("/%D0%A1%D0%BB%D1%83%D0%B6%D0%B5%D0%B1%D0%BD%D0%B0%D1%8F", default_domain="xn--e1afmkfd.xn--80akhbyknj4f"))
|
|
134
|
+
# Output: https://пример.испытание/Служебная
|
|
135
|
+
```
|
|
136
|
+
|
|
137
|
+
### Command-line Usage
|
|
138
|
+
|
|
139
|
+
You can also use `url-normalize` directly from the terminal to process URLs.
|
|
140
|
+
|
|
141
|
+
```bash
|
|
142
|
+
$ url-normalize "www.foo.com:80/foo"
|
|
143
|
+
# Output: https://www.foo.com:80/foo
|
|
144
|
+
|
|
145
|
+
# With custom default scheme
|
|
146
|
+
$ url-normalize -s http "www.foo.com/foo"
|
|
147
|
+
# Output: http://www.foo.com/foo
|
|
148
|
+
|
|
149
|
+
# With query parameter filtering
|
|
150
|
+
$ url-normalize -f "www.google.com/search?q=test&utm_source=test"
|
|
151
|
+
# Output: https://www.google.com/search?q=test
|
|
152
|
+
|
|
153
|
+
# With custom allowlist
|
|
154
|
+
$ url-normalize -f -p page,id "example.com?page=1&id=123&ref=test"
|
|
155
|
+
# Output: https://example.com/?page=1&id=123
|
|
156
|
+
|
|
157
|
+
# With default domain for absolute paths
|
|
158
|
+
$ url-normalize -d example.com "/images/logo.png"
|
|
159
|
+
# Output: https://example.com/images/logo.png
|
|
160
|
+
|
|
161
|
+
# With default domain and custom scheme
|
|
162
|
+
$ url-normalize -d example.com -s http "/images/logo.png"
|
|
163
|
+
# Output: http://example.com/images/logo.png
|
|
164
|
+
|
|
165
|
+
# Human-readable display form
|
|
166
|
+
$ url-normalize -H "https://xn--e1afmkfd.xn--80akhbyknj4f/%D0%A1%D0%BB%D1%83%D0%B6%D0%B5%D0%B1%D0%BD%D0%B0%D1%8F"
|
|
167
|
+
# Output: https://пример.испытание/Служебная
|
|
168
|
+
|
|
169
|
+
# Via uv tool/uvx
|
|
170
|
+
$ uvx url-normalize www.foo.com:80/foo
|
|
171
|
+
# Output: https://www.foo.com:80/foo
|
|
172
|
+
```
|
|
173
|
+
|
|
174
|
+
## Documentation
|
|
175
|
+
|
|
176
|
+
For a complete history of changes, see [CHANGELOG.md](CHANGELOG.md).
|
|
177
|
+
|
|
178
|
+
## Contributing
|
|
179
|
+
|
|
180
|
+
Contributions are welcome! Please feel free to submit a Pull Request.
|
|
181
|
+
|
|
182
|
+
## License
|
|
183
|
+
|
|
184
|
+
MIT License
|
|
@@ -1,12 +1,21 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "url-normalize"
|
|
3
|
-
version = "
|
|
3
|
+
version = "3.0.1"
|
|
4
4
|
description = "URL normalization for Python"
|
|
5
5
|
authors = [{ name = "Nikolay Panov", email = "github@npanov.com" }]
|
|
6
|
-
license =
|
|
6
|
+
license = "MIT"
|
|
7
7
|
readme = "README.md"
|
|
8
|
-
requires-python = ">=3.
|
|
9
|
-
keywords = ["url", "normalization", "normalize"]
|
|
8
|
+
requires-python = ">=3.10"
|
|
9
|
+
keywords = ["url", "normalization", "normalize", "normalizer"]
|
|
10
|
+
classifiers = [
|
|
11
|
+
"Programming Language :: Python :: 3",
|
|
12
|
+
"Programming Language :: Python :: 3 :: Only",
|
|
13
|
+
"Programming Language :: Python :: 3.10",
|
|
14
|
+
"Programming Language :: Python :: 3.11",
|
|
15
|
+
"Programming Language :: Python :: 3.12",
|
|
16
|
+
"Programming Language :: Python :: 3.13",
|
|
17
|
+
"Programming Language :: Python :: 3.14",
|
|
18
|
+
]
|
|
10
19
|
dependencies = ["idna>=3.3"]
|
|
11
20
|
|
|
12
21
|
[project.urls]
|
|
@@ -19,19 +28,10 @@ Changelog = "https://github.com/niksite/url-normalize/blob/master/CHANGELOG.md"
|
|
|
19
28
|
url-normalize = "url_normalize.cli:main"
|
|
20
29
|
|
|
21
30
|
[project.optional-dependencies]
|
|
22
|
-
dev = [
|
|
23
|
-
"mypy",
|
|
24
|
-
"pre-commit",
|
|
25
|
-
"pytest-cov",
|
|
26
|
-
"pytest-ruff",
|
|
27
|
-
"pytest-socket",
|
|
28
|
-
"pytest",
|
|
29
|
-
"ruff",
|
|
30
|
-
"tox",
|
|
31
|
-
]
|
|
31
|
+
dev = ["mypy", "pre-commit", "pytest-cov", "pytest-socket", "pytest", "ruff"]
|
|
32
32
|
|
|
33
33
|
[tool.ruff]
|
|
34
|
-
target-version = "
|
|
34
|
+
target-version = "py310"
|
|
35
35
|
line-length = 88
|
|
36
36
|
unsafe-fixes = true
|
|
37
37
|
|
|
@@ -62,20 +62,19 @@ indent-style = "space"
|
|
|
62
62
|
[tool.mypy]
|
|
63
63
|
ignore_missing_imports = true
|
|
64
64
|
exclude = ["tests"]
|
|
65
|
-
python_version = "3.
|
|
65
|
+
python_version = "3.10"
|
|
66
66
|
show_error_codes = true
|
|
67
67
|
|
|
68
68
|
[build-system]
|
|
69
|
-
requires = ["setuptools>=
|
|
69
|
+
requires = ["setuptools>=77", "wheel"]
|
|
70
70
|
build-backend = "setuptools.build_meta"
|
|
71
71
|
|
|
72
72
|
[tool.pytest.ini_options]
|
|
73
73
|
addopts = [
|
|
74
|
-
"--cov-fail-under=100",
|
|
75
74
|
"--cov-report=term-missing:skip-covered",
|
|
76
75
|
"--cov=url_normalize",
|
|
76
|
+
"--cov-fail-under=100",
|
|
77
77
|
"--disable-socket",
|
|
78
|
-
"--ruff",
|
|
79
78
|
"-v",
|
|
80
79
|
]
|
|
81
80
|
python_files = ["tests.py", "test_*.py", "*_tests.py"]
|
|
@@ -22,10 +22,42 @@ def run_cli(*args: str) -> subprocess.CompletedProcess:
|
|
|
22
22
|
"""
|
|
23
23
|
command = [sys.executable, "-m", "url_normalize.cli", *list(args)]
|
|
24
24
|
return subprocess.run( # noqa: S603
|
|
25
|
-
command, capture_output=True, text=True, check=False
|
|
25
|
+
command, capture_output=True, text=True, encoding="utf-8", check=False
|
|
26
26
|
)
|
|
27
27
|
|
|
28
28
|
|
|
29
|
+
def test_cli_help_describes_allowlist_filtering() -> None:
|
|
30
|
+
"""Describe filtering as an allowlist, not a tracking-parameter blocklist."""
|
|
31
|
+
result = run_cli("--help")
|
|
32
|
+
assert result.returncode == 0
|
|
33
|
+
help_text = " ".join(result.stdout.split())
|
|
34
|
+
assert "Keep only allowlisted query parameters." in help_text
|
|
35
|
+
assert "Unknown domains have an empty default allowlist." in help_text
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def test_cli_help_describes_fixed_utf8_encoding() -> None:
|
|
39
|
+
"""Describe charset as a compatibility option for Unicode URL input."""
|
|
40
|
+
result = run_cli("--help")
|
|
41
|
+
assert result.returncode == 0
|
|
42
|
+
help_text = " ".join(result.stdout.split())
|
|
43
|
+
assert "Retained for compatibility." in help_text
|
|
44
|
+
assert "Unicode characters use UTF-8 percent encoding." in help_text
|
|
45
|
+
normalized = run_cli("--charset", "iso-8859-1", "https://example.com/é")
|
|
46
|
+
assert normalized.returncode == 0
|
|
47
|
+
assert normalized.stdout.strip() == "https://example.com/%C3%A9"
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def test_cli_successful_main(capsys, monkeypatch):
|
|
51
|
+
"""Exercise successful CLI output in the measured process."""
|
|
52
|
+
monkeypatch.setattr(
|
|
53
|
+
sys, "argv", ["url-normalize", "http://EXAMPLE.com/./path/../other/"]
|
|
54
|
+
)
|
|
55
|
+
main()
|
|
56
|
+
captured = capsys.readouterr()
|
|
57
|
+
assert captured.out == "http://example.com/other/\n"
|
|
58
|
+
assert not captured.err
|
|
59
|
+
|
|
60
|
+
|
|
29
61
|
def test_cli_error_handling(capsys, monkeypatch):
|
|
30
62
|
"""Test CLI error handling when URL normalization fails."""
|
|
31
63
|
with patch("url_normalize.cli.url_normalize") as mock_normalize:
|
|
@@ -66,6 +98,56 @@ def test_cli_basic_normalization_short_args() -> None:
|
|
|
66
98
|
assert not result.stderr
|
|
67
99
|
|
|
68
100
|
|
|
101
|
+
def test_cli_humanize() -> None:
|
|
102
|
+
"""Test human-readable URL output via CLI."""
|
|
103
|
+
url = (
|
|
104
|
+
"https://xn--e1afmkfd.xn--80akhbyknj4f/"
|
|
105
|
+
"%D0%A1%D0%BB%D1%83%D0%B6%D0%B5%D0%B1%D0%BD%D0%B0%D1%8F"
|
|
106
|
+
)
|
|
107
|
+
expected = "https://пример.испытание/Служебная"
|
|
108
|
+
|
|
109
|
+
result = run_cli("--humanize", url)
|
|
110
|
+
|
|
111
|
+
assert result.returncode == 0
|
|
112
|
+
assert result.stdout.strip() == expected
|
|
113
|
+
assert not result.stderr
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def test_cli_humanize_short_arg() -> None:
|
|
117
|
+
"""Test human-readable URL output via CLI using short argument."""
|
|
118
|
+
url = (
|
|
119
|
+
"https://xn--e1afmkfd.xn--80akhbyknj4f/"
|
|
120
|
+
"%D0%A1%D0%BB%D1%83%D0%B6%D0%B5%D0%B1%D0%BD%D0%B0%D1%8F"
|
|
121
|
+
)
|
|
122
|
+
expected = "https://пример.испытание/Служебная"
|
|
123
|
+
|
|
124
|
+
result = run_cli("-H", url)
|
|
125
|
+
|
|
126
|
+
assert result.returncode == 0
|
|
127
|
+
assert result.stdout.strip() == expected
|
|
128
|
+
assert not result.stderr
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def test_cli_humanize_respects_normalization_options() -> None:
|
|
132
|
+
"""Test humanized output keeps the existing CLI normalization options."""
|
|
133
|
+
url = "/%D0%A1%D0%BB%D1%83%D0%B6%D0%B5%D0%B1%D0%BD%D0%B0%D1%8F"
|
|
134
|
+
expected = "https://пример.испытание/Служебная?keep=Ç"
|
|
135
|
+
|
|
136
|
+
result = run_cli(
|
|
137
|
+
"--humanize",
|
|
138
|
+
"--default-domain",
|
|
139
|
+
"xn--e1afmkfd.xn--80akhbyknj4f",
|
|
140
|
+
"--filter-params",
|
|
141
|
+
"--param-allowlist",
|
|
142
|
+
"keep",
|
|
143
|
+
f"{url}?utm_source=ignored&keep=%C3%87",
|
|
144
|
+
)
|
|
145
|
+
|
|
146
|
+
assert result.returncode == 0
|
|
147
|
+
assert result.stdout.strip() == expected
|
|
148
|
+
assert not result.stderr
|
|
149
|
+
|
|
150
|
+
|
|
69
151
|
def test_cli_default_scheme() -> None:
|
|
70
152
|
"""Test default scheme addition via CLI."""
|
|
71
153
|
url = "//example.com"
|
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
"""Deconstruct url tests."""
|
|
2
|
+
|
|
3
|
+
import pytest
|
|
4
|
+
|
|
5
|
+
from url_normalize import url_normalize
|
|
6
|
+
from url_normalize.tools import URL, deconstruct_url
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
@pytest.mark.parametrize(
|
|
10
|
+
("url", "expected"),
|
|
11
|
+
[
|
|
12
|
+
(
|
|
13
|
+
"http://site.com",
|
|
14
|
+
URL(
|
|
15
|
+
fragment="",
|
|
16
|
+
host="site.com",
|
|
17
|
+
path="",
|
|
18
|
+
port="",
|
|
19
|
+
query="",
|
|
20
|
+
scheme="http",
|
|
21
|
+
userinfo="",
|
|
22
|
+
),
|
|
23
|
+
),
|
|
24
|
+
(
|
|
25
|
+
"http://user@www.example.com:8080/path/index.html?param=val#fragment",
|
|
26
|
+
URL(
|
|
27
|
+
fragment="fragment",
|
|
28
|
+
host="www.example.com",
|
|
29
|
+
path="/path/index.html",
|
|
30
|
+
port="8080",
|
|
31
|
+
query="param=val",
|
|
32
|
+
scheme="http",
|
|
33
|
+
userinfo="user@",
|
|
34
|
+
),
|
|
35
|
+
),
|
|
36
|
+
],
|
|
37
|
+
)
|
|
38
|
+
def test_deconstruct_url_result_is_expected(url: str, expected: URL) -> None:
|
|
39
|
+
"""Assert we got expected results from the deconstruct_url function."""
|
|
40
|
+
result = deconstruct_url(url)
|
|
41
|
+
assert result == expected, url
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
@pytest.mark.parametrize(
|
|
45
|
+
("authority", "userinfo", "host", "port", "normalized_authority"),
|
|
46
|
+
[
|
|
47
|
+
("[2001:DB8::1]:080", "", "[2001:DB8::1]", "080", "[2001:db8::1]"),
|
|
48
|
+
("[2001:DB8::1]:0081", "", "[2001:DB8::1]", "0081", "[2001:db8::1]:81"),
|
|
49
|
+
("[::1]", "", "[::1]", "", "[::1]"),
|
|
50
|
+
("[::1]:", "", "[::1]", "", "[::1]"),
|
|
51
|
+
("user:pass@[::1]:080", "user:pass@", "[::1]", "080", "user:pass@[::1]"),
|
|
52
|
+
("EXAMPLE.com:080", "", "EXAMPLE.com", "080", "example.com"),
|
|
53
|
+
("127.0.0.1:0081", "", "127.0.0.1", "0081", "127.0.0.1:81"),
|
|
54
|
+
],
|
|
55
|
+
)
|
|
56
|
+
def test_deconstruct_url_separates_bracketed_host_and_port(
|
|
57
|
+
authority, userinfo, host, port, normalized_authority
|
|
58
|
+
):
|
|
59
|
+
"""Parse the entire bracketed host before considering a port separator."""
|
|
60
|
+
url = f"http://{authority}/path?q=1#fragment"
|
|
61
|
+
assert deconstruct_url(url) == URL(
|
|
62
|
+
"http", userinfo, host, port, "/path", "q=1", "fragment"
|
|
63
|
+
)
|
|
64
|
+
assert url_normalize(url) == f"http://{normalized_authority}/path?q=1#fragment"
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
@pytest.mark.parametrize("whitespace", [" ", "\u00a0", " \t\r\n"])
|
|
68
|
+
@pytest.mark.parametrize("suffix", ["", "/path ", "?q=x ", "#part "])
|
|
69
|
+
@pytest.mark.parametrize(
|
|
70
|
+
("authority", "normalized"),
|
|
71
|
+
[
|
|
72
|
+
("EXAMPLE.com", "example.com"),
|
|
73
|
+
("example.com:443", "example.com"),
|
|
74
|
+
("user:pass@example.com:443", "user:pass@example.com"),
|
|
75
|
+
("[FE80::1%25ethA]:443", "[fe80::1%25ethA]"),
|
|
76
|
+
],
|
|
77
|
+
)
|
|
78
|
+
def test_deconstruct_url_trims_authority_whitespace_only(
|
|
79
|
+
authority, normalized, whitespace, suffix
|
|
80
|
+
):
|
|
81
|
+
"""Trim the authority without deleting whitespace from other components."""
|
|
82
|
+
value = f"https://{authority}{whitespace}{suffix}"
|
|
83
|
+
assert deconstruct_url(value) == deconstruct_url(f"https://{authority}{suffix}")
|
|
84
|
+
path_suffix = suffix if suffix.startswith("/") else "/" + suffix
|
|
85
|
+
assert (
|
|
86
|
+
url_normalize(value) == f"https://{normalized}{path_suffix.replace(' ', '%20')}"
|
|
87
|
+
)
|