url-normalize 2.1.0__tar.gz → 2.2.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Potentially problematic release.
This version of url-normalize might be problematic. Click here for more details.
- url_normalize-2.2.1/PKG-INFO +175 -0
- url_normalize-2.2.1/README.md +151 -0
- {url_normalize-2.1.0 → url_normalize-2.2.1}/pyproject.toml +3 -14
- {url_normalize-2.1.0 → url_normalize-2.2.1}/tests/test_cli.py +61 -0
- url_normalize-2.2.1/tests/test_deconstruct_url.py +40 -0
- url_normalize-2.2.1/tests/test_normalize_fragment.py +23 -0
- url_normalize-2.2.1/tests/test_normalize_host.py +32 -0
- url_normalize-2.2.1/tests/test_normalize_path.py +44 -0
- url_normalize-2.2.1/tests/test_normalize_port.py +26 -0
- {url_normalize-2.1.0 → url_normalize-2.2.1}/tests/test_normalize_query.py +2 -0
- url_normalize-2.2.1/tests/test_normalize_scheme.py +18 -0
- url_normalize-2.2.1/tests/test_normalize_userinfo.py +21 -0
- url_normalize-2.2.1/tests/test_provide_url_domain.py +31 -0
- url_normalize-2.2.1/tests/test_provide_url_scheme.py +32 -0
- url_normalize-2.2.1/tests/test_reconstruct_url.py +42 -0
- url_normalize-2.2.1/tests/test_url_normalize.py +163 -0
- {url_normalize-2.1.0 → url_normalize-2.2.1}/url_normalize/__init__.py +1 -1
- {url_normalize-2.1.0 → url_normalize-2.2.1}/url_normalize/cli.py +7 -0
- {url_normalize-2.1.0 → url_normalize-2.2.1}/url_normalize/normalize_path.py +4 -4
- {url_normalize-2.1.0 → url_normalize-2.2.1}/url_normalize/normalize_query.py +2 -2
- url_normalize-2.2.1/url_normalize/provide_url_domain.py +28 -0
- {url_normalize-2.1.0 → url_normalize-2.2.1}/url_normalize/provide_url_scheme.py +11 -4
- url_normalize-2.2.1/url_normalize/py.typed +0 -0
- {url_normalize-2.1.0 → url_normalize-2.2.1}/url_normalize/url_normalize.py +7 -2
- url_normalize-2.2.1/url_normalize.egg-info/PKG-INFO +175 -0
- {url_normalize-2.1.0 → url_normalize-2.2.1}/url_normalize.egg-info/SOURCES.txt +3 -0
- {url_normalize-2.1.0 → url_normalize-2.2.1}/url_normalize.egg-info/requires.txt +0 -2
- url_normalize-2.1.0/PKG-INFO +0 -133
- url_normalize-2.1.0/README.md +0 -107
- url_normalize-2.1.0/tests/test_deconstruct_url.py +0 -35
- url_normalize-2.1.0/tests/test_normalize_fragment.py +0 -21
- url_normalize-2.1.0/tests/test_normalize_host.py +0 -29
- url_normalize-2.1.0/tests/test_normalize_path.py +0 -40
- url_normalize-2.1.0/tests/test_normalize_port.py +0 -13
- url_normalize-2.1.0/tests/test_normalize_scheme.py +0 -13
- url_normalize-2.1.0/tests/test_normalize_userinfo.py +0 -19
- url_normalize-2.1.0/tests/test_provide_url_scheme.py +0 -30
- url_normalize-2.1.0/tests/test_reconstruct_url.py +0 -41
- url_normalize-2.1.0/tests/test_url_normalize.py +0 -127
- url_normalize-2.1.0/url_normalize.egg-info/PKG-INFO +0 -133
- {url_normalize-2.1.0 → url_normalize-2.2.1}/LICENSE +0 -0
- {url_normalize-2.1.0 → url_normalize-2.2.1}/setup.cfg +0 -0
- {url_normalize-2.1.0 → url_normalize-2.2.1}/tests/test_generic_url_cleanup.py +0 -0
- {url_normalize-2.1.0 → url_normalize-2.2.1}/tests/test_normalize_query_filters.py +0 -0
- {url_normalize-2.1.0 → url_normalize-2.2.1}/tests/test_tools.py +0 -0
- {url_normalize-2.1.0 → url_normalize-2.2.1}/url_normalize/generic_url_cleanup.py +0 -0
- {url_normalize-2.1.0 → url_normalize-2.2.1}/url_normalize/normalize_fragment.py +0 -0
- {url_normalize-2.1.0 → url_normalize-2.2.1}/url_normalize/normalize_host.py +0 -0
- {url_normalize-2.1.0 → url_normalize-2.2.1}/url_normalize/normalize_port.py +0 -0
- {url_normalize-2.1.0 → url_normalize-2.2.1}/url_normalize/normalize_scheme.py +0 -0
- {url_normalize-2.1.0 → url_normalize-2.2.1}/url_normalize/normalize_userinfo.py +0 -0
- {url_normalize-2.1.0 → url_normalize-2.2.1}/url_normalize/param_allowlist.py +0 -0
- {url_normalize-2.1.0 → url_normalize-2.2.1}/url_normalize/tools.py +0 -0
- {url_normalize-2.1.0 → url_normalize-2.2.1}/url_normalize.egg-info/dependency_links.txt +0 -0
- {url_normalize-2.1.0 → url_normalize-2.2.1}/url_normalize.egg-info/entry_points.txt +0 -0
- {url_normalize-2.1.0 → url_normalize-2.2.1}/url_normalize.egg-info/top_level.txt +0 -0
|
@@ -0,0 +1,175 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: url-normalize
|
|
3
|
+
Version: 2.2.1
|
|
4
|
+
Summary: URL normalization for Python
|
|
5
|
+
Author-email: Nikolay Panov <github@npanov.com>
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/niksite/url-normalize
|
|
8
|
+
Project-URL: Repository, https://github.com/niksite/url-normalize
|
|
9
|
+
Project-URL: Issues, https://github.com/niksite/url-normalize/issues
|
|
10
|
+
Project-URL: Changelog, https://github.com/niksite/url-normalize/blob/master/CHANGELOG.md
|
|
11
|
+
Keywords: url,normalization,normalize,normalizer
|
|
12
|
+
Requires-Python: >=3.8
|
|
13
|
+
Description-Content-Type: text/markdown
|
|
14
|
+
License-File: LICENSE
|
|
15
|
+
Requires-Dist: idna>=3.3
|
|
16
|
+
Provides-Extra: dev
|
|
17
|
+
Requires-Dist: mypy; extra == "dev"
|
|
18
|
+
Requires-Dist: pre-commit; extra == "dev"
|
|
19
|
+
Requires-Dist: pytest-cov; extra == "dev"
|
|
20
|
+
Requires-Dist: pytest-socket; extra == "dev"
|
|
21
|
+
Requires-Dist: pytest; extra == "dev"
|
|
22
|
+
Requires-Dist: ruff; extra == "dev"
|
|
23
|
+
Dynamic: license-file
|
|
24
|
+
|
|
25
|
+
# url-normalize
|
|
26
|
+
|
|
27
|
+
[](https://github.com/niksite/url-normalize/actions/workflows/ci.yml)
|
|
28
|
+
[](https://coveralls.io/r/niksite/url-normalize)
|
|
29
|
+
[](https://pypi.org/project/url-normalize/)
|
|
30
|
+
|
|
31
|
+
A Python library for standardizing and normalizing URLs with support for internationalized domain names (IDN).
|
|
32
|
+
|
|
33
|
+
## Table of Contents
|
|
34
|
+
|
|
35
|
+
- [Introduction](#introduction)
|
|
36
|
+
- [Features](#features)
|
|
37
|
+
- [Installation](#installation)
|
|
38
|
+
- [Usage](#usage)
|
|
39
|
+
- [Python API](#python-api)
|
|
40
|
+
- [Command Line](#command-line-usage)
|
|
41
|
+
- [Documentation](#documentation)
|
|
42
|
+
- [Contributing](#contributing)
|
|
43
|
+
- [License](#license)
|
|
44
|
+
|
|
45
|
+
## Introduction
|
|
46
|
+
|
|
47
|
+
url-normalize provides a robust URI normalization function that:
|
|
48
|
+
|
|
49
|
+
- Takes care of IDN domains.
|
|
50
|
+
- Always provides the URI scheme in lowercase characters.
|
|
51
|
+
- Always provides the host, if any, in lowercase characters.
|
|
52
|
+
- Only performs percent-encoding where it is essential.
|
|
53
|
+
- Always uses uppercase A-through-F characters when percent-encoding.
|
|
54
|
+
- Prevents dot-segments appearing in non-relative URI paths.
|
|
55
|
+
- For schemes that define a default authority, uses an empty authority if the
|
|
56
|
+
default is desired.
|
|
57
|
+
- For schemes that define an empty path to be equivalent to a path of "/",
|
|
58
|
+
uses "/".
|
|
59
|
+
- For schemes that define a port, uses an empty port if the default is desired
|
|
60
|
+
- Ensures all portions of the URI are utf-8 encoded NFC from Unicode strings
|
|
61
|
+
|
|
62
|
+
Inspired by Sam Ruby's [urlnorm.py](http://intertwingly.net/blog/2004/08/04/Urlnorm)
|
|
63
|
+
|
|
64
|
+
## Features
|
|
65
|
+
|
|
66
|
+
- **IDN Support**: Full internationalized domain name handling
|
|
67
|
+
- **Configurable Defaults**:
|
|
68
|
+
- Customizable default scheme (https by default)
|
|
69
|
+
- Configurable default domain for absolute paths
|
|
70
|
+
- **Query Parameter Control**:
|
|
71
|
+
- Parameter filtering with allowlists
|
|
72
|
+
- Support for domain-specific parameter rules
|
|
73
|
+
- **Versatile URL Handling**:
|
|
74
|
+
- Empty string URLs
|
|
75
|
+
- Double slash URLs (//domain.tld)
|
|
76
|
+
- Shebang (#!) URLs
|
|
77
|
+
- **Developer Friendly**:
|
|
78
|
+
- Cross-version Python compatibility (3.8+)
|
|
79
|
+
- 100% test coverage
|
|
80
|
+
- Modern type hints and string handling
|
|
81
|
+
|
|
82
|
+
## Installation
|
|
83
|
+
|
|
84
|
+
```sh
|
|
85
|
+
pip install url-normalize
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
## Usage
|
|
89
|
+
|
|
90
|
+
### Python API
|
|
91
|
+
|
|
92
|
+
```python
|
|
93
|
+
from url_normalize import url_normalize
|
|
94
|
+
|
|
95
|
+
# Basic normalization (uses https by default)
|
|
96
|
+
print(url_normalize("www.foo.com:80/foo"))
|
|
97
|
+
# Output: https://www.foo.com/foo
|
|
98
|
+
|
|
99
|
+
# With custom default scheme
|
|
100
|
+
print(url_normalize("www.foo.com/foo", default_scheme="http"))
|
|
101
|
+
# Output: http://www.foo.com/foo
|
|
102
|
+
|
|
103
|
+
# With query parameter filtering enabled
|
|
104
|
+
print(url_normalize("www.google.com/search?q=test&utm_source=test", filter_params=True))
|
|
105
|
+
# Output: https://www.google.com/search?q=test
|
|
106
|
+
|
|
107
|
+
# With custom parameter allowlist as a dict
|
|
108
|
+
print(url_normalize(
|
|
109
|
+
"example.com?page=1&id=123&ref=test",
|
|
110
|
+
filter_params=True,
|
|
111
|
+
param_allowlist={"example.com": ["page", "id"]}
|
|
112
|
+
))
|
|
113
|
+
# Output: https://example.com?page=1&id=123
|
|
114
|
+
|
|
115
|
+
# With custom parameter allowlist as a list
|
|
116
|
+
print(url_normalize(
|
|
117
|
+
"example.com?page=1&id=123&ref=test",
|
|
118
|
+
filter_params=True,
|
|
119
|
+
param_allowlist=["page", "id"]
|
|
120
|
+
))
|
|
121
|
+
# Output: https://example.com?page=1&id=123
|
|
122
|
+
|
|
123
|
+
# With default domain for absolute paths
|
|
124
|
+
print(url_normalize("/images/logo.png", default_domain="example.com"))
|
|
125
|
+
# Output: https://example.com/images/logo.png
|
|
126
|
+
|
|
127
|
+
# With default domain and custom scheme
|
|
128
|
+
print(url_normalize("/images/logo.png", default_scheme="http", default_domain="example.com"))
|
|
129
|
+
# Output: http://example.com/images/logo.png
|
|
130
|
+
```
|
|
131
|
+
|
|
132
|
+
### Command-line Usage
|
|
133
|
+
|
|
134
|
+
You can also use `url-normalize` from the command line:
|
|
135
|
+
|
|
136
|
+
```bash
|
|
137
|
+
$ url-normalize "www.foo.com:80/foo"
|
|
138
|
+
# Output: https://www.foo.com/foo
|
|
139
|
+
|
|
140
|
+
# With custom default scheme
|
|
141
|
+
$ url-normalize -s http "www.foo.com/foo"
|
|
142
|
+
# Output: http://www.foo.com/foo
|
|
143
|
+
|
|
144
|
+
# With query parameter filtering
|
|
145
|
+
$ url-normalize -f "www.google.com/search?q=test&utm_source=test"
|
|
146
|
+
# Output: https://www.google.com/search?q=test
|
|
147
|
+
|
|
148
|
+
# With custom allowlist
|
|
149
|
+
$ url-normalize -f -p page,id "example.com?page=1&id=123&ref=test"
|
|
150
|
+
# Output: https://example.com/?page=1&id=123
|
|
151
|
+
|
|
152
|
+
# With default domain for absolute paths
|
|
153
|
+
$ url-normalize -d example.com "/images/logo.png"
|
|
154
|
+
# Output: https://example.com/images/logo.png
|
|
155
|
+
|
|
156
|
+
# With default domain and custom scheme
|
|
157
|
+
$ url-normalize -d example.com -s http "/images/logo.png"
|
|
158
|
+
# Output: http://example.com/images/logo.png
|
|
159
|
+
|
|
160
|
+
# Via uv tool/uvx
|
|
161
|
+
$ uvx url-normalize www.foo.com:80/foo
|
|
162
|
+
# Output: https://www.foo.com:80/foo
|
|
163
|
+
```
|
|
164
|
+
|
|
165
|
+
## Documentation
|
|
166
|
+
|
|
167
|
+
For a complete history of changes, see [CHANGELOG.md](CHANGELOG.md).
|
|
168
|
+
|
|
169
|
+
## Contributing
|
|
170
|
+
|
|
171
|
+
Contributions are welcome! Please feel free to submit a Pull Request.
|
|
172
|
+
|
|
173
|
+
## License
|
|
174
|
+
|
|
175
|
+
MIT License
|
|
@@ -0,0 +1,151 @@
|
|
|
1
|
+
# url-normalize
|
|
2
|
+
|
|
3
|
+
[](https://github.com/niksite/url-normalize/actions/workflows/ci.yml)
|
|
4
|
+
[](https://coveralls.io/r/niksite/url-normalize)
|
|
5
|
+
[](https://pypi.org/project/url-normalize/)
|
|
6
|
+
|
|
7
|
+
A Python library for standardizing and normalizing URLs with support for internationalized domain names (IDN).
|
|
8
|
+
|
|
9
|
+
## Table of Contents
|
|
10
|
+
|
|
11
|
+
- [Introduction](#introduction)
|
|
12
|
+
- [Features](#features)
|
|
13
|
+
- [Installation](#installation)
|
|
14
|
+
- [Usage](#usage)
|
|
15
|
+
- [Python API](#python-api)
|
|
16
|
+
- [Command Line](#command-line-usage)
|
|
17
|
+
- [Documentation](#documentation)
|
|
18
|
+
- [Contributing](#contributing)
|
|
19
|
+
- [License](#license)
|
|
20
|
+
|
|
21
|
+
## Introduction
|
|
22
|
+
|
|
23
|
+
url-normalize provides a robust URI normalization function that:
|
|
24
|
+
|
|
25
|
+
- Takes care of IDN domains.
|
|
26
|
+
- Always provides the URI scheme in lowercase characters.
|
|
27
|
+
- Always provides the host, if any, in lowercase characters.
|
|
28
|
+
- Only performs percent-encoding where it is essential.
|
|
29
|
+
- Always uses uppercase A-through-F characters when percent-encoding.
|
|
30
|
+
- Prevents dot-segments appearing in non-relative URI paths.
|
|
31
|
+
- For schemes that define a default authority, uses an empty authority if the
|
|
32
|
+
default is desired.
|
|
33
|
+
- For schemes that define an empty path to be equivalent to a path of "/",
|
|
34
|
+
uses "/".
|
|
35
|
+
- For schemes that define a port, uses an empty port if the default is desired
|
|
36
|
+
- Ensures all portions of the URI are utf-8 encoded NFC from Unicode strings
|
|
37
|
+
|
|
38
|
+
Inspired by Sam Ruby's [urlnorm.py](http://intertwingly.net/blog/2004/08/04/Urlnorm)
|
|
39
|
+
|
|
40
|
+
## Features
|
|
41
|
+
|
|
42
|
+
- **IDN Support**: Full internationalized domain name handling
|
|
43
|
+
- **Configurable Defaults**:
|
|
44
|
+
- Customizable default scheme (https by default)
|
|
45
|
+
- Configurable default domain for absolute paths
|
|
46
|
+
- **Query Parameter Control**:
|
|
47
|
+
- Parameter filtering with allowlists
|
|
48
|
+
- Support for domain-specific parameter rules
|
|
49
|
+
- **Versatile URL Handling**:
|
|
50
|
+
- Empty string URLs
|
|
51
|
+
- Double slash URLs (//domain.tld)
|
|
52
|
+
- Shebang (#!) URLs
|
|
53
|
+
- **Developer Friendly**:
|
|
54
|
+
- Cross-version Python compatibility (3.8+)
|
|
55
|
+
- 100% test coverage
|
|
56
|
+
- Modern type hints and string handling
|
|
57
|
+
|
|
58
|
+
## Installation
|
|
59
|
+
|
|
60
|
+
```sh
|
|
61
|
+
pip install url-normalize
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
## Usage
|
|
65
|
+
|
|
66
|
+
### Python API
|
|
67
|
+
|
|
68
|
+
```python
|
|
69
|
+
from url_normalize import url_normalize
|
|
70
|
+
|
|
71
|
+
# Basic normalization (uses https by default)
|
|
72
|
+
print(url_normalize("www.foo.com:80/foo"))
|
|
73
|
+
# Output: https://www.foo.com/foo
|
|
74
|
+
|
|
75
|
+
# With custom default scheme
|
|
76
|
+
print(url_normalize("www.foo.com/foo", default_scheme="http"))
|
|
77
|
+
# Output: http://www.foo.com/foo
|
|
78
|
+
|
|
79
|
+
# With query parameter filtering enabled
|
|
80
|
+
print(url_normalize("www.google.com/search?q=test&utm_source=test", filter_params=True))
|
|
81
|
+
# Output: https://www.google.com/search?q=test
|
|
82
|
+
|
|
83
|
+
# With custom parameter allowlist as a dict
|
|
84
|
+
print(url_normalize(
|
|
85
|
+
"example.com?page=1&id=123&ref=test",
|
|
86
|
+
filter_params=True,
|
|
87
|
+
param_allowlist={"example.com": ["page", "id"]}
|
|
88
|
+
))
|
|
89
|
+
# Output: https://example.com?page=1&id=123
|
|
90
|
+
|
|
91
|
+
# With custom parameter allowlist as a list
|
|
92
|
+
print(url_normalize(
|
|
93
|
+
"example.com?page=1&id=123&ref=test",
|
|
94
|
+
filter_params=True,
|
|
95
|
+
param_allowlist=["page", "id"]
|
|
96
|
+
))
|
|
97
|
+
# Output: https://example.com?page=1&id=123
|
|
98
|
+
|
|
99
|
+
# With default domain for absolute paths
|
|
100
|
+
print(url_normalize("/images/logo.png", default_domain="example.com"))
|
|
101
|
+
# Output: https://example.com/images/logo.png
|
|
102
|
+
|
|
103
|
+
# With default domain and custom scheme
|
|
104
|
+
print(url_normalize("/images/logo.png", default_scheme="http", default_domain="example.com"))
|
|
105
|
+
# Output: http://example.com/images/logo.png
|
|
106
|
+
```
|
|
107
|
+
|
|
108
|
+
### Command-line Usage
|
|
109
|
+
|
|
110
|
+
You can also use `url-normalize` from the command line:
|
|
111
|
+
|
|
112
|
+
```bash
|
|
113
|
+
$ url-normalize "www.foo.com:80/foo"
|
|
114
|
+
# Output: https://www.foo.com/foo
|
|
115
|
+
|
|
116
|
+
# With custom default scheme
|
|
117
|
+
$ url-normalize -s http "www.foo.com/foo"
|
|
118
|
+
# Output: http://www.foo.com/foo
|
|
119
|
+
|
|
120
|
+
# With query parameter filtering
|
|
121
|
+
$ url-normalize -f "www.google.com/search?q=test&utm_source=test"
|
|
122
|
+
# Output: https://www.google.com/search?q=test
|
|
123
|
+
|
|
124
|
+
# With custom allowlist
|
|
125
|
+
$ url-normalize -f -p page,id "example.com?page=1&id=123&ref=test"
|
|
126
|
+
# Output: https://example.com/?page=1&id=123
|
|
127
|
+
|
|
128
|
+
# With default domain for absolute paths
|
|
129
|
+
$ url-normalize -d example.com "/images/logo.png"
|
|
130
|
+
# Output: https://example.com/images/logo.png
|
|
131
|
+
|
|
132
|
+
# With default domain and custom scheme
|
|
133
|
+
$ url-normalize -d example.com -s http "/images/logo.png"
|
|
134
|
+
# Output: http://example.com/images/logo.png
|
|
135
|
+
|
|
136
|
+
# Via uv tool/uvx
|
|
137
|
+
$ uvx url-normalize www.foo.com:80/foo
|
|
138
|
+
# Output: https://www.foo.com:80/foo
|
|
139
|
+
```
|
|
140
|
+
|
|
141
|
+
## Documentation
|
|
142
|
+
|
|
143
|
+
For a complete history of changes, see [CHANGELOG.md](CHANGELOG.md).
|
|
144
|
+
|
|
145
|
+
## Contributing
|
|
146
|
+
|
|
147
|
+
Contributions are welcome! Please feel free to submit a Pull Request.
|
|
148
|
+
|
|
149
|
+
## License
|
|
150
|
+
|
|
151
|
+
MIT License
|
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "url-normalize"
|
|
3
|
-
version = "2.1
|
|
3
|
+
version = "2.2.1"
|
|
4
4
|
description = "URL normalization for Python"
|
|
5
5
|
authors = [{ name = "Nikolay Panov", email = "github@npanov.com" }]
|
|
6
6
|
license = { text = "MIT" }
|
|
7
7
|
readme = "README.md"
|
|
8
8
|
requires-python = ">=3.8"
|
|
9
|
-
keywords = ["url", "normalization", "normalize"]
|
|
9
|
+
keywords = ["url", "normalization", "normalize", "normalizer"]
|
|
10
10
|
dependencies = ["idna>=3.3"]
|
|
11
11
|
|
|
12
12
|
[project.urls]
|
|
@@ -19,16 +19,7 @@ Changelog = "https://github.com/niksite/url-normalize/blob/master/CHANGELOG.md"
|
|
|
19
19
|
url-normalize = "url_normalize.cli:main"
|
|
20
20
|
|
|
21
21
|
[project.optional-dependencies]
|
|
22
|
-
dev = [
|
|
23
|
-
"mypy",
|
|
24
|
-
"pre-commit",
|
|
25
|
-
"pytest-cov",
|
|
26
|
-
"pytest-ruff",
|
|
27
|
-
"pytest-socket",
|
|
28
|
-
"pytest",
|
|
29
|
-
"ruff",
|
|
30
|
-
"tox",
|
|
31
|
-
]
|
|
22
|
+
dev = ["mypy", "pre-commit", "pytest-cov", "pytest-socket", "pytest", "ruff"]
|
|
32
23
|
|
|
33
24
|
[tool.ruff]
|
|
34
25
|
target-version = "py38"
|
|
@@ -71,11 +62,9 @@ build-backend = "setuptools.build_meta"
|
|
|
71
62
|
|
|
72
63
|
[tool.pytest.ini_options]
|
|
73
64
|
addopts = [
|
|
74
|
-
"--cov-fail-under=100",
|
|
75
65
|
"--cov-report=term-missing:skip-covered",
|
|
76
66
|
"--cov=url_normalize",
|
|
77
67
|
"--disable-socket",
|
|
78
|
-
"--ruff",
|
|
79
68
|
"-v",
|
|
80
69
|
]
|
|
81
70
|
python_files = ["tests.py", "test_*.py", "*_tests.py"]
|
|
@@ -231,3 +231,64 @@ def test_cli_charset() -> None:
|
|
|
231
231
|
assert result_charset_short.returncode == 0
|
|
232
232
|
assert result_charset_short.stdout.strip() == expected_idn
|
|
233
233
|
assert not result_charset_short.stderr
|
|
234
|
+
|
|
235
|
+
|
|
236
|
+
def test_cli_default_domain() -> None:
|
|
237
|
+
"""Test adding default domain to absolute path via CLI."""
|
|
238
|
+
url = "/path/to/image.png"
|
|
239
|
+
expected = "https://example.com/path/to/image.png"
|
|
240
|
+
|
|
241
|
+
result = run_cli("--default-domain", "example.com", url)
|
|
242
|
+
|
|
243
|
+
assert result.returncode == 0
|
|
244
|
+
assert result.stdout.strip() == expected
|
|
245
|
+
assert not result.stderr
|
|
246
|
+
|
|
247
|
+
|
|
248
|
+
def test_cli_default_domain_short_arg() -> None:
|
|
249
|
+
"""Test adding default domain using short argument."""
|
|
250
|
+
url = "/path/to/image.png"
|
|
251
|
+
expected = "https://example.com/path/to/image.png"
|
|
252
|
+
|
|
253
|
+
result = run_cli("-d", "example.com", url)
|
|
254
|
+
|
|
255
|
+
assert result.returncode == 0
|
|
256
|
+
assert result.stdout.strip() == expected
|
|
257
|
+
assert not result.stderr
|
|
258
|
+
|
|
259
|
+
|
|
260
|
+
def test_cli_default_domain_with_scheme() -> None:
|
|
261
|
+
"""Test adding default domain with custom scheme."""
|
|
262
|
+
url = "/path/to/image.png"
|
|
263
|
+
expected = "http://example.com/path/to/image.png"
|
|
264
|
+
|
|
265
|
+
result = run_cli("-d", "example.com", "-s", "http", url)
|
|
266
|
+
|
|
267
|
+
assert result.returncode == 0
|
|
268
|
+
assert result.stdout.strip() == expected
|
|
269
|
+
assert not result.stderr
|
|
270
|
+
|
|
271
|
+
|
|
272
|
+
def test_cli_default_domain_no_effect_on_absolute_urls() -> None:
|
|
273
|
+
"""Test default domain has no effect on absolute URLs."""
|
|
274
|
+
url = "http://original-domain.com/path"
|
|
275
|
+
expected = "http://original-domain.com/path"
|
|
276
|
+
|
|
277
|
+
result = run_cli("-d", "example.com", url)
|
|
278
|
+
|
|
279
|
+
assert result.returncode == 0
|
|
280
|
+
assert result.stdout.strip() == expected
|
|
281
|
+
assert not result.stderr
|
|
282
|
+
|
|
283
|
+
|
|
284
|
+
def test_cli_default_domain_no_effect_on_relative_paths() -> None:
|
|
285
|
+
"""Test default domain has no effect on relative paths."""
|
|
286
|
+
url = "path/to/file.html"
|
|
287
|
+
# This becomes a regular URL with the default scheme
|
|
288
|
+
expected = "https://path/to/file.html"
|
|
289
|
+
|
|
290
|
+
result = run_cli("-d", "example.com", url)
|
|
291
|
+
|
|
292
|
+
assert result.returncode == 0
|
|
293
|
+
assert result.stdout.strip() == expected
|
|
294
|
+
assert not result.stderr
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
"""Deconstruct url tests."""
|
|
2
|
+
|
|
3
|
+
import pytest
|
|
4
|
+
|
|
5
|
+
from url_normalize.tools import URL, deconstruct_url
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
@pytest.mark.parametrize(
|
|
9
|
+
("url", "expected"),
|
|
10
|
+
[
|
|
11
|
+
(
|
|
12
|
+
"http://site.com",
|
|
13
|
+
URL(
|
|
14
|
+
fragment="",
|
|
15
|
+
host="site.com",
|
|
16
|
+
path="",
|
|
17
|
+
port="",
|
|
18
|
+
query="",
|
|
19
|
+
scheme="http",
|
|
20
|
+
userinfo="",
|
|
21
|
+
),
|
|
22
|
+
),
|
|
23
|
+
(
|
|
24
|
+
"http://user@www.example.com:8080/path/index.html?param=val#fragment",
|
|
25
|
+
URL(
|
|
26
|
+
fragment="fragment",
|
|
27
|
+
host="www.example.com",
|
|
28
|
+
path="/path/index.html",
|
|
29
|
+
port="8080",
|
|
30
|
+
query="param=val",
|
|
31
|
+
scheme="http",
|
|
32
|
+
userinfo="user@",
|
|
33
|
+
),
|
|
34
|
+
),
|
|
35
|
+
],
|
|
36
|
+
)
|
|
37
|
+
def test_deconstruct_url_result_is_expected(url: str, expected: URL) -> None:
|
|
38
|
+
"""Assert we got expected results from the deconstruct_url function."""
|
|
39
|
+
result = deconstruct_url(url)
|
|
40
|
+
assert result == expected, url
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
"""Tests for normalize_fragment function."""
|
|
2
|
+
|
|
3
|
+
import pytest
|
|
4
|
+
|
|
5
|
+
from url_normalize.url_normalize import normalize_fragment
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
@pytest.mark.parametrize(
|
|
9
|
+
("fragment", "expected"),
|
|
10
|
+
[
|
|
11
|
+
("", ""),
|
|
12
|
+
("fragment", "fragment"),
|
|
13
|
+
("пример", "%D0%BF%D1%80%D0%B8%D0%BC%D0%B5%D1%80"),
|
|
14
|
+
("!fragment", "%21fragment"),
|
|
15
|
+
("~fragment", "~fragment"),
|
|
16
|
+
# Issue #36: Equal sign should not be encoded
|
|
17
|
+
("gid=1234", "gid=1234"),
|
|
18
|
+
],
|
|
19
|
+
)
|
|
20
|
+
def test_normalize_fragment_result_is_expected(fragment: str, expected: str) -> None:
|
|
21
|
+
"""Assert we got expected results from the normalize_fragment function."""
|
|
22
|
+
result = normalize_fragment(fragment)
|
|
23
|
+
assert result == expected, fragment
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
"""Tests for normalize_host function."""
|
|
2
|
+
|
|
3
|
+
import pytest
|
|
4
|
+
|
|
5
|
+
from url_normalize.url_normalize import normalize_host
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
@pytest.mark.parametrize(
|
|
9
|
+
("host", "expected"),
|
|
10
|
+
[
|
|
11
|
+
# Basic cases
|
|
12
|
+
("site.com", "site.com"),
|
|
13
|
+
("SITE.COM", "site.com"),
|
|
14
|
+
("site.com.", "site.com"),
|
|
15
|
+
# Cyrillic domains
|
|
16
|
+
("пример.испытание", "xn--e1afmkfd.xn--80akhbyknj4f"),
|
|
17
|
+
# Mixed case with Cyrillic
|
|
18
|
+
("ExAmPle.РФ", "example.xn--p1ai"),
|
|
19
|
+
# IDNA2008 with UTS46
|
|
20
|
+
("faß.de", "fass.de"), # Normalize using transitional rules
|
|
21
|
+
# Edge cases
|
|
22
|
+
("ドメイン.テスト", "xn--eckwd4c7c.xn--zckzah"), # Japanese
|
|
23
|
+
("domain.café", "domain.xn--caf-dma"), # Latin with diacritic
|
|
24
|
+
# Normalization tests
|
|
25
|
+
("über.example", "xn--ber-goa.example"), # IDNA 2008 for umlaut
|
|
26
|
+
("example。com", "example.com"), # Normalize full-width punctuation
|
|
27
|
+
],
|
|
28
|
+
)
|
|
29
|
+
def test_normalize_host_result_is_expected(host: str, expected: str) -> None:
|
|
30
|
+
"""Assert we got expected results from the normalize_host function."""
|
|
31
|
+
result = normalize_host(host)
|
|
32
|
+
assert result == expected, host
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
"""Tests for normalize_path function."""
|
|
2
|
+
|
|
3
|
+
import pytest
|
|
4
|
+
|
|
5
|
+
from url_normalize.url_normalize import normalize_path
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
@pytest.mark.parametrize(
|
|
9
|
+
("path", "expected"),
|
|
10
|
+
[
|
|
11
|
+
("..", "/"),
|
|
12
|
+
("", "/"),
|
|
13
|
+
("/../foo", "/foo"),
|
|
14
|
+
("/..foo", "/..foo"),
|
|
15
|
+
("/./../foo", "/foo"),
|
|
16
|
+
("/./foo", "/foo"),
|
|
17
|
+
("/./foo/.", "/foo/"),
|
|
18
|
+
("/.foo", "/.foo"),
|
|
19
|
+
("/", "/"),
|
|
20
|
+
("/foo..", "/foo.."),
|
|
21
|
+
("/foo.", "/foo."),
|
|
22
|
+
("/FOO", "/FOO"),
|
|
23
|
+
("/foo/../bar", "/bar"),
|
|
24
|
+
("/foo/./bar", "/foo/bar"),
|
|
25
|
+
("/foo//", "/foo/"),
|
|
26
|
+
("/foo///bar//", "/foo/bar/"),
|
|
27
|
+
("/foo/bar/..", "/foo/"),
|
|
28
|
+
("/foo/bar/../..", "/"),
|
|
29
|
+
("/foo/bar/../../../../baz", "/baz"),
|
|
30
|
+
("/foo/bar/../../../baz", "/baz"),
|
|
31
|
+
("/foo/bar/../../", "/"),
|
|
32
|
+
("/foo/bar/../../baz", "/baz"),
|
|
33
|
+
("/foo/bar/../", "/foo/"),
|
|
34
|
+
("/foo/bar/../baz", "/foo/baz"),
|
|
35
|
+
("/foo/bar/.", "/foo/bar/"),
|
|
36
|
+
("/foo/bar/./", "/foo/bar/"),
|
|
37
|
+
# Issue #25: we should preserve ? in the path
|
|
38
|
+
("/More+Tea+Vicar%3F/discussion", "/More+Tea+Vicar%3F/discussion"),
|
|
39
|
+
],
|
|
40
|
+
)
|
|
41
|
+
def test_normalize_path_result_is_expected(path: str, expected: str) -> None:
|
|
42
|
+
"""Assert we got expected results from the normalize_path function."""
|
|
43
|
+
result = normalize_path(path, "http")
|
|
44
|
+
assert result == expected, path
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
"""Tests for normalize_port function."""
|
|
2
|
+
|
|
3
|
+
import pytest
|
|
4
|
+
|
|
5
|
+
from url_normalize.url_normalize import normalize_port
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
@pytest.mark.parametrize(
|
|
9
|
+
("port", "expected"),
|
|
10
|
+
[
|
|
11
|
+
("8080", "8080"), # Non-default port
|
|
12
|
+
("", ""), # Empty port
|
|
13
|
+
("80", ""), # Default HTTP port
|
|
14
|
+
("string", "string"), # Non-numeric port (should pass through)
|
|
15
|
+
# Add more cases as needed, e.g., for HTTPS
|
|
16
|
+
pytest.param("443", "", id="https_default_port"),
|
|
17
|
+
],
|
|
18
|
+
)
|
|
19
|
+
def test_normalize_port_result_is_expected(port: str, expected: str):
|
|
20
|
+
"""Assert we got expected results from the normalize_port function."""
|
|
21
|
+
# Test with 'http' scheme for most cases
|
|
22
|
+
scheme = "https" if port == "443" else "http"
|
|
23
|
+
|
|
24
|
+
result = normalize_port(port, scheme)
|
|
25
|
+
|
|
26
|
+
assert result == expected
|
|
@@ -14,6 +14,8 @@ from url_normalize.url_normalize import normalize_query
|
|
|
14
14
|
("Ç=Ç", "%C3%87=%C3%87"),
|
|
15
15
|
("%C3%87=%C3%87", "%C3%87=%C3%87"),
|
|
16
16
|
("q=C%CC%A7", "q=%C3%87"),
|
|
17
|
+
("q=%23test", "q=%23test"), # Preserve encoded # in value, #31
|
|
18
|
+
("where=code%3D123", "where=code%3D123"), # Preserve encoded = in value, #25
|
|
17
19
|
],
|
|
18
20
|
)
|
|
19
21
|
def test_normalize_query_result_is_expected(query, expected):
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
"""Tests for normalize_scheme function."""
|
|
2
|
+
|
|
3
|
+
import pytest
|
|
4
|
+
|
|
5
|
+
from url_normalize.url_normalize import normalize_scheme
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
@pytest.mark.parametrize(
|
|
9
|
+
("scheme", "expected"),
|
|
10
|
+
[
|
|
11
|
+
("http", "http"),
|
|
12
|
+
("HTTP", "http"),
|
|
13
|
+
],
|
|
14
|
+
)
|
|
15
|
+
def test_normalize_scheme_result_is_expected(scheme: str, expected: str) -> None:
|
|
16
|
+
"""Assert we got expected results from the normalize_scheme function."""
|
|
17
|
+
result = normalize_scheme(scheme)
|
|
18
|
+
assert result == expected, scheme
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
"""Tests for normalize_userinfo function."""
|
|
2
|
+
|
|
3
|
+
import pytest
|
|
4
|
+
|
|
5
|
+
from url_normalize.url_normalize import normalize_userinfo
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
@pytest.mark.parametrize(
|
|
9
|
+
("userinfo", "expected"),
|
|
10
|
+
[
|
|
11
|
+
(":@", ""),
|
|
12
|
+
("", ""),
|
|
13
|
+
("@", ""),
|
|
14
|
+
("user:password@", "user:password@"),
|
|
15
|
+
("user@", "user@"),
|
|
16
|
+
],
|
|
17
|
+
)
|
|
18
|
+
def test_normalize_userinfo_result_is_expected(userinfo: str, expected: str) -> None:
|
|
19
|
+
"""Assert we got expected results from the normalize_userinfo function."""
|
|
20
|
+
result = normalize_userinfo(userinfo)
|
|
21
|
+
assert result == expected, userinfo
|