url-normalize 2.1.0__tar.gz → 2.2.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Potentially problematic release.


This version of url-normalize might be problematic. Click here for more details.

Files changed (56) hide show
  1. url_normalize-2.2.1/PKG-INFO +175 -0
  2. url_normalize-2.2.1/README.md +151 -0
  3. {url_normalize-2.1.0 → url_normalize-2.2.1}/pyproject.toml +3 -14
  4. {url_normalize-2.1.0 → url_normalize-2.2.1}/tests/test_cli.py +61 -0
  5. url_normalize-2.2.1/tests/test_deconstruct_url.py +40 -0
  6. url_normalize-2.2.1/tests/test_normalize_fragment.py +23 -0
  7. url_normalize-2.2.1/tests/test_normalize_host.py +32 -0
  8. url_normalize-2.2.1/tests/test_normalize_path.py +44 -0
  9. url_normalize-2.2.1/tests/test_normalize_port.py +26 -0
  10. {url_normalize-2.1.0 → url_normalize-2.2.1}/tests/test_normalize_query.py +2 -0
  11. url_normalize-2.2.1/tests/test_normalize_scheme.py +18 -0
  12. url_normalize-2.2.1/tests/test_normalize_userinfo.py +21 -0
  13. url_normalize-2.2.1/tests/test_provide_url_domain.py +31 -0
  14. url_normalize-2.2.1/tests/test_provide_url_scheme.py +32 -0
  15. url_normalize-2.2.1/tests/test_reconstruct_url.py +42 -0
  16. url_normalize-2.2.1/tests/test_url_normalize.py +163 -0
  17. {url_normalize-2.1.0 → url_normalize-2.2.1}/url_normalize/__init__.py +1 -1
  18. {url_normalize-2.1.0 → url_normalize-2.2.1}/url_normalize/cli.py +7 -0
  19. {url_normalize-2.1.0 → url_normalize-2.2.1}/url_normalize/normalize_path.py +4 -4
  20. {url_normalize-2.1.0 → url_normalize-2.2.1}/url_normalize/normalize_query.py +2 -2
  21. url_normalize-2.2.1/url_normalize/provide_url_domain.py +28 -0
  22. {url_normalize-2.1.0 → url_normalize-2.2.1}/url_normalize/provide_url_scheme.py +11 -4
  23. url_normalize-2.2.1/url_normalize/py.typed +0 -0
  24. {url_normalize-2.1.0 → url_normalize-2.2.1}/url_normalize/url_normalize.py +7 -2
  25. url_normalize-2.2.1/url_normalize.egg-info/PKG-INFO +175 -0
  26. {url_normalize-2.1.0 → url_normalize-2.2.1}/url_normalize.egg-info/SOURCES.txt +3 -0
  27. {url_normalize-2.1.0 → url_normalize-2.2.1}/url_normalize.egg-info/requires.txt +0 -2
  28. url_normalize-2.1.0/PKG-INFO +0 -133
  29. url_normalize-2.1.0/README.md +0 -107
  30. url_normalize-2.1.0/tests/test_deconstruct_url.py +0 -35
  31. url_normalize-2.1.0/tests/test_normalize_fragment.py +0 -21
  32. url_normalize-2.1.0/tests/test_normalize_host.py +0 -29
  33. url_normalize-2.1.0/tests/test_normalize_path.py +0 -40
  34. url_normalize-2.1.0/tests/test_normalize_port.py +0 -13
  35. url_normalize-2.1.0/tests/test_normalize_scheme.py +0 -13
  36. url_normalize-2.1.0/tests/test_normalize_userinfo.py +0 -19
  37. url_normalize-2.1.0/tests/test_provide_url_scheme.py +0 -30
  38. url_normalize-2.1.0/tests/test_reconstruct_url.py +0 -41
  39. url_normalize-2.1.0/tests/test_url_normalize.py +0 -127
  40. url_normalize-2.1.0/url_normalize.egg-info/PKG-INFO +0 -133
  41. {url_normalize-2.1.0 → url_normalize-2.2.1}/LICENSE +0 -0
  42. {url_normalize-2.1.0 → url_normalize-2.2.1}/setup.cfg +0 -0
  43. {url_normalize-2.1.0 → url_normalize-2.2.1}/tests/test_generic_url_cleanup.py +0 -0
  44. {url_normalize-2.1.0 → url_normalize-2.2.1}/tests/test_normalize_query_filters.py +0 -0
  45. {url_normalize-2.1.0 → url_normalize-2.2.1}/tests/test_tools.py +0 -0
  46. {url_normalize-2.1.0 → url_normalize-2.2.1}/url_normalize/generic_url_cleanup.py +0 -0
  47. {url_normalize-2.1.0 → url_normalize-2.2.1}/url_normalize/normalize_fragment.py +0 -0
  48. {url_normalize-2.1.0 → url_normalize-2.2.1}/url_normalize/normalize_host.py +0 -0
  49. {url_normalize-2.1.0 → url_normalize-2.2.1}/url_normalize/normalize_port.py +0 -0
  50. {url_normalize-2.1.0 → url_normalize-2.2.1}/url_normalize/normalize_scheme.py +0 -0
  51. {url_normalize-2.1.0 → url_normalize-2.2.1}/url_normalize/normalize_userinfo.py +0 -0
  52. {url_normalize-2.1.0 → url_normalize-2.2.1}/url_normalize/param_allowlist.py +0 -0
  53. {url_normalize-2.1.0 → url_normalize-2.2.1}/url_normalize/tools.py +0 -0
  54. {url_normalize-2.1.0 → url_normalize-2.2.1}/url_normalize.egg-info/dependency_links.txt +0 -0
  55. {url_normalize-2.1.0 → url_normalize-2.2.1}/url_normalize.egg-info/entry_points.txt +0 -0
  56. {url_normalize-2.1.0 → url_normalize-2.2.1}/url_normalize.egg-info/top_level.txt +0 -0
@@ -0,0 +1,175 @@
1
+ Metadata-Version: 2.4
2
+ Name: url-normalize
3
+ Version: 2.2.1
4
+ Summary: URL normalization for Python
5
+ Author-email: Nikolay Panov <github@npanov.com>
6
+ License: MIT
7
+ Project-URL: Homepage, https://github.com/niksite/url-normalize
8
+ Project-URL: Repository, https://github.com/niksite/url-normalize
9
+ Project-URL: Issues, https://github.com/niksite/url-normalize/issues
10
+ Project-URL: Changelog, https://github.com/niksite/url-normalize/blob/master/CHANGELOG.md
11
+ Keywords: url,normalization,normalize,normalizer
12
+ Requires-Python: >=3.8
13
+ Description-Content-Type: text/markdown
14
+ License-File: LICENSE
15
+ Requires-Dist: idna>=3.3
16
+ Provides-Extra: dev
17
+ Requires-Dist: mypy; extra == "dev"
18
+ Requires-Dist: pre-commit; extra == "dev"
19
+ Requires-Dist: pytest-cov; extra == "dev"
20
+ Requires-Dist: pytest-socket; extra == "dev"
21
+ Requires-Dist: pytest; extra == "dev"
22
+ Requires-Dist: ruff; extra == "dev"
23
+ Dynamic: license-file
24
+
25
+ # url-normalize
26
+
27
+ [![tests](https://github.com/niksite/url-normalize/actions/workflows/ci.yml/badge.svg)](https://github.com/niksite/url-normalize/actions/workflows/ci.yml)
28
+ [![Coveralls](https://img.shields.io/coveralls/github/niksite/url-normalize/master.svg)](https://coveralls.io/r/niksite/url-normalize)
29
+ [![PyPI](https://img.shields.io/pypi/v/url-normalize.svg)](https://pypi.org/project/url-normalize/)
30
+
31
+ A Python library for standardizing and normalizing URLs with support for internationalized domain names (IDN).
32
+
33
+ ## Table of Contents
34
+
35
+ - [Introduction](#introduction)
36
+ - [Features](#features)
37
+ - [Installation](#installation)
38
+ - [Usage](#usage)
39
+ - [Python API](#python-api)
40
+ - [Command Line](#command-line-usage)
41
+ - [Documentation](#documentation)
42
+ - [Contributing](#contributing)
43
+ - [License](#license)
44
+
45
+ ## Introduction
46
+
47
+ url-normalize provides a robust URI normalization function that:
48
+
49
+ - Takes care of IDN domains.
50
+ - Always provides the URI scheme in lowercase characters.
51
+ - Always provides the host, if any, in lowercase characters.
52
+ - Only performs percent-encoding where it is essential.
53
+ - Always uses uppercase A-through-F characters when percent-encoding.
54
+ - Prevents dot-segments appearing in non-relative URI paths.
55
+ - For schemes that define a default authority, uses an empty authority if the
56
+ default is desired.
57
+ - For schemes that define an empty path to be equivalent to a path of "/",
58
+ uses "/".
59
+ - For schemes that define a port, uses an empty port if the default is desired
60
+ - Ensures all portions of the URI are utf-8 encoded NFC from Unicode strings
61
+
62
+ Inspired by Sam Ruby's [urlnorm.py](http://intertwingly.net/blog/2004/08/04/Urlnorm)
63
+
64
+ ## Features
65
+
66
+ - **IDN Support**: Full internationalized domain name handling
67
+ - **Configurable Defaults**:
68
+ - Customizable default scheme (https by default)
69
+ - Configurable default domain for absolute paths
70
+ - **Query Parameter Control**:
71
+ - Parameter filtering with allowlists
72
+ - Support for domain-specific parameter rules
73
+ - **Versatile URL Handling**:
74
+ - Empty string URLs
75
+ - Double slash URLs (//domain.tld)
76
+ - Shebang (#!) URLs
77
+ - **Developer Friendly**:
78
+ - Cross-version Python compatibility (3.8+)
79
+ - 100% test coverage
80
+ - Modern type hints and string handling
81
+
82
+ ## Installation
83
+
84
+ ```sh
85
+ pip install url-normalize
86
+ ```
87
+
88
+ ## Usage
89
+
90
+ ### Python API
91
+
92
+ ```python
93
+ from url_normalize import url_normalize
94
+
95
+ # Basic normalization (uses https by default)
96
+ print(url_normalize("www.foo.com:80/foo"))
97
+ # Output: https://www.foo.com/foo
98
+
99
+ # With custom default scheme
100
+ print(url_normalize("www.foo.com/foo", default_scheme="http"))
101
+ # Output: http://www.foo.com/foo
102
+
103
+ # With query parameter filtering enabled
104
+ print(url_normalize("www.google.com/search?q=test&utm_source=test", filter_params=True))
105
+ # Output: https://www.google.com/search?q=test
106
+
107
+ # With custom parameter allowlist as a dict
108
+ print(url_normalize(
109
+ "example.com?page=1&id=123&ref=test",
110
+ filter_params=True,
111
+ param_allowlist={"example.com": ["page", "id"]}
112
+ ))
113
+ # Output: https://example.com?page=1&id=123
114
+
115
+ # With custom parameter allowlist as a list
116
+ print(url_normalize(
117
+ "example.com?page=1&id=123&ref=test",
118
+ filter_params=True,
119
+ param_allowlist=["page", "id"]
120
+ ))
121
+ # Output: https://example.com?page=1&id=123
122
+
123
+ # With default domain for absolute paths
124
+ print(url_normalize("/images/logo.png", default_domain="example.com"))
125
+ # Output: https://example.com/images/logo.png
126
+
127
+ # With default domain and custom scheme
128
+ print(url_normalize("/images/logo.png", default_scheme="http", default_domain="example.com"))
129
+ # Output: http://example.com/images/logo.png
130
+ ```
131
+
132
+ ### Command-line Usage
133
+
134
+ You can also use `url-normalize` from the command line:
135
+
136
+ ```bash
137
+ $ url-normalize "www.foo.com:80/foo"
138
+ # Output: https://www.foo.com/foo
139
+
140
+ # With custom default scheme
141
+ $ url-normalize -s http "www.foo.com/foo"
142
+ # Output: http://www.foo.com/foo
143
+
144
+ # With query parameter filtering
145
+ $ url-normalize -f "www.google.com/search?q=test&utm_source=test"
146
+ # Output: https://www.google.com/search?q=test
147
+
148
+ # With custom allowlist
149
+ $ url-normalize -f -p page,id "example.com?page=1&id=123&ref=test"
150
+ # Output: https://example.com/?page=1&id=123
151
+
152
+ # With default domain for absolute paths
153
+ $ url-normalize -d example.com "/images/logo.png"
154
+ # Output: https://example.com/images/logo.png
155
+
156
+ # With default domain and custom scheme
157
+ $ url-normalize -d example.com -s http "/images/logo.png"
158
+ # Output: http://example.com/images/logo.png
159
+
160
+ # Via uv tool/uvx
161
+ $ uvx url-normalize www.foo.com:80/foo
162
+ # Output: https://www.foo.com:80/foo
163
+ ```
164
+
165
+ ## Documentation
166
+
167
+ For a complete history of changes, see [CHANGELOG.md](CHANGELOG.md).
168
+
169
+ ## Contributing
170
+
171
+ Contributions are welcome! Please feel free to submit a Pull Request.
172
+
173
+ ## License
174
+
175
+ MIT License
@@ -0,0 +1,151 @@
1
+ # url-normalize
2
+
3
+ [![tests](https://github.com/niksite/url-normalize/actions/workflows/ci.yml/badge.svg)](https://github.com/niksite/url-normalize/actions/workflows/ci.yml)
4
+ [![Coveralls](https://img.shields.io/coveralls/github/niksite/url-normalize/master.svg)](https://coveralls.io/r/niksite/url-normalize)
5
+ [![PyPI](https://img.shields.io/pypi/v/url-normalize.svg)](https://pypi.org/project/url-normalize/)
6
+
7
+ A Python library for standardizing and normalizing URLs with support for internationalized domain names (IDN).
8
+
9
+ ## Table of Contents
10
+
11
+ - [Introduction](#introduction)
12
+ - [Features](#features)
13
+ - [Installation](#installation)
14
+ - [Usage](#usage)
15
+ - [Python API](#python-api)
16
+ - [Command Line](#command-line-usage)
17
+ - [Documentation](#documentation)
18
+ - [Contributing](#contributing)
19
+ - [License](#license)
20
+
21
+ ## Introduction
22
+
23
+ url-normalize provides a robust URI normalization function that:
24
+
25
+ - Takes care of IDN domains.
26
+ - Always provides the URI scheme in lowercase characters.
27
+ - Always provides the host, if any, in lowercase characters.
28
+ - Only performs percent-encoding where it is essential.
29
+ - Always uses uppercase A-through-F characters when percent-encoding.
30
+ - Prevents dot-segments appearing in non-relative URI paths.
31
+ - For schemes that define a default authority, uses an empty authority if the
32
+ default is desired.
33
+ - For schemes that define an empty path to be equivalent to a path of "/",
34
+ uses "/".
35
+ - For schemes that define a port, uses an empty port if the default is desired
36
+ - Ensures all portions of the URI are utf-8 encoded NFC from Unicode strings
37
+
38
+ Inspired by Sam Ruby's [urlnorm.py](http://intertwingly.net/blog/2004/08/04/Urlnorm)
39
+
40
+ ## Features
41
+
42
+ - **IDN Support**: Full internationalized domain name handling
43
+ - **Configurable Defaults**:
44
+ - Customizable default scheme (https by default)
45
+ - Configurable default domain for absolute paths
46
+ - **Query Parameter Control**:
47
+ - Parameter filtering with allowlists
48
+ - Support for domain-specific parameter rules
49
+ - **Versatile URL Handling**:
50
+ - Empty string URLs
51
+ - Double slash URLs (//domain.tld)
52
+ - Shebang (#!) URLs
53
+ - **Developer Friendly**:
54
+ - Cross-version Python compatibility (3.8+)
55
+ - 100% test coverage
56
+ - Modern type hints and string handling
57
+
58
+ ## Installation
59
+
60
+ ```sh
61
+ pip install url-normalize
62
+ ```
63
+
64
+ ## Usage
65
+
66
+ ### Python API
67
+
68
+ ```python
69
+ from url_normalize import url_normalize
70
+
71
+ # Basic normalization (uses https by default)
72
+ print(url_normalize("www.foo.com:80/foo"))
73
+ # Output: https://www.foo.com/foo
74
+
75
+ # With custom default scheme
76
+ print(url_normalize("www.foo.com/foo", default_scheme="http"))
77
+ # Output: http://www.foo.com/foo
78
+
79
+ # With query parameter filtering enabled
80
+ print(url_normalize("www.google.com/search?q=test&utm_source=test", filter_params=True))
81
+ # Output: https://www.google.com/search?q=test
82
+
83
+ # With custom parameter allowlist as a dict
84
+ print(url_normalize(
85
+ "example.com?page=1&id=123&ref=test",
86
+ filter_params=True,
87
+ param_allowlist={"example.com": ["page", "id"]}
88
+ ))
89
+ # Output: https://example.com?page=1&id=123
90
+
91
+ # With custom parameter allowlist as a list
92
+ print(url_normalize(
93
+ "example.com?page=1&id=123&ref=test",
94
+ filter_params=True,
95
+ param_allowlist=["page", "id"]
96
+ ))
97
+ # Output: https://example.com?page=1&id=123
98
+
99
+ # With default domain for absolute paths
100
+ print(url_normalize("/images/logo.png", default_domain="example.com"))
101
+ # Output: https://example.com/images/logo.png
102
+
103
+ # With default domain and custom scheme
104
+ print(url_normalize("/images/logo.png", default_scheme="http", default_domain="example.com"))
105
+ # Output: http://example.com/images/logo.png
106
+ ```
107
+
108
+ ### Command-line Usage
109
+
110
+ You can also use `url-normalize` from the command line:
111
+
112
+ ```bash
113
+ $ url-normalize "www.foo.com:80/foo"
114
+ # Output: https://www.foo.com/foo
115
+
116
+ # With custom default scheme
117
+ $ url-normalize -s http "www.foo.com/foo"
118
+ # Output: http://www.foo.com/foo
119
+
120
+ # With query parameter filtering
121
+ $ url-normalize -f "www.google.com/search?q=test&utm_source=test"
122
+ # Output: https://www.google.com/search?q=test
123
+
124
+ # With custom allowlist
125
+ $ url-normalize -f -p page,id "example.com?page=1&id=123&ref=test"
126
+ # Output: https://example.com/?page=1&id=123
127
+
128
+ # With default domain for absolute paths
129
+ $ url-normalize -d example.com "/images/logo.png"
130
+ # Output: https://example.com/images/logo.png
131
+
132
+ # With default domain and custom scheme
133
+ $ url-normalize -d example.com -s http "/images/logo.png"
134
+ # Output: http://example.com/images/logo.png
135
+
136
+ # Via uv tool/uvx
137
+ $ uvx url-normalize www.foo.com:80/foo
138
+ # Output: https://www.foo.com:80/foo
139
+ ```
140
+
141
+ ## Documentation
142
+
143
+ For a complete history of changes, see [CHANGELOG.md](CHANGELOG.md).
144
+
145
+ ## Contributing
146
+
147
+ Contributions are welcome! Please feel free to submit a Pull Request.
148
+
149
+ ## License
150
+
151
+ MIT License
@@ -1,12 +1,12 @@
1
1
  [project]
2
2
  name = "url-normalize"
3
- version = "2.1.0"
3
+ version = "2.2.1"
4
4
  description = "URL normalization for Python"
5
5
  authors = [{ name = "Nikolay Panov", email = "github@npanov.com" }]
6
6
  license = { text = "MIT" }
7
7
  readme = "README.md"
8
8
  requires-python = ">=3.8"
9
- keywords = ["url", "normalization", "normalize"]
9
+ keywords = ["url", "normalization", "normalize", "normalizer"]
10
10
  dependencies = ["idna>=3.3"]
11
11
 
12
12
  [project.urls]
@@ -19,16 +19,7 @@ Changelog = "https://github.com/niksite/url-normalize/blob/master/CHANGELOG.md"
19
19
  url-normalize = "url_normalize.cli:main"
20
20
 
21
21
  [project.optional-dependencies]
22
- dev = [
23
- "mypy",
24
- "pre-commit",
25
- "pytest-cov",
26
- "pytest-ruff",
27
- "pytest-socket",
28
- "pytest",
29
- "ruff",
30
- "tox",
31
- ]
22
+ dev = ["mypy", "pre-commit", "pytest-cov", "pytest-socket", "pytest", "ruff"]
32
23
 
33
24
  [tool.ruff]
34
25
  target-version = "py38"
@@ -71,11 +62,9 @@ build-backend = "setuptools.build_meta"
71
62
 
72
63
  [tool.pytest.ini_options]
73
64
  addopts = [
74
- "--cov-fail-under=100",
75
65
  "--cov-report=term-missing:skip-covered",
76
66
  "--cov=url_normalize",
77
67
  "--disable-socket",
78
- "--ruff",
79
68
  "-v",
80
69
  ]
81
70
  python_files = ["tests.py", "test_*.py", "*_tests.py"]
@@ -231,3 +231,64 @@ def test_cli_charset() -> None:
231
231
  assert result_charset_short.returncode == 0
232
232
  assert result_charset_short.stdout.strip() == expected_idn
233
233
  assert not result_charset_short.stderr
234
+
235
+
236
+ def test_cli_default_domain() -> None:
237
+ """Test adding default domain to absolute path via CLI."""
238
+ url = "/path/to/image.png"
239
+ expected = "https://example.com/path/to/image.png"
240
+
241
+ result = run_cli("--default-domain", "example.com", url)
242
+
243
+ assert result.returncode == 0
244
+ assert result.stdout.strip() == expected
245
+ assert not result.stderr
246
+
247
+
248
+ def test_cli_default_domain_short_arg() -> None:
249
+ """Test adding default domain using short argument."""
250
+ url = "/path/to/image.png"
251
+ expected = "https://example.com/path/to/image.png"
252
+
253
+ result = run_cli("-d", "example.com", url)
254
+
255
+ assert result.returncode == 0
256
+ assert result.stdout.strip() == expected
257
+ assert not result.stderr
258
+
259
+
260
+ def test_cli_default_domain_with_scheme() -> None:
261
+ """Test adding default domain with custom scheme."""
262
+ url = "/path/to/image.png"
263
+ expected = "http://example.com/path/to/image.png"
264
+
265
+ result = run_cli("-d", "example.com", "-s", "http", url)
266
+
267
+ assert result.returncode == 0
268
+ assert result.stdout.strip() == expected
269
+ assert not result.stderr
270
+
271
+
272
+ def test_cli_default_domain_no_effect_on_absolute_urls() -> None:
273
+ """Test default domain has no effect on absolute URLs."""
274
+ url = "http://original-domain.com/path"
275
+ expected = "http://original-domain.com/path"
276
+
277
+ result = run_cli("-d", "example.com", url)
278
+
279
+ assert result.returncode == 0
280
+ assert result.stdout.strip() == expected
281
+ assert not result.stderr
282
+
283
+
284
+ def test_cli_default_domain_no_effect_on_relative_paths() -> None:
285
+ """Test default domain has no effect on relative paths."""
286
+ url = "path/to/file.html"
287
+ # This becomes a regular URL with the default scheme
288
+ expected = "https://path/to/file.html"
289
+
290
+ result = run_cli("-d", "example.com", url)
291
+
292
+ assert result.returncode == 0
293
+ assert result.stdout.strip() == expected
294
+ assert not result.stderr
@@ -0,0 +1,40 @@
1
+ """Deconstruct url tests."""
2
+
3
+ import pytest
4
+
5
+ from url_normalize.tools import URL, deconstruct_url
6
+
7
+
8
+ @pytest.mark.parametrize(
9
+ ("url", "expected"),
10
+ [
11
+ (
12
+ "http://site.com",
13
+ URL(
14
+ fragment="",
15
+ host="site.com",
16
+ path="",
17
+ port="",
18
+ query="",
19
+ scheme="http",
20
+ userinfo="",
21
+ ),
22
+ ),
23
+ (
24
+ "http://user@www.example.com:8080/path/index.html?param=val#fragment",
25
+ URL(
26
+ fragment="fragment",
27
+ host="www.example.com",
28
+ path="/path/index.html",
29
+ port="8080",
30
+ query="param=val",
31
+ scheme="http",
32
+ userinfo="user@",
33
+ ),
34
+ ),
35
+ ],
36
+ )
37
+ def test_deconstruct_url_result_is_expected(url: str, expected: URL) -> None:
38
+ """Assert we got expected results from the deconstruct_url function."""
39
+ result = deconstruct_url(url)
40
+ assert result == expected, url
@@ -0,0 +1,23 @@
1
+ """Tests for normalize_fragment function."""
2
+
3
+ import pytest
4
+
5
+ from url_normalize.url_normalize import normalize_fragment
6
+
7
+
8
+ @pytest.mark.parametrize(
9
+ ("fragment", "expected"),
10
+ [
11
+ ("", ""),
12
+ ("fragment", "fragment"),
13
+ ("пример", "%D0%BF%D1%80%D0%B8%D0%BC%D0%B5%D1%80"),
14
+ ("!fragment", "%21fragment"),
15
+ ("~fragment", "~fragment"),
16
+ # Issue #36: Equal sign should not be encoded
17
+ ("gid=1234", "gid=1234"),
18
+ ],
19
+ )
20
+ def test_normalize_fragment_result_is_expected(fragment: str, expected: str) -> None:
21
+ """Assert we got expected results from the normalize_fragment function."""
22
+ result = normalize_fragment(fragment)
23
+ assert result == expected, fragment
@@ -0,0 +1,32 @@
1
+ """Tests for normalize_host function."""
2
+
3
+ import pytest
4
+
5
+ from url_normalize.url_normalize import normalize_host
6
+
7
+
8
+ @pytest.mark.parametrize(
9
+ ("host", "expected"),
10
+ [
11
+ # Basic cases
12
+ ("site.com", "site.com"),
13
+ ("SITE.COM", "site.com"),
14
+ ("site.com.", "site.com"),
15
+ # Cyrillic domains
16
+ ("пример.испытание", "xn--e1afmkfd.xn--80akhbyknj4f"),
17
+ # Mixed case with Cyrillic
18
+ ("ExAmPle.РФ", "example.xn--p1ai"),
19
+ # IDNA2008 with UTS46
20
+ ("faß.de", "fass.de"), # Normalize using transitional rules
21
+ # Edge cases
22
+ ("ドメイン.テスト", "xn--eckwd4c7c.xn--zckzah"), # Japanese
23
+ ("domain.café", "domain.xn--caf-dma"), # Latin with diacritic
24
+ # Normalization tests
25
+ ("über.example", "xn--ber-goa.example"), # IDNA 2008 for umlaut
26
+ ("example。com", "example.com"), # Normalize full-width punctuation
27
+ ],
28
+ )
29
+ def test_normalize_host_result_is_expected(host: str, expected: str) -> None:
30
+ """Assert we got expected results from the normalize_host function."""
31
+ result = normalize_host(host)
32
+ assert result == expected, host
@@ -0,0 +1,44 @@
1
+ """Tests for normalize_path function."""
2
+
3
+ import pytest
4
+
5
+ from url_normalize.url_normalize import normalize_path
6
+
7
+
8
+ @pytest.mark.parametrize(
9
+ ("path", "expected"),
10
+ [
11
+ ("..", "/"),
12
+ ("", "/"),
13
+ ("/../foo", "/foo"),
14
+ ("/..foo", "/..foo"),
15
+ ("/./../foo", "/foo"),
16
+ ("/./foo", "/foo"),
17
+ ("/./foo/.", "/foo/"),
18
+ ("/.foo", "/.foo"),
19
+ ("/", "/"),
20
+ ("/foo..", "/foo.."),
21
+ ("/foo.", "/foo."),
22
+ ("/FOO", "/FOO"),
23
+ ("/foo/../bar", "/bar"),
24
+ ("/foo/./bar", "/foo/bar"),
25
+ ("/foo//", "/foo/"),
26
+ ("/foo///bar//", "/foo/bar/"),
27
+ ("/foo/bar/..", "/foo/"),
28
+ ("/foo/bar/../..", "/"),
29
+ ("/foo/bar/../../../../baz", "/baz"),
30
+ ("/foo/bar/../../../baz", "/baz"),
31
+ ("/foo/bar/../../", "/"),
32
+ ("/foo/bar/../../baz", "/baz"),
33
+ ("/foo/bar/../", "/foo/"),
34
+ ("/foo/bar/../baz", "/foo/baz"),
35
+ ("/foo/bar/.", "/foo/bar/"),
36
+ ("/foo/bar/./", "/foo/bar/"),
37
+ # Issue #25: we should preserve ? in the path
38
+ ("/More+Tea+Vicar%3F/discussion", "/More+Tea+Vicar%3F/discussion"),
39
+ ],
40
+ )
41
+ def test_normalize_path_result_is_expected(path: str, expected: str) -> None:
42
+ """Assert we got expected results from the normalize_path function."""
43
+ result = normalize_path(path, "http")
44
+ assert result == expected, path
@@ -0,0 +1,26 @@
1
+ """Tests for normalize_port function."""
2
+
3
+ import pytest
4
+
5
+ from url_normalize.url_normalize import normalize_port
6
+
7
+
8
+ @pytest.mark.parametrize(
9
+ ("port", "expected"),
10
+ [
11
+ ("8080", "8080"), # Non-default port
12
+ ("", ""), # Empty port
13
+ ("80", ""), # Default HTTP port
14
+ ("string", "string"), # Non-numeric port (should pass through)
15
+ # Add more cases as needed, e.g., for HTTPS
16
+ pytest.param("443", "", id="https_default_port"),
17
+ ],
18
+ )
19
+ def test_normalize_port_result_is_expected(port: str, expected: str):
20
+ """Assert we got expected results from the normalize_port function."""
21
+ # Test with 'http' scheme for most cases
22
+ scheme = "https" if port == "443" else "http"
23
+
24
+ result = normalize_port(port, scheme)
25
+
26
+ assert result == expected
@@ -14,6 +14,8 @@ from url_normalize.url_normalize import normalize_query
14
14
  ("Ç=Ç", "%C3%87=%C3%87"),
15
15
  ("%C3%87=%C3%87", "%C3%87=%C3%87"),
16
16
  ("q=C%CC%A7", "q=%C3%87"),
17
+ ("q=%23test", "q=%23test"), # Preserve encoded # in value, #31
18
+ ("where=code%3D123", "where=code%3D123"), # Preserve encoded = in value, #25
17
19
  ],
18
20
  )
19
21
  def test_normalize_query_result_is_expected(query, expected):
@@ -0,0 +1,18 @@
1
+ """Tests for normalize_scheme function."""
2
+
3
+ import pytest
4
+
5
+ from url_normalize.url_normalize import normalize_scheme
6
+
7
+
8
+ @pytest.mark.parametrize(
9
+ ("scheme", "expected"),
10
+ [
11
+ ("http", "http"),
12
+ ("HTTP", "http"),
13
+ ],
14
+ )
15
+ def test_normalize_scheme_result_is_expected(scheme: str, expected: str) -> None:
16
+ """Assert we got expected results from the normalize_scheme function."""
17
+ result = normalize_scheme(scheme)
18
+ assert result == expected, scheme
@@ -0,0 +1,21 @@
1
+ """Tests for normalize_userinfo function."""
2
+
3
+ import pytest
4
+
5
+ from url_normalize.url_normalize import normalize_userinfo
6
+
7
+
8
+ @pytest.mark.parametrize(
9
+ ("userinfo", "expected"),
10
+ [
11
+ (":@", ""),
12
+ ("", ""),
13
+ ("@", ""),
14
+ ("user:password@", "user:password@"),
15
+ ("user@", "user@"),
16
+ ],
17
+ )
18
+ def test_normalize_userinfo_result_is_expected(userinfo: str, expected: str) -> None:
19
+ """Assert we got expected results from the normalize_userinfo function."""
20
+ result = normalize_userinfo(userinfo)
21
+ assert result == expected, userinfo