url-normalize 1.4.3__tar.gz → 2.0.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Potentially problematic release.
This version of url-normalize might be problematic. Click here for more details.
- url_normalize-2.0.0/PKG-INFO +105 -0
- url_normalize-2.0.0/README.md +79 -0
- url_normalize-2.0.0/pyproject.toml +78 -0
- url_normalize-2.0.0/setup.cfg +4 -0
- url_normalize-2.0.0/tests/test_deconstruct_url.py +35 -0
- url_normalize-2.0.0/tests/test_generic_url_cleanup.py +21 -0
- url_normalize-2.0.0/tests/test_normalize_fragment.py +19 -0
- url_normalize-2.0.0/tests/test_normalize_host.py +29 -0
- url_normalize-2.0.0/tests/test_normalize_path.py +40 -0
- url_normalize-2.0.0/tests/test_normalize_port.py +13 -0
- url_normalize-2.0.0/tests/test_normalize_query.py +22 -0
- url_normalize-2.0.0/tests/test_normalize_query_filters.py +98 -0
- url_normalize-2.0.0/tests/test_normalize_scheme.py +13 -0
- url_normalize-2.0.0/tests/test_normalize_userinfo.py +19 -0
- url_normalize-2.0.0/tests/test_provide_url_scheme.py +30 -0
- url_normalize-2.0.0/tests/test_reconstruct_url.py +41 -0
- url_normalize-2.0.0/tests/test_tools.py +12 -0
- url_normalize-2.0.0/tests/test_url_normalize.py +125 -0
- url_normalize-2.0.0/url_normalize/__init__.py +13 -0
- url_normalize-2.0.0/url_normalize/generic_url_cleanup.py +19 -0
- url_normalize-2.0.0/url_normalize/normalize_fragment.py +18 -0
- url_normalize-2.0.0/url_normalize/normalize_host.py +42 -0
- url_normalize-2.0.0/url_normalize/normalize_path.py +48 -0
- url_normalize-2.0.0/url_normalize/normalize_port.py +38 -0
- url_normalize-2.0.0/url_normalize/normalize_query.py +64 -0
- url_normalize-2.0.0/url_normalize/normalize_scheme.py +18 -0
- url_normalize-2.0.0/url_normalize/normalize_userinfo.py +18 -0
- url_normalize-2.0.0/url_normalize/param_allowlist.py +49 -0
- url_normalize-2.0.0/url_normalize/provide_url_scheme.py +26 -0
- url_normalize-2.0.0/url_normalize/tools.py +113 -0
- url_normalize-2.0.0/url_normalize/url_normalize.py +76 -0
- url_normalize-2.0.0/url_normalize.egg-info/PKG-INFO +105 -0
- url_normalize-2.0.0/url_normalize.egg-info/SOURCES.txt +35 -0
- url_normalize-2.0.0/url_normalize.egg-info/dependency_links.txt +1 -0
- url_normalize-2.0.0/url_normalize.egg-info/requires.txt +11 -0
- url_normalize-2.0.0/url_normalize.egg-info/top_level.txt +1 -0
- url-normalize-1.4.3/PKG-INFO +0 -79
- url-normalize-1.4.3/README.md +0 -54
- url-normalize-1.4.3/pyproject.toml +0 -34
- url-normalize-1.4.3/setup.py +0 -30
- url-normalize-1.4.3/url_normalize/__init__.py +0 -32
- url-normalize-1.4.3/url_normalize/tools.py +0 -100
- url-normalize-1.4.3/url_normalize/url_normalize.py +0 -244
- {url-normalize-1.4.3 → url_normalize-2.0.0}/LICENSE +0 -0
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: url-normalize
|
|
3
|
+
Version: 2.0.0
|
|
4
|
+
Summary: URL normalization for Python
|
|
5
|
+
Author-email: Nikolay Panov <github@npanov.com>
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/niksite/url-normalize
|
|
8
|
+
Project-URL: Repository, https://github.com/niksite/url-normalize
|
|
9
|
+
Project-URL: Issues, https://github.com/niksite/url-normalize/issues
|
|
10
|
+
Project-URL: Changelog, https://github.com/niksite/url-normalize/blob/master/CHANGELOG.md
|
|
11
|
+
Keywords: url,normalization,normalize
|
|
12
|
+
Requires-Python: >=3.8
|
|
13
|
+
Description-Content-Type: text/markdown
|
|
14
|
+
License-File: LICENSE
|
|
15
|
+
Requires-Dist: idna>=3.3
|
|
16
|
+
Provides-Extra: dev
|
|
17
|
+
Requires-Dist: mypy; extra == "dev"
|
|
18
|
+
Requires-Dist: pre-commit; extra == "dev"
|
|
19
|
+
Requires-Dist: pytest-cov; extra == "dev"
|
|
20
|
+
Requires-Dist: pytest-ruff; extra == "dev"
|
|
21
|
+
Requires-Dist: pytest-socket; extra == "dev"
|
|
22
|
+
Requires-Dist: pytest; extra == "dev"
|
|
23
|
+
Requires-Dist: ruff; extra == "dev"
|
|
24
|
+
Requires-Dist: tox; extra == "dev"
|
|
25
|
+
Dynamic: license-file
|
|
26
|
+
|
|
27
|
+
# url-normalize
|
|
28
|
+
|
|
29
|
+
URI Normalization function:
|
|
30
|
+
|
|
31
|
+
* Take care of IDN domains.
|
|
32
|
+
* Always provide the URI scheme in lowercase characters.
|
|
33
|
+
* Always provide the host, if any, in lowercase characters.
|
|
34
|
+
* Only perform percent-encoding where it is essential.
|
|
35
|
+
* Always use uppercase A-through-F characters when percent-encoding.
|
|
36
|
+
* Prevent dot-segments appearing in non-relative URI paths.
|
|
37
|
+
* For schemes that define a default authority, use an empty authority if the
|
|
38
|
+
default is desired.
|
|
39
|
+
* For schemes that define an empty path to be equivalent to a path of "/",
|
|
40
|
+
use "/".
|
|
41
|
+
* For schemes that define a port, use an empty port if the default is desired
|
|
42
|
+
* All portions of the URI must be utf-8 encoded NFC from Unicode strings
|
|
43
|
+
|
|
44
|
+
Inspired by Sam Ruby's [urlnorm.py](<http://intertwingly.net/blog/2004/08/04/Urlnorm>)
|
|
45
|
+
|
|
46
|
+
## Features
|
|
47
|
+
|
|
48
|
+
* IDN (Internationalized Domain Name) support
|
|
49
|
+
* Configurable default scheme (https by default)
|
|
50
|
+
* Query parameter filtering with allowlists
|
|
51
|
+
* Support for various URL formats including:
|
|
52
|
+
* Empty string URLs
|
|
53
|
+
* Double slash URLs (//domain.tld)
|
|
54
|
+
* Shebang (#!) URLs
|
|
55
|
+
* Cross-version Python compatibility (3.8+)
|
|
56
|
+
* 100% test coverage
|
|
57
|
+
* Modern type hints and string handling
|
|
58
|
+
|
|
59
|
+
## Installation
|
|
60
|
+
|
|
61
|
+
```sh
|
|
62
|
+
pip install url-normalize
|
|
63
|
+
```
|
|
64
|
+
|
|
65
|
+
## Usage
|
|
66
|
+
|
|
67
|
+
Basic usage:
|
|
68
|
+
|
|
69
|
+
```python
|
|
70
|
+
from url_normalize import url_normalize
|
|
71
|
+
|
|
72
|
+
# Basic normalization (uses https by default)
|
|
73
|
+
print(url_normalize("www.foo.com:80/foo"))
|
|
74
|
+
# Output: https://www.foo.com/foo
|
|
75
|
+
|
|
76
|
+
# With custom default scheme
|
|
77
|
+
print(url_normalize("www.foo.com/foo", default_scheme="http"))
|
|
78
|
+
# Output: http://www.foo.com/foo
|
|
79
|
+
|
|
80
|
+
# With query parameter filtering enabled
|
|
81
|
+
print(url_normalize("www.google.com/search?q=test&utm_source=test", filter_params=True))
|
|
82
|
+
# Output: https://www.google.com/search?q=test
|
|
83
|
+
|
|
84
|
+
# With custom parameter allowlist
|
|
85
|
+
print(url_normalize(
|
|
86
|
+
"example.com?page=1&id=123&ref=test",
|
|
87
|
+
filter_params=True,
|
|
88
|
+
param_allowlist={"example.com": ["page", "id"]}
|
|
89
|
+
))
|
|
90
|
+
# Output: https://example.com?page=1&id=123
|
|
91
|
+
print(url_normalize(
|
|
92
|
+
"example.com?page=1&id=123&ref=test",
|
|
93
|
+
filter_params=True,
|
|
94
|
+
param_allowlist=["page", "id"]
|
|
95
|
+
))
|
|
96
|
+
# Output: https://example.com?page=1&id=123
|
|
97
|
+
```
|
|
98
|
+
|
|
99
|
+
## Documentation
|
|
100
|
+
|
|
101
|
+
For a complete history of changes, see [CHANGELOG.md](CHANGELOG.md).
|
|
102
|
+
|
|
103
|
+
## License
|
|
104
|
+
|
|
105
|
+
MIT License
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
# url-normalize
|
|
2
|
+
|
|
3
|
+
URI Normalization function:
|
|
4
|
+
|
|
5
|
+
* Take care of IDN domains.
|
|
6
|
+
* Always provide the URI scheme in lowercase characters.
|
|
7
|
+
* Always provide the host, if any, in lowercase characters.
|
|
8
|
+
* Only perform percent-encoding where it is essential.
|
|
9
|
+
* Always use uppercase A-through-F characters when percent-encoding.
|
|
10
|
+
* Prevent dot-segments appearing in non-relative URI paths.
|
|
11
|
+
* For schemes that define a default authority, use an empty authority if the
|
|
12
|
+
default is desired.
|
|
13
|
+
* For schemes that define an empty path to be equivalent to a path of "/",
|
|
14
|
+
use "/".
|
|
15
|
+
* For schemes that define a port, use an empty port if the default is desired
|
|
16
|
+
* All portions of the URI must be utf-8 encoded NFC from Unicode strings
|
|
17
|
+
|
|
18
|
+
Inspired by Sam Ruby's [urlnorm.py](<http://intertwingly.net/blog/2004/08/04/Urlnorm>)
|
|
19
|
+
|
|
20
|
+
## Features
|
|
21
|
+
|
|
22
|
+
* IDN (Internationalized Domain Name) support
|
|
23
|
+
* Configurable default scheme (https by default)
|
|
24
|
+
* Query parameter filtering with allowlists
|
|
25
|
+
* Support for various URL formats including:
|
|
26
|
+
* Empty string URLs
|
|
27
|
+
* Double slash URLs (//domain.tld)
|
|
28
|
+
* Shebang (#!) URLs
|
|
29
|
+
* Cross-version Python compatibility (3.8+)
|
|
30
|
+
* 100% test coverage
|
|
31
|
+
* Modern type hints and string handling
|
|
32
|
+
|
|
33
|
+
## Installation
|
|
34
|
+
|
|
35
|
+
```sh
|
|
36
|
+
pip install url-normalize
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
## Usage
|
|
40
|
+
|
|
41
|
+
Basic usage:
|
|
42
|
+
|
|
43
|
+
```python
|
|
44
|
+
from url_normalize import url_normalize
|
|
45
|
+
|
|
46
|
+
# Basic normalization (uses https by default)
|
|
47
|
+
print(url_normalize("www.foo.com:80/foo"))
|
|
48
|
+
# Output: https://www.foo.com/foo
|
|
49
|
+
|
|
50
|
+
# With custom default scheme
|
|
51
|
+
print(url_normalize("www.foo.com/foo", default_scheme="http"))
|
|
52
|
+
# Output: http://www.foo.com/foo
|
|
53
|
+
|
|
54
|
+
# With query parameter filtering enabled
|
|
55
|
+
print(url_normalize("www.google.com/search?q=test&utm_source=test", filter_params=True))
|
|
56
|
+
# Output: https://www.google.com/search?q=test
|
|
57
|
+
|
|
58
|
+
# With custom parameter allowlist
|
|
59
|
+
print(url_normalize(
|
|
60
|
+
"example.com?page=1&id=123&ref=test",
|
|
61
|
+
filter_params=True,
|
|
62
|
+
param_allowlist={"example.com": ["page", "id"]}
|
|
63
|
+
))
|
|
64
|
+
# Output: https://example.com?page=1&id=123
|
|
65
|
+
print(url_normalize(
|
|
66
|
+
"example.com?page=1&id=123&ref=test",
|
|
67
|
+
filter_params=True,
|
|
68
|
+
param_allowlist=["page", "id"]
|
|
69
|
+
))
|
|
70
|
+
# Output: https://example.com?page=1&id=123
|
|
71
|
+
```
|
|
72
|
+
|
|
73
|
+
## Documentation
|
|
74
|
+
|
|
75
|
+
For a complete history of changes, see [CHANGELOG.md](CHANGELOG.md).
|
|
76
|
+
|
|
77
|
+
## License
|
|
78
|
+
|
|
79
|
+
MIT License
|
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "url-normalize"
|
|
3
|
+
version = "2.0.0"
|
|
4
|
+
description = "URL normalization for Python"
|
|
5
|
+
authors = [{ name = "Nikolay Panov", email = "github@npanov.com" }]
|
|
6
|
+
license = "MIT"
|
|
7
|
+
readme = "README.md"
|
|
8
|
+
requires-python = ">=3.8"
|
|
9
|
+
keywords = ["url", "normalization", "normalize"]
|
|
10
|
+
dependencies = ["idna>=3.3"]
|
|
11
|
+
|
|
12
|
+
[project.urls]
|
|
13
|
+
Homepage = "https://github.com/niksite/url-normalize"
|
|
14
|
+
Repository = "https://github.com/niksite/url-normalize"
|
|
15
|
+
Issues = "https://github.com/niksite/url-normalize/issues"
|
|
16
|
+
Changelog = "https://github.com/niksite/url-normalize/blob/master/CHANGELOG.md"
|
|
17
|
+
|
|
18
|
+
[project.optional-dependencies]
|
|
19
|
+
dev = [
|
|
20
|
+
"mypy",
|
|
21
|
+
"pre-commit",
|
|
22
|
+
"pytest-cov",
|
|
23
|
+
"pytest-ruff",
|
|
24
|
+
"pytest-socket",
|
|
25
|
+
"pytest",
|
|
26
|
+
"ruff",
|
|
27
|
+
"tox",
|
|
28
|
+
]
|
|
29
|
+
|
|
30
|
+
[tool.ruff]
|
|
31
|
+
target-version = "py38"
|
|
32
|
+
line-length = 88
|
|
33
|
+
unsafe-fixes = true
|
|
34
|
+
|
|
35
|
+
[tool.ruff.lint]
|
|
36
|
+
select = ["ALL"]
|
|
37
|
+
extend-select = [
|
|
38
|
+
"D400", # First line should end with a period
|
|
39
|
+
"D401", # First line should be in imperative mood
|
|
40
|
+
"D413", # Missing blank line after the last section of a multiline docstring
|
|
41
|
+
]
|
|
42
|
+
fixable = ["ALL"]
|
|
43
|
+
ignore = [
|
|
44
|
+
"COM812", # missing-trailing-comma
|
|
45
|
+
"D203", # One blank line before class - we prefer D211 instead
|
|
46
|
+
"D213", # multi-line-summary-second-line - we prefer D212 instead
|
|
47
|
+
]
|
|
48
|
+
|
|
49
|
+
[tool.ruff.lint.pydocstyle]
|
|
50
|
+
convention = "google"
|
|
51
|
+
|
|
52
|
+
[tool.ruff.lint.per-file-ignores]
|
|
53
|
+
"tests/**" = ["INP001", "ANN001", "ANN201", "S101", "CPY001"]
|
|
54
|
+
|
|
55
|
+
[tool.ruff.format]
|
|
56
|
+
quote-style = "double"
|
|
57
|
+
indent-style = "space"
|
|
58
|
+
|
|
59
|
+
[tool.mypy]
|
|
60
|
+
ignore_missing_imports = true
|
|
61
|
+
exclude = ["tests"]
|
|
62
|
+
python_version = "3.8"
|
|
63
|
+
show_error_codes = true
|
|
64
|
+
|
|
65
|
+
[build-system]
|
|
66
|
+
requires = ["setuptools>=42", "wheel"]
|
|
67
|
+
build-backend = "setuptools.build_meta"
|
|
68
|
+
|
|
69
|
+
[tool.pytest.ini_options]
|
|
70
|
+
addopts = [
|
|
71
|
+
"--cov-fail-under=100",
|
|
72
|
+
"--cov-report=term-missing:skip-covered",
|
|
73
|
+
"--cov=url_normalize",
|
|
74
|
+
"--disable-socket",
|
|
75
|
+
"--ruff",
|
|
76
|
+
"-v",
|
|
77
|
+
]
|
|
78
|
+
python_files = ["tests.py", "test_*.py", "*_tests.py"]
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
"""Deconstruct url tests."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Final
|
|
6
|
+
|
|
7
|
+
from url_normalize.tools import URL, deconstruct_url
|
|
8
|
+
|
|
9
|
+
EXPECTED_DATA: Final[dict[str, URL]] = {
|
|
10
|
+
"http://site.com": URL(
|
|
11
|
+
fragment="",
|
|
12
|
+
host="site.com",
|
|
13
|
+
path="",
|
|
14
|
+
port="",
|
|
15
|
+
query="",
|
|
16
|
+
scheme="http",
|
|
17
|
+
userinfo="",
|
|
18
|
+
),
|
|
19
|
+
"http://user@www.example.com:8080/path/index.html?param=val#fragment": URL(
|
|
20
|
+
fragment="fragment",
|
|
21
|
+
host="www.example.com",
|
|
22
|
+
path="/path/index.html",
|
|
23
|
+
port="8080",
|
|
24
|
+
query="param=val",
|
|
25
|
+
scheme="http",
|
|
26
|
+
userinfo="user@",
|
|
27
|
+
),
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def test_deconstruct_url_result_is_expected() -> None:
|
|
32
|
+
"""Assert we got expected results from the deconstruct_url function."""
|
|
33
|
+
for url, expected in EXPECTED_DATA.items():
|
|
34
|
+
result = deconstruct_url(url)
|
|
35
|
+
assert result == expected, url
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
"""Tests for generic_url_cleanup function."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import pytest
|
|
6
|
+
|
|
7
|
+
from url_normalize.url_normalize import generic_url_cleanup
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
@pytest.mark.parametrize(
|
|
11
|
+
("url", "expected"),
|
|
12
|
+
[
|
|
13
|
+
("//site/#!fragment", "//site/?_escaped_fragment_=fragment"),
|
|
14
|
+
("//site/page", "//site/page"),
|
|
15
|
+
("//site/?& ", "//site/"),
|
|
16
|
+
],
|
|
17
|
+
)
|
|
18
|
+
def test_generic_url_cleanup_result_is_expected(url: str, expected: str) -> None:
|
|
19
|
+
"""Assert we got expected results from the generic_url_cleanup function."""
|
|
20
|
+
result = generic_url_cleanup(url)
|
|
21
|
+
assert result == expected
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
"""Tests for normalize_fragment function."""
|
|
2
|
+
|
|
3
|
+
from url_normalize.url_normalize import normalize_fragment
|
|
4
|
+
|
|
5
|
+
EXPECTED_DATA = {
|
|
6
|
+
"": "",
|
|
7
|
+
"fragment": "fragment",
|
|
8
|
+
"пример": "%D0%BF%D1%80%D0%B8%D0%BC%D0%B5%D1%80",
|
|
9
|
+
"!fragment": "%21fragment",
|
|
10
|
+
"~fragment": "~fragment",
|
|
11
|
+
}
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def test_normalize_fragment_result_is_expected():
|
|
15
|
+
"""Assert we got expected results from the normalize_fragment function."""
|
|
16
|
+
for url, expected in EXPECTED_DATA.items():
|
|
17
|
+
result = normalize_fragment(url)
|
|
18
|
+
|
|
19
|
+
assert result == expected, url
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
"""Tests for normalize_host function."""
|
|
2
|
+
|
|
3
|
+
from url_normalize.url_normalize import normalize_host
|
|
4
|
+
|
|
5
|
+
EXPECTED_DATA = {
|
|
6
|
+
# Basic cases
|
|
7
|
+
"site.com": "site.com",
|
|
8
|
+
"SITE.COM": "site.com",
|
|
9
|
+
"site.com.": "site.com",
|
|
10
|
+
# Cyrillic domains
|
|
11
|
+
"пример.испытание": "xn--e1afmkfd.xn--80akhbyknj4f",
|
|
12
|
+
# Mixed case with Cyrillic
|
|
13
|
+
"ExAmPle.РФ": "example.xn--p1ai",
|
|
14
|
+
# IDNA2008 with UTS46
|
|
15
|
+
"faß.de": "fass.de", # Normalize using transitional rules
|
|
16
|
+
# Edge cases
|
|
17
|
+
"ドメイン.テスト": "xn--eckwd4c7c.xn--zckzah", # Japanese
|
|
18
|
+
"domain.café": "domain.xn--caf-dma", # Latin with diacritic
|
|
19
|
+
# Normalization tests
|
|
20
|
+
"über.example": "xn--ber-goa.example", # IDNA 2008 for umlaut
|
|
21
|
+
"example。com": "example.com", # Normalize full-width punctuation
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def test_normalize_host_result_is_expected() -> None:
|
|
26
|
+
"""Assert we got expected results from the normalize_host function."""
|
|
27
|
+
for url, expected in EXPECTED_DATA.items():
|
|
28
|
+
result = normalize_host(url)
|
|
29
|
+
assert result == expected, url
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
"""Tests for normalize_path function."""
|
|
2
|
+
|
|
3
|
+
from url_normalize.url_normalize import normalize_path
|
|
4
|
+
|
|
5
|
+
EXPECTED_DATA = {
|
|
6
|
+
"..": "/",
|
|
7
|
+
"": "/",
|
|
8
|
+
"/../foo": "/foo",
|
|
9
|
+
"/..foo": "/..foo",
|
|
10
|
+
"/./../foo": "/foo",
|
|
11
|
+
"/./foo": "/foo",
|
|
12
|
+
"/./foo/.": "/foo/",
|
|
13
|
+
"/.foo": "/.foo",
|
|
14
|
+
"/": "/",
|
|
15
|
+
"/foo..": "/foo..",
|
|
16
|
+
"/foo.": "/foo.",
|
|
17
|
+
"/FOO": "/FOO",
|
|
18
|
+
"/foo/../bar": "/bar",
|
|
19
|
+
"/foo/./bar": "/foo/bar",
|
|
20
|
+
"/foo//": "/foo/",
|
|
21
|
+
"/foo///bar//": "/foo/bar/",
|
|
22
|
+
"/foo/bar/..": "/foo/",
|
|
23
|
+
"/foo/bar/../..": "/",
|
|
24
|
+
"/foo/bar/../../../../baz": "/baz",
|
|
25
|
+
"/foo/bar/../../../baz": "/baz",
|
|
26
|
+
"/foo/bar/../../": "/",
|
|
27
|
+
"/foo/bar/../../baz": "/baz",
|
|
28
|
+
"/foo/bar/../": "/foo/",
|
|
29
|
+
"/foo/bar/../baz": "/foo/baz",
|
|
30
|
+
"/foo/bar/.": "/foo/bar/",
|
|
31
|
+
"/foo/bar/./": "/foo/bar/",
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def test_normalize_host_result_is_expected():
|
|
36
|
+
"""Assert we got expected results from the normalize_path function."""
|
|
37
|
+
for url, expected in EXPECTED_DATA.items():
|
|
38
|
+
result = normalize_path(url, "http")
|
|
39
|
+
|
|
40
|
+
assert result == expected, url
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
"""Tests for normalize_port function."""
|
|
2
|
+
|
|
3
|
+
from url_normalize.url_normalize import normalize_port
|
|
4
|
+
|
|
5
|
+
EXPECTED_DATA = {"8080": "8080", "": "", "80": "", "string": "string"}
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def test_normalize_port_result_is_expected():
|
|
9
|
+
"""Assert we got expected results from the normalize_port function."""
|
|
10
|
+
for url, expected in EXPECTED_DATA.items():
|
|
11
|
+
result = normalize_port(url, "http")
|
|
12
|
+
|
|
13
|
+
assert result == expected, url
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
"""Tests for normalize_query function."""
|
|
2
|
+
|
|
3
|
+
import pytest
|
|
4
|
+
|
|
5
|
+
from url_normalize.url_normalize import normalize_query
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
@pytest.mark.parametrize(
|
|
9
|
+
("query", "expected"),
|
|
10
|
+
[
|
|
11
|
+
("", ""),
|
|
12
|
+
("&&&", ""),
|
|
13
|
+
("param1=val1¶m2=val2", "param1=val1¶m2=val2"),
|
|
14
|
+
("Ç=Ç", "%C3%87=%C3%87"),
|
|
15
|
+
("%C3%87=%C3%87", "%C3%87=%C3%87"),
|
|
16
|
+
("q=C%CC%A7", "q=%C3%87"),
|
|
17
|
+
],
|
|
18
|
+
)
|
|
19
|
+
def test_normalize_query_result_is_expected(query, expected):
|
|
20
|
+
"""Assert we got expected results from the normalize_query function."""
|
|
21
|
+
result = normalize_query(query)
|
|
22
|
+
assert result == expected, query
|
|
@@ -0,0 +1,98 @@
|
|
|
1
|
+
"""URL parameter filtering test module."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import pytest
|
|
6
|
+
|
|
7
|
+
from url_normalize import url_normalize
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def test_param_filtering_disabled_by_default():
|
|
11
|
+
"""Test that parameter filtering is disabled by default."""
|
|
12
|
+
url = "https://www.google.com/search?q=test&utm_source=test"
|
|
13
|
+
assert url_normalize(url) == url
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def test_empty_query():
|
|
17
|
+
"""Test handling empty query strings."""
|
|
18
|
+
assert url_normalize("https://example.com/page?") == "https://example.com/page"
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def test_custom_allowlist():
|
|
22
|
+
"""Test custom allowlist functionality with preserved order."""
|
|
23
|
+
custom_allowlist = {"example.com": ["page", "id"], "google.com": ["q", "lang"]}
|
|
24
|
+
|
|
25
|
+
# Order should match input query string order
|
|
26
|
+
assert (
|
|
27
|
+
url_normalize(
|
|
28
|
+
"https://example.com/search?page=1&id=123&utm_source=test",
|
|
29
|
+
filter_params=True,
|
|
30
|
+
param_allowlist=custom_allowlist,
|
|
31
|
+
)
|
|
32
|
+
== "https://example.com/search?page=1&id=123"
|
|
33
|
+
)
|
|
34
|
+
|
|
35
|
+
assert (
|
|
36
|
+
url_normalize(
|
|
37
|
+
"https://google.com/search?q=test&ie=utf8&lang=en",
|
|
38
|
+
filter_params=True,
|
|
39
|
+
param_allowlist=custom_allowlist,
|
|
40
|
+
)
|
|
41
|
+
== "https://google.com/search?q=test&lang=en"
|
|
42
|
+
)
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def test_custom_list_allowlist():
|
|
46
|
+
"""Test custom list allowlist functionality."""
|
|
47
|
+
assert (
|
|
48
|
+
url_normalize(
|
|
49
|
+
"https://google.com/search?qq=test&ie=utf8&utm_source=test",
|
|
50
|
+
filter_params=True,
|
|
51
|
+
param_allowlist=["ie", "qq"],
|
|
52
|
+
)
|
|
53
|
+
== "https://google.com/search?qq=test&ie=utf8"
|
|
54
|
+
)
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
@pytest.mark.parametrize(
|
|
58
|
+
("url", "expected"),
|
|
59
|
+
[
|
|
60
|
+
# Basic parameter filtering
|
|
61
|
+
(
|
|
62
|
+
"https://www.google.com/search?q=test&utm_source=test",
|
|
63
|
+
"https://www.google.com/search?q=test",
|
|
64
|
+
),
|
|
65
|
+
(
|
|
66
|
+
"https://www.youtube.com/watch?v=12345&utm_source=share",
|
|
67
|
+
"https://www.youtube.com/watch?v=12345",
|
|
68
|
+
),
|
|
69
|
+
# With www subdomain
|
|
70
|
+
(
|
|
71
|
+
"https://www.google.com/search?q=test&ref=test",
|
|
72
|
+
"https://www.google.com/search?q=test",
|
|
73
|
+
),
|
|
74
|
+
# With port number
|
|
75
|
+
(
|
|
76
|
+
"https://google.com:8080/search?q=test&ref=test",
|
|
77
|
+
"https://google.com:8080/search?q=test",
|
|
78
|
+
),
|
|
79
|
+
# Default allowlist cases
|
|
80
|
+
(
|
|
81
|
+
"https://www.google.com/search?q=test&utm_source=test&ie=utf8",
|
|
82
|
+
"https://www.google.com/search?q=test&ie=utf8",
|
|
83
|
+
),
|
|
84
|
+
(
|
|
85
|
+
"https://www.baidu.com/s?wd=test&utm_source=test&ie=utf8",
|
|
86
|
+
"https://www.baidu.com/s?wd=test&ie=utf8",
|
|
87
|
+
),
|
|
88
|
+
(
|
|
89
|
+
"https://youtube.com/watch?v=12345&utm_source=test&search_query=test",
|
|
90
|
+
"https://youtube.com/watch?v=12345&search_query=test",
|
|
91
|
+
),
|
|
92
|
+
# Non-allowlisted domain
|
|
93
|
+
("https://example.org/page?a=1&b=2", "https://example.org/page"),
|
|
94
|
+
],
|
|
95
|
+
)
|
|
96
|
+
def test_parameter_filtering(url: str, expected: str):
|
|
97
|
+
"""Test URL parameter filtering functionality with various scenarios."""
|
|
98
|
+
assert url_normalize(url, filter_params=True) == expected
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
"""Tests for normalize_scheme function."""
|
|
2
|
+
|
|
3
|
+
from url_normalize.url_normalize import normalize_scheme
|
|
4
|
+
|
|
5
|
+
EXPECTED_DATA = {"http": "http", "HTTP": "http"}
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def test_normalize_scheme_result_is_expected():
|
|
9
|
+
"""Assert we got expected results from the normalize_scheme function."""
|
|
10
|
+
for url, expected in EXPECTED_DATA.items():
|
|
11
|
+
result = normalize_scheme(url)
|
|
12
|
+
|
|
13
|
+
assert result == expected, url
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
"""Tests for normalize_userinfo function."""
|
|
2
|
+
|
|
3
|
+
from url_normalize.url_normalize import normalize_userinfo
|
|
4
|
+
|
|
5
|
+
EXPECTED_DATA = {
|
|
6
|
+
":@": "",
|
|
7
|
+
"": "",
|
|
8
|
+
"@": "",
|
|
9
|
+
"user:password@": "user:password@",
|
|
10
|
+
"user@": "user@",
|
|
11
|
+
}
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def test_normalize_userinfo_result_is_expected():
|
|
15
|
+
"""Assert we got expected results from the normalize_userinfo function."""
|
|
16
|
+
for url, expected in EXPECTED_DATA.items():
|
|
17
|
+
result = normalize_userinfo(url)
|
|
18
|
+
|
|
19
|
+
assert result == expected, url
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
"""Tests for provide_url_scheme function."""
|
|
2
|
+
|
|
3
|
+
from url_normalize.url_normalize import provide_url_scheme
|
|
4
|
+
|
|
5
|
+
EXPECTED_DATA = {
|
|
6
|
+
"": "",
|
|
7
|
+
"-": "-",
|
|
8
|
+
"/file/path": "/file/path",
|
|
9
|
+
"//site/path": "https://site/path",
|
|
10
|
+
"ftp://site/": "ftp://site/",
|
|
11
|
+
"site/page": "https://site/page",
|
|
12
|
+
}
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def test_provide_url_scheme_result_is_expected():
|
|
16
|
+
"""Assert we got expected results from the provide_url_scheme function."""
|
|
17
|
+
for url, expected in EXPECTED_DATA.items():
|
|
18
|
+
result = provide_url_scheme(url)
|
|
19
|
+
|
|
20
|
+
assert result == expected, url
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def test_provide_url_scheme_accept_default_scheme_param():
|
|
24
|
+
"""Assert we could provide default_scheme param other than https."""
|
|
25
|
+
url = "//site/path"
|
|
26
|
+
expected = "http://site/path"
|
|
27
|
+
|
|
28
|
+
actual = provide_url_scheme(url, default_scheme="http")
|
|
29
|
+
|
|
30
|
+
assert actual == expected
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
"""Reconstruct url tests."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Final
|
|
6
|
+
|
|
7
|
+
from url_normalize.tools import URL, reconstruct_url
|
|
8
|
+
|
|
9
|
+
EXPECTED_DATA: Final[tuple[tuple[URL, str], ...]] = (
|
|
10
|
+
(
|
|
11
|
+
URL(
|
|
12
|
+
fragment="",
|
|
13
|
+
host="site.com",
|
|
14
|
+
path="",
|
|
15
|
+
port="",
|
|
16
|
+
query="",
|
|
17
|
+
scheme="http",
|
|
18
|
+
userinfo="",
|
|
19
|
+
),
|
|
20
|
+
"http://site.com",
|
|
21
|
+
),
|
|
22
|
+
(
|
|
23
|
+
URL(
|
|
24
|
+
fragment="fragment",
|
|
25
|
+
host="www.example.com",
|
|
26
|
+
path="/path/index.html",
|
|
27
|
+
port="8080",
|
|
28
|
+
query="param=val",
|
|
29
|
+
scheme="http",
|
|
30
|
+
userinfo="user@",
|
|
31
|
+
),
|
|
32
|
+
"http://user@www.example.com:8080/path/index.html?param=val#fragment",
|
|
33
|
+
),
|
|
34
|
+
)
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def test_deconstruct_url_result_is_expected() -> None:
|
|
38
|
+
"""Assert we got expected results from the deconstruct_url function."""
|
|
39
|
+
for url, expected in EXPECTED_DATA:
|
|
40
|
+
result = reconstruct_url(url)
|
|
41
|
+
assert result == expected, url
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
"""Tools module tests."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from url_normalize.tools import force_unicode
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def test_force_unicode_with_bytes() -> None:
|
|
9
|
+
"""Test force_unicode handles bytes input correctly."""
|
|
10
|
+
test_bytes = b"hello world"
|
|
11
|
+
result = force_unicode(test_bytes)
|
|
12
|
+
assert result == "hello world"
|