url-normalize 2.0.0__tar.gz → 2.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {url_normalize-2.0.0/url_normalize.egg-info → url_normalize-2.1.0}/PKG-INFO +33 -5
- {url_normalize-2.0.0 → url_normalize-2.1.0}/README.md +31 -3
- {url_normalize-2.0.0 → url_normalize-2.1.0}/pyproject.toml +5 -2
- url_normalize-2.1.0/tests/test_cli.py +233 -0
- {url_normalize-2.0.0 → url_normalize-2.1.0}/tests/test_normalize_fragment.py +2 -0
- {url_normalize-2.0.0 → url_normalize-2.1.0}/tests/test_url_normalize.py +2 -0
- {url_normalize-2.0.0 → url_normalize-2.1.0}/url_normalize/__init__.py +1 -1
- url_normalize-2.1.0/url_normalize/cli.py +66 -0
- url_normalize-2.1.0/url_normalize/normalize_fragment.py +27 -0
- {url_normalize-2.0.0 → url_normalize-2.1.0/url_normalize.egg-info}/PKG-INFO +33 -5
- {url_normalize-2.0.0 → url_normalize-2.1.0}/url_normalize.egg-info/SOURCES.txt +3 -0
- url_normalize-2.1.0/url_normalize.egg-info/entry_points.txt +2 -0
- url_normalize-2.0.0/url_normalize/normalize_fragment.py +0 -18
- {url_normalize-2.0.0 → url_normalize-2.1.0}/LICENSE +0 -0
- {url_normalize-2.0.0 → url_normalize-2.1.0}/setup.cfg +0 -0
- {url_normalize-2.0.0 → url_normalize-2.1.0}/tests/test_deconstruct_url.py +0 -0
- {url_normalize-2.0.0 → url_normalize-2.1.0}/tests/test_generic_url_cleanup.py +0 -0
- {url_normalize-2.0.0 → url_normalize-2.1.0}/tests/test_normalize_host.py +0 -0
- {url_normalize-2.0.0 → url_normalize-2.1.0}/tests/test_normalize_path.py +0 -0
- {url_normalize-2.0.0 → url_normalize-2.1.0}/tests/test_normalize_port.py +0 -0
- {url_normalize-2.0.0 → url_normalize-2.1.0}/tests/test_normalize_query.py +0 -0
- {url_normalize-2.0.0 → url_normalize-2.1.0}/tests/test_normalize_query_filters.py +0 -0
- {url_normalize-2.0.0 → url_normalize-2.1.0}/tests/test_normalize_scheme.py +0 -0
- {url_normalize-2.0.0 → url_normalize-2.1.0}/tests/test_normalize_userinfo.py +0 -0
- {url_normalize-2.0.0 → url_normalize-2.1.0}/tests/test_provide_url_scheme.py +0 -0
- {url_normalize-2.0.0 → url_normalize-2.1.0}/tests/test_reconstruct_url.py +0 -0
- {url_normalize-2.0.0 → url_normalize-2.1.0}/tests/test_tools.py +0 -0
- {url_normalize-2.0.0 → url_normalize-2.1.0}/url_normalize/generic_url_cleanup.py +0 -0
- {url_normalize-2.0.0 → url_normalize-2.1.0}/url_normalize/normalize_host.py +0 -0
- {url_normalize-2.0.0 → url_normalize-2.1.0}/url_normalize/normalize_path.py +0 -0
- {url_normalize-2.0.0 → url_normalize-2.1.0}/url_normalize/normalize_port.py +0 -0
- {url_normalize-2.0.0 → url_normalize-2.1.0}/url_normalize/normalize_query.py +0 -0
- {url_normalize-2.0.0 → url_normalize-2.1.0}/url_normalize/normalize_scheme.py +0 -0
- {url_normalize-2.0.0 → url_normalize-2.1.0}/url_normalize/normalize_userinfo.py +0 -0
- {url_normalize-2.0.0 → url_normalize-2.1.0}/url_normalize/param_allowlist.py +0 -0
- {url_normalize-2.0.0 → url_normalize-2.1.0}/url_normalize/provide_url_scheme.py +0 -0
- {url_normalize-2.0.0 → url_normalize-2.1.0}/url_normalize/tools.py +0 -0
- {url_normalize-2.0.0 → url_normalize-2.1.0}/url_normalize/url_normalize.py +0 -0
- {url_normalize-2.0.0 → url_normalize-2.1.0}/url_normalize.egg-info/dependency_links.txt +0 -0
- {url_normalize-2.0.0 → url_normalize-2.1.0}/url_normalize.egg-info/requires.txt +0 -0
- {url_normalize-2.0.0 → url_normalize-2.1.0}/url_normalize.egg-info/top_level.txt +0 -0
|
@@ -1,9 +1,9 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: url-normalize
|
|
3
|
-
Version: 2.
|
|
3
|
+
Version: 2.1.0
|
|
4
4
|
Summary: URL normalization for Python
|
|
5
5
|
Author-email: Nikolay Panov <github@npanov.com>
|
|
6
|
-
License
|
|
6
|
+
License: MIT
|
|
7
7
|
Project-URL: Homepage, https://github.com/niksite/url-normalize
|
|
8
8
|
Project-URL: Repository, https://github.com/niksite/url-normalize
|
|
9
9
|
Project-URL: Issues, https://github.com/niksite/url-normalize/issues
|
|
@@ -26,6 +26,9 @@ Dynamic: license-file
|
|
|
26
26
|
|
|
27
27
|
# url-normalize
|
|
28
28
|
|
|
29
|
+
[](https://github.com/niksite/url-normalize/actions/workflows/ci.yml)
|
|
30
|
+
[](https://github.com/niksite/url-normalize/actions/workflows/publish.yml)
|
|
31
|
+
|
|
29
32
|
URI Normalization function:
|
|
30
33
|
|
|
31
34
|
* Take care of IDN domains.
|
|
@@ -64,8 +67,6 @@ pip install url-normalize
|
|
|
64
67
|
|
|
65
68
|
## Usage
|
|
66
69
|
|
|
67
|
-
Basic usage:
|
|
68
|
-
|
|
69
70
|
```python
|
|
70
71
|
from url_normalize import url_normalize
|
|
71
72
|
|
|
@@ -81,13 +82,15 @@ print(url_normalize("www.foo.com/foo", default_scheme="http"))
|
|
|
81
82
|
print(url_normalize("www.google.com/search?q=test&utm_source=test", filter_params=True))
|
|
82
83
|
# Output: https://www.google.com/search?q=test
|
|
83
84
|
|
|
84
|
-
# With custom parameter allowlist
|
|
85
|
+
# With custom parameter allowlist as a dict
|
|
85
86
|
print(url_normalize(
|
|
86
87
|
"example.com?page=1&id=123&ref=test",
|
|
87
88
|
filter_params=True,
|
|
88
89
|
param_allowlist={"example.com": ["page", "id"]}
|
|
89
90
|
))
|
|
90
91
|
# Output: https://example.com?page=1&id=123
|
|
92
|
+
|
|
93
|
+
# With custom parameter allowlist as a list
|
|
91
94
|
print(url_normalize(
|
|
92
95
|
"example.com?page=1&id=123&ref=test",
|
|
93
96
|
filter_params=True,
|
|
@@ -96,6 +99,31 @@ print(url_normalize(
|
|
|
96
99
|
# Output: https://example.com?page=1&id=123
|
|
97
100
|
```
|
|
98
101
|
|
|
102
|
+
### Command-line usage
|
|
103
|
+
|
|
104
|
+
You can also use `url-normalize` from the command line:
|
|
105
|
+
|
|
106
|
+
```bash
|
|
107
|
+
$ url-normalize "www.foo.com:80/foo"
|
|
108
|
+
# Output: https://www.foo.com/foo
|
|
109
|
+
|
|
110
|
+
# With custom default scheme
|
|
111
|
+
$ url-normalize -s http "www.foo.com/foo"
|
|
112
|
+
# Output: http://www.foo.com/foo
|
|
113
|
+
|
|
114
|
+
# With query parameter filtering
|
|
115
|
+
$ url-normalize -f "www.google.com/search?q=test&utm_source=test"
|
|
116
|
+
# Output: https://www.google.com/search?q=test
|
|
117
|
+
|
|
118
|
+
# With custom allowlist
|
|
119
|
+
$ url-normalize -f -p page,id "example.com?page=1&id=123&ref=test"
|
|
120
|
+
# Output: https://example.com/?page=1&id=123
|
|
121
|
+
|
|
122
|
+
# Via uv tool/uvx
|
|
123
|
+
$ uvx url-normalize www.foo.com:80/foo
|
|
124
|
+
# Output: https://www.foo.com:80/foo
|
|
125
|
+
```
|
|
126
|
+
|
|
99
127
|
## Documentation
|
|
100
128
|
|
|
101
129
|
For a complete history of changes, see [CHANGELOG.md](CHANGELOG.md).
|
|
@@ -1,5 +1,8 @@
|
|
|
1
1
|
# url-normalize
|
|
2
2
|
|
|
3
|
+
[](https://github.com/niksite/url-normalize/actions/workflows/ci.yml)
|
|
4
|
+
[](https://github.com/niksite/url-normalize/actions/workflows/publish.yml)
|
|
5
|
+
|
|
3
6
|
URI Normalization function:
|
|
4
7
|
|
|
5
8
|
* Take care of IDN domains.
|
|
@@ -38,8 +41,6 @@ pip install url-normalize
|
|
|
38
41
|
|
|
39
42
|
## Usage
|
|
40
43
|
|
|
41
|
-
Basic usage:
|
|
42
|
-
|
|
43
44
|
```python
|
|
44
45
|
from url_normalize import url_normalize
|
|
45
46
|
|
|
@@ -55,13 +56,15 @@ print(url_normalize("www.foo.com/foo", default_scheme="http"))
|
|
|
55
56
|
print(url_normalize("www.google.com/search?q=test&utm_source=test", filter_params=True))
|
|
56
57
|
# Output: https://www.google.com/search?q=test
|
|
57
58
|
|
|
58
|
-
# With custom parameter allowlist
|
|
59
|
+
# With custom parameter allowlist as a dict
|
|
59
60
|
print(url_normalize(
|
|
60
61
|
"example.com?page=1&id=123&ref=test",
|
|
61
62
|
filter_params=True,
|
|
62
63
|
param_allowlist={"example.com": ["page", "id"]}
|
|
63
64
|
))
|
|
64
65
|
# Output: https://example.com?page=1&id=123
|
|
66
|
+
|
|
67
|
+
# With custom parameter allowlist as a list
|
|
65
68
|
print(url_normalize(
|
|
66
69
|
"example.com?page=1&id=123&ref=test",
|
|
67
70
|
filter_params=True,
|
|
@@ -70,6 +73,31 @@ print(url_normalize(
|
|
|
70
73
|
# Output: https://example.com?page=1&id=123
|
|
71
74
|
```
|
|
72
75
|
|
|
76
|
+
### Command-line usage
|
|
77
|
+
|
|
78
|
+
You can also use `url-normalize` from the command line:
|
|
79
|
+
|
|
80
|
+
```bash
|
|
81
|
+
$ url-normalize "www.foo.com:80/foo"
|
|
82
|
+
# Output: https://www.foo.com/foo
|
|
83
|
+
|
|
84
|
+
# With custom default scheme
|
|
85
|
+
$ url-normalize -s http "www.foo.com/foo"
|
|
86
|
+
# Output: http://www.foo.com/foo
|
|
87
|
+
|
|
88
|
+
# With query parameter filtering
|
|
89
|
+
$ url-normalize -f "www.google.com/search?q=test&utm_source=test"
|
|
90
|
+
# Output: https://www.google.com/search?q=test
|
|
91
|
+
|
|
92
|
+
# With custom allowlist
|
|
93
|
+
$ url-normalize -f -p page,id "example.com?page=1&id=123&ref=test"
|
|
94
|
+
# Output: https://example.com/?page=1&id=123
|
|
95
|
+
|
|
96
|
+
# Via uv tool/uvx
|
|
97
|
+
$ uvx url-normalize www.foo.com:80/foo
|
|
98
|
+
# Output: https://www.foo.com:80/foo
|
|
99
|
+
```
|
|
100
|
+
|
|
73
101
|
## Documentation
|
|
74
102
|
|
|
75
103
|
For a complete history of changes, see [CHANGELOG.md](CHANGELOG.md).
|
|
@@ -1,9 +1,9 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "url-normalize"
|
|
3
|
-
version = "2.
|
|
3
|
+
version = "2.1.0"
|
|
4
4
|
description = "URL normalization for Python"
|
|
5
5
|
authors = [{ name = "Nikolay Panov", email = "github@npanov.com" }]
|
|
6
|
-
license = "MIT"
|
|
6
|
+
license = { text = "MIT" }
|
|
7
7
|
readme = "README.md"
|
|
8
8
|
requires-python = ">=3.8"
|
|
9
9
|
keywords = ["url", "normalization", "normalize"]
|
|
@@ -15,6 +15,9 @@ Repository = "https://github.com/niksite/url-normalize"
|
|
|
15
15
|
Issues = "https://github.com/niksite/url-normalize/issues"
|
|
16
16
|
Changelog = "https://github.com/niksite/url-normalize/blob/master/CHANGELOG.md"
|
|
17
17
|
|
|
18
|
+
[project.scripts]
|
|
19
|
+
url-normalize = "url_normalize.cli:main"
|
|
20
|
+
|
|
18
21
|
[project.optional-dependencies]
|
|
19
22
|
dev = [
|
|
20
23
|
"mypy",
|
|
@@ -0,0 +1,233 @@
|
|
|
1
|
+
"""Tests for the command line interface."""
|
|
2
|
+
|
|
3
|
+
import subprocess
|
|
4
|
+
import sys
|
|
5
|
+
from unittest.mock import patch
|
|
6
|
+
|
|
7
|
+
import pytest
|
|
8
|
+
|
|
9
|
+
from url_normalize import __version__
|
|
10
|
+
from url_normalize.cli import main
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def run_cli(*args: str) -> subprocess.CompletedProcess:
|
|
14
|
+
"""Run the CLI command with given arguments.
|
|
15
|
+
|
|
16
|
+
Params:
|
|
17
|
+
*args: Command line arguments to pass to the CLI.
|
|
18
|
+
|
|
19
|
+
Returns:
|
|
20
|
+
A completed process with stdout, stderr, and return code.
|
|
21
|
+
|
|
22
|
+
"""
|
|
23
|
+
command = [sys.executable, "-m", "url_normalize.cli", *list(args)]
|
|
24
|
+
return subprocess.run( # noqa: S603
|
|
25
|
+
command, capture_output=True, text=True, check=False
|
|
26
|
+
)
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def test_cli_error_handling(capsys, monkeypatch):
|
|
30
|
+
"""Test CLI error handling when URL normalization fails."""
|
|
31
|
+
with patch("url_normalize.cli.url_normalize") as mock_normalize:
|
|
32
|
+
mock_normalize.side_effect = Exception("Simulated error")
|
|
33
|
+
monkeypatch.setattr("sys.argv", ["url-normalize", "http://example.com"])
|
|
34
|
+
|
|
35
|
+
with pytest.raises(SystemExit) as excinfo:
|
|
36
|
+
main()
|
|
37
|
+
|
|
38
|
+
assert excinfo.value.code == 1
|
|
39
|
+
captured = capsys.readouterr()
|
|
40
|
+
assert "Error normalizing URL: Simulated error" in captured.err
|
|
41
|
+
assert not captured.out
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def test_cli_basic_normalization() -> None:
|
|
45
|
+
"""Test basic URL normalization via CLI."""
|
|
46
|
+
url = "http://EXAMPLE.com/./path/../other/"
|
|
47
|
+
expected = "http://example.com/other/"
|
|
48
|
+
|
|
49
|
+
result = run_cli(url)
|
|
50
|
+
|
|
51
|
+
assert result.returncode == 0
|
|
52
|
+
assert result.stdout.strip() == expected
|
|
53
|
+
assert not result.stderr
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def test_cli_basic_normalization_short_args() -> None:
|
|
57
|
+
"""Test basic URL normalization via CLI using short arguments."""
|
|
58
|
+
url = "http://EXAMPLE.com/./path/../other/"
|
|
59
|
+
expected = "http://example.com/other/"
|
|
60
|
+
# Using short args where applicable (none for the URL itself)
|
|
61
|
+
|
|
62
|
+
result = run_cli(url) # No short args needed for basic case
|
|
63
|
+
|
|
64
|
+
assert result.returncode == 0
|
|
65
|
+
assert result.stdout.strip() == expected
|
|
66
|
+
assert not result.stderr
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def test_cli_default_scheme() -> None:
|
|
70
|
+
"""Test default scheme addition via CLI."""
|
|
71
|
+
url = "//example.com"
|
|
72
|
+
expected = "https://example.com/"
|
|
73
|
+
|
|
74
|
+
result = run_cli(url)
|
|
75
|
+
|
|
76
|
+
assert result.returncode == 0
|
|
77
|
+
assert result.stdout.strip() == expected
|
|
78
|
+
assert not result.stderr
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def test_cli_default_scheme_short_arg() -> None:
|
|
82
|
+
"""Test default scheme addition via CLI using short argument."""
|
|
83
|
+
url = "//example.com"
|
|
84
|
+
expected = "https://example.com/"
|
|
85
|
+
|
|
86
|
+
result = run_cli(url) # Default scheme is implicit, no arg needed
|
|
87
|
+
|
|
88
|
+
assert result.returncode == 0
|
|
89
|
+
assert result.stdout.strip() == expected
|
|
90
|
+
assert not result.stderr
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def test_cli_custom_default_scheme() -> None:
|
|
94
|
+
"""Test custom default scheme via CLI."""
|
|
95
|
+
url = "//example.com"
|
|
96
|
+
expected = "ftp://example.com/"
|
|
97
|
+
|
|
98
|
+
result = run_cli("--default-scheme", "ftp", url)
|
|
99
|
+
|
|
100
|
+
assert result.returncode == 0
|
|
101
|
+
assert result.stdout.strip() == expected
|
|
102
|
+
assert not result.stderr
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def test_cli_custom_default_scheme_short_arg() -> None:
|
|
106
|
+
"""Test custom default scheme via CLI using short argument."""
|
|
107
|
+
url = "//example.com"
|
|
108
|
+
expected = "ftp://example.com/"
|
|
109
|
+
|
|
110
|
+
result = run_cli("-s", "ftp", url)
|
|
111
|
+
|
|
112
|
+
assert result.returncode == 0
|
|
113
|
+
assert result.stdout.strip() == expected
|
|
114
|
+
assert not result.stderr
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def test_cli_filter_params() -> None:
|
|
118
|
+
"""Test parameter filtering via CLI."""
|
|
119
|
+
url = "http://google.com?utm_source=test&q=1"
|
|
120
|
+
expected = "http://google.com/?q=1"
|
|
121
|
+
|
|
122
|
+
result = run_cli("--filter-params", url)
|
|
123
|
+
|
|
124
|
+
assert result.returncode == 0
|
|
125
|
+
assert result.stdout.strip() == expected
|
|
126
|
+
assert not result.stderr
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def test_cli_filter_params_short_arg() -> None:
|
|
130
|
+
"""Test parameter filtering via CLI using short argument."""
|
|
131
|
+
url = "http://google.com?utm_source=test&q=1"
|
|
132
|
+
expected = "http://google.com/?q=1"
|
|
133
|
+
|
|
134
|
+
result = run_cli("-f", url)
|
|
135
|
+
|
|
136
|
+
assert result.returncode == 0
|
|
137
|
+
assert result.stdout.strip() == expected
|
|
138
|
+
assert not result.stderr
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def test_cli_param_allowlist() -> None:
|
|
142
|
+
"""Test parameter allowlist via CLI."""
|
|
143
|
+
url = "http://example.com?remove=me&keep=this&remove_too=true"
|
|
144
|
+
expected = "http://example.com/?keep=this"
|
|
145
|
+
# Use filter_params to enable filtering, then allowlist to keep specific ones
|
|
146
|
+
|
|
147
|
+
result = run_cli("-f", "-p", "keep", url)
|
|
148
|
+
|
|
149
|
+
assert result.returncode == 0
|
|
150
|
+
assert result.stdout.strip() == expected
|
|
151
|
+
assert not result.stderr
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def test_cli_param_allowlist_multiple() -> None:
|
|
155
|
+
"""Test parameter allowlist with multiple params via CLI."""
|
|
156
|
+
url = "http://example.com?remove=me&keep=this&keep_too=yes&remove_too=true"
|
|
157
|
+
expected = "http://example.com/?keep=this&keep_too=yes"
|
|
158
|
+
|
|
159
|
+
result = run_cli("-f", "-p", "keep,keep_too", url)
|
|
160
|
+
|
|
161
|
+
assert result.returncode == 0
|
|
162
|
+
assert result.stdout.strip() == expected
|
|
163
|
+
assert not result.stderr
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def test_cli_param_allowlist_without_filtering() -> None:
|
|
167
|
+
"""Test allowlist has no effect if filtering is not enabled."""
|
|
168
|
+
url = "http://example.com?remove=me&keep=this&remove_too=true"
|
|
169
|
+
expected = "http://example.com/?remove=me&keep=this&remove_too=true"
|
|
170
|
+
# Not using -f, so allowlist should be ignored
|
|
171
|
+
|
|
172
|
+
result = run_cli("-p", "keep", url)
|
|
173
|
+
|
|
174
|
+
assert result.returncode == 0
|
|
175
|
+
assert result.stdout.strip() == expected
|
|
176
|
+
assert not result.stderr
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
def test_cli_no_url() -> None:
|
|
180
|
+
"""Test CLI error when no URL is provided."""
|
|
181
|
+
result = run_cli()
|
|
182
|
+
|
|
183
|
+
assert result.returncode != 0
|
|
184
|
+
assert "the following arguments are required: url" in result.stderr
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
def test_cli_version_long() -> None:
|
|
188
|
+
"""Test version output with --version flag."""
|
|
189
|
+
result = run_cli("--version")
|
|
190
|
+
|
|
191
|
+
assert result.returncode == 0
|
|
192
|
+
assert __version__ in result.stdout
|
|
193
|
+
assert not result.stderr
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def test_cli_version_short() -> None:
|
|
197
|
+
"""Test version output with -v flag."""
|
|
198
|
+
result = run_cli("-v")
|
|
199
|
+
|
|
200
|
+
assert result.returncode == 0
|
|
201
|
+
assert __version__ in result.stdout
|
|
202
|
+
assert not result.stderr
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
@pytest.mark.skipif(
|
|
206
|
+
sys.platform == "win32", reason="Charset handling differs on Windows CLI"
|
|
207
|
+
)
|
|
208
|
+
def test_cli_charset() -> None:
|
|
209
|
+
"""Test charset handling via CLI (might be platform-dependent)."""
|
|
210
|
+
# Example using Cyrillic characters which need correct encoding
|
|
211
|
+
url = "http://пример.рф/path"
|
|
212
|
+
expected_idn = "http://xn--e1afmkfd.xn--p1ai/path"
|
|
213
|
+
|
|
214
|
+
# Test with default UTF-8
|
|
215
|
+
result_utf8 = run_cli(url)
|
|
216
|
+
|
|
217
|
+
assert result_utf8.returncode == 0
|
|
218
|
+
assert result_utf8.stdout.strip() == expected_idn
|
|
219
|
+
assert not result_utf8.stderr
|
|
220
|
+
|
|
221
|
+
# Test specifying UTF-8 explicitly
|
|
222
|
+
result_charset = run_cli("--charset", "utf-8", url)
|
|
223
|
+
|
|
224
|
+
assert result_charset.returncode == 0
|
|
225
|
+
assert result_charset.stdout.strip() == expected_idn
|
|
226
|
+
assert not result_charset.stderr
|
|
227
|
+
|
|
228
|
+
# Test specifying UTF-8 explicitly using short arg
|
|
229
|
+
result_charset_short = run_cli("-c", "utf-8", url)
|
|
230
|
+
|
|
231
|
+
assert result_charset_short.returncode == 0
|
|
232
|
+
assert result_charset_short.stdout.strip() == expected_idn
|
|
233
|
+
assert not result_charset_short.stderr
|
|
@@ -87,6 +87,8 @@ NO_CHANGES_EXPECTED: Final[tuple[str, ...]] = (
|
|
|
87
87
|
"tel:+1-816-555-1212",
|
|
88
88
|
"telnet://192.0.2.16:80/",
|
|
89
89
|
"urn:oasis:names:specification:docbook:dtd:xml:4.1.2",
|
|
90
|
+
# Issue #36: Fragment with '=' should not be encoded
|
|
91
|
+
"https://docs.google.com/spreadsheets/d/abcd/edit#gid=1234",
|
|
90
92
|
)
|
|
91
93
|
|
|
92
94
|
|
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
#!/usr/bin/env python
|
|
2
|
+
"""Command line interface for url-normalize."""
|
|
3
|
+
|
|
4
|
+
import argparse
|
|
5
|
+
import sys
|
|
6
|
+
from importlib.metadata import version
|
|
7
|
+
|
|
8
|
+
from .url_normalize import url_normalize
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def main() -> None:
|
|
12
|
+
"""Parse arguments and run url_normalize."""
|
|
13
|
+
parser = argparse.ArgumentParser(description="Normalize a URL.")
|
|
14
|
+
parser.add_argument(
|
|
15
|
+
"-v",
|
|
16
|
+
"--version",
|
|
17
|
+
action="version",
|
|
18
|
+
version=f"%(prog)s {version('url-normalize')}",
|
|
19
|
+
)
|
|
20
|
+
parser.add_argument("url", help="The URL to normalize.")
|
|
21
|
+
parser.add_argument(
|
|
22
|
+
"-c",
|
|
23
|
+
"--charset",
|
|
24
|
+
default="utf-8",
|
|
25
|
+
help="The charset of the URL. Default: utf-8",
|
|
26
|
+
)
|
|
27
|
+
parser.add_argument(
|
|
28
|
+
"-s",
|
|
29
|
+
"--default-scheme",
|
|
30
|
+
default="https",
|
|
31
|
+
help="The default scheme to use if missing. Default: https",
|
|
32
|
+
)
|
|
33
|
+
parser.add_argument(
|
|
34
|
+
"-f",
|
|
35
|
+
"--filter-params",
|
|
36
|
+
action="store_true",
|
|
37
|
+
help="Filter common tracking parameters.",
|
|
38
|
+
)
|
|
39
|
+
parser.add_argument(
|
|
40
|
+
"-p",
|
|
41
|
+
"--param-allowlist",
|
|
42
|
+
type=str,
|
|
43
|
+
help="Comma-separated list of query parameters to allow (e.g., 'q,id').",
|
|
44
|
+
)
|
|
45
|
+
|
|
46
|
+
args = parser.parse_args()
|
|
47
|
+
|
|
48
|
+
allowlist = args.param_allowlist.split(",") if args.param_allowlist else None
|
|
49
|
+
|
|
50
|
+
try:
|
|
51
|
+
normalized_url = url_normalize(
|
|
52
|
+
args.url,
|
|
53
|
+
charset=args.charset,
|
|
54
|
+
default_scheme=args.default_scheme,
|
|
55
|
+
filter_params=args.filter_params,
|
|
56
|
+
param_allowlist=allowlist,
|
|
57
|
+
)
|
|
58
|
+
except Exception as e: # noqa: BLE001
|
|
59
|
+
print(f"Error normalizing URL: {e}", file=sys.stderr) # noqa: T201
|
|
60
|
+
sys.exit(1)
|
|
61
|
+
else:
|
|
62
|
+
print(normalized_url) # noqa: T201
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
if __name__ == "__main__":
|
|
66
|
+
main() # pragma: no cover
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
"""URL fragment normalization."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from .tools import quote, unquote
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def normalize_fragment(fragment: str) -> str:
|
|
9
|
+
"""Normalize fragment part of the url.
|
|
10
|
+
|
|
11
|
+
Params:
|
|
12
|
+
fragment : string : url fragment, e.g., 'fragment'
|
|
13
|
+
|
|
14
|
+
Returns:
|
|
15
|
+
string : normalized fragment data.
|
|
16
|
+
|
|
17
|
+
Notes:
|
|
18
|
+
According to RFC 3986, the following characters are allowed in a fragment:
|
|
19
|
+
fragment = *( pchar / "/" / "?" )
|
|
20
|
+
pchar = unreserved / pct-encoded / sub-delims / ":" / "@"
|
|
21
|
+
unreserved = ALPHA / DIGIT / "-" / "." / "_" / "~"
|
|
22
|
+
sub-delims = "!" / "$" / "&" / "'" / "(" / ")" / "*" / "+" / "," / ";" / "="
|
|
23
|
+
We specifically allow "~" and "=" as safe characters during normalization.
|
|
24
|
+
Other sub-delimiters could potentially be added to the `safe` list if needed.
|
|
25
|
+
|
|
26
|
+
"""
|
|
27
|
+
return quote(unquote(fragment), safe="~=")
|
|
@@ -1,9 +1,9 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: url-normalize
|
|
3
|
-
Version: 2.
|
|
3
|
+
Version: 2.1.0
|
|
4
4
|
Summary: URL normalization for Python
|
|
5
5
|
Author-email: Nikolay Panov <github@npanov.com>
|
|
6
|
-
License
|
|
6
|
+
License: MIT
|
|
7
7
|
Project-URL: Homepage, https://github.com/niksite/url-normalize
|
|
8
8
|
Project-URL: Repository, https://github.com/niksite/url-normalize
|
|
9
9
|
Project-URL: Issues, https://github.com/niksite/url-normalize/issues
|
|
@@ -26,6 +26,9 @@ Dynamic: license-file
|
|
|
26
26
|
|
|
27
27
|
# url-normalize
|
|
28
28
|
|
|
29
|
+
[](https://github.com/niksite/url-normalize/actions/workflows/ci.yml)
|
|
30
|
+
[](https://github.com/niksite/url-normalize/actions/workflows/publish.yml)
|
|
31
|
+
|
|
29
32
|
URI Normalization function:
|
|
30
33
|
|
|
31
34
|
* Take care of IDN domains.
|
|
@@ -64,8 +67,6 @@ pip install url-normalize
|
|
|
64
67
|
|
|
65
68
|
## Usage
|
|
66
69
|
|
|
67
|
-
Basic usage:
|
|
68
|
-
|
|
69
70
|
```python
|
|
70
71
|
from url_normalize import url_normalize
|
|
71
72
|
|
|
@@ -81,13 +82,15 @@ print(url_normalize("www.foo.com/foo", default_scheme="http"))
|
|
|
81
82
|
print(url_normalize("www.google.com/search?q=test&utm_source=test", filter_params=True))
|
|
82
83
|
# Output: https://www.google.com/search?q=test
|
|
83
84
|
|
|
84
|
-
# With custom parameter allowlist
|
|
85
|
+
# With custom parameter allowlist as a dict
|
|
85
86
|
print(url_normalize(
|
|
86
87
|
"example.com?page=1&id=123&ref=test",
|
|
87
88
|
filter_params=True,
|
|
88
89
|
param_allowlist={"example.com": ["page", "id"]}
|
|
89
90
|
))
|
|
90
91
|
# Output: https://example.com?page=1&id=123
|
|
92
|
+
|
|
93
|
+
# With custom parameter allowlist as a list
|
|
91
94
|
print(url_normalize(
|
|
92
95
|
"example.com?page=1&id=123&ref=test",
|
|
93
96
|
filter_params=True,
|
|
@@ -96,6 +99,31 @@ print(url_normalize(
|
|
|
96
99
|
# Output: https://example.com?page=1&id=123
|
|
97
100
|
```
|
|
98
101
|
|
|
102
|
+
### Command-line usage
|
|
103
|
+
|
|
104
|
+
You can also use `url-normalize` from the command line:
|
|
105
|
+
|
|
106
|
+
```bash
|
|
107
|
+
$ url-normalize "www.foo.com:80/foo"
|
|
108
|
+
# Output: https://www.foo.com/foo
|
|
109
|
+
|
|
110
|
+
# With custom default scheme
|
|
111
|
+
$ url-normalize -s http "www.foo.com/foo"
|
|
112
|
+
# Output: http://www.foo.com/foo
|
|
113
|
+
|
|
114
|
+
# With query parameter filtering
|
|
115
|
+
$ url-normalize -f "www.google.com/search?q=test&utm_source=test"
|
|
116
|
+
# Output: https://www.google.com/search?q=test
|
|
117
|
+
|
|
118
|
+
# With custom allowlist
|
|
119
|
+
$ url-normalize -f -p page,id "example.com?page=1&id=123&ref=test"
|
|
120
|
+
# Output: https://example.com/?page=1&id=123
|
|
121
|
+
|
|
122
|
+
# Via uv tool/uvx
|
|
123
|
+
$ uvx url-normalize www.foo.com:80/foo
|
|
124
|
+
# Output: https://www.foo.com:80/foo
|
|
125
|
+
```
|
|
126
|
+
|
|
99
127
|
## Documentation
|
|
100
128
|
|
|
101
129
|
For a complete history of changes, see [CHANGELOG.md](CHANGELOG.md).
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
LICENSE
|
|
2
2
|
README.md
|
|
3
3
|
pyproject.toml
|
|
4
|
+
tests/test_cli.py
|
|
4
5
|
tests/test_deconstruct_url.py
|
|
5
6
|
tests/test_generic_url_cleanup.py
|
|
6
7
|
tests/test_normalize_fragment.py
|
|
@@ -16,6 +17,7 @@ tests/test_reconstruct_url.py
|
|
|
16
17
|
tests/test_tools.py
|
|
17
18
|
tests/test_url_normalize.py
|
|
18
19
|
url_normalize/__init__.py
|
|
20
|
+
url_normalize/cli.py
|
|
19
21
|
url_normalize/generic_url_cleanup.py
|
|
20
22
|
url_normalize/normalize_fragment.py
|
|
21
23
|
url_normalize/normalize_host.py
|
|
@@ -31,5 +33,6 @@ url_normalize/url_normalize.py
|
|
|
31
33
|
url_normalize.egg-info/PKG-INFO
|
|
32
34
|
url_normalize.egg-info/SOURCES.txt
|
|
33
35
|
url_normalize.egg-info/dependency_links.txt
|
|
36
|
+
url_normalize.egg-info/entry_points.txt
|
|
34
37
|
url_normalize.egg-info/requires.txt
|
|
35
38
|
url_normalize.egg-info/top_level.txt
|
|
@@ -1,18 +0,0 @@
|
|
|
1
|
-
"""URL fragment normalization."""
|
|
2
|
-
|
|
3
|
-
from __future__ import annotations
|
|
4
|
-
|
|
5
|
-
from .tools import quote, unquote
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
def normalize_fragment(fragment: str) -> str:
|
|
9
|
-
"""Normalize fragment part of the url.
|
|
10
|
-
|
|
11
|
-
Params:
|
|
12
|
-
fragment : string : url fragment, e.g., 'fragment'
|
|
13
|
-
|
|
14
|
-
Returns:
|
|
15
|
-
string : normalized fragment data.
|
|
16
|
-
|
|
17
|
-
"""
|
|
18
|
-
return quote(unquote(fragment), "~")
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|