kliz 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- kliz/__init__.py +29 -0
- kliz/_http.py +64 -0
- kliz/_validation.py +34 -0
- kliz/cli.py +164 -0
- kliz/core.py +177 -0
- kliz/exceptions.py +24 -0
- kliz/providers/__init__.py +8 -0
- kliz/providers/base.py +32 -0
- kliz/providers/batch.py +79 -0
- kliz/providers/google.py +105 -0
- kliz/providers/indexnow.py +88 -0
- kliz/py.typed +1 -0
- kliz/results.py +15 -0
- kliz-0.2.0.dist-info/METADATA +436 -0
- kliz-0.2.0.dist-info/RECORD +19 -0
- kliz-0.2.0.dist-info/WHEEL +5 -0
- kliz-0.2.0.dist-info/entry_points.txt +2 -0
- kliz-0.2.0.dist-info/licenses/LICENSE +21 -0
- kliz-0.2.0.dist-info/top_level.txt +1 -0
kliz/__init__.py
ADDED
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
"""Public API for kliz."""
|
|
2
|
+
|
|
3
|
+
from importlib.metadata import PackageNotFoundError, version
|
|
4
|
+
|
|
5
|
+
from kliz.core import Kliz
|
|
6
|
+
from kliz.exceptions import KlizError, ProviderError
|
|
7
|
+
from kliz.providers import (
|
|
8
|
+
BaseProvider,
|
|
9
|
+
BatchProvider,
|
|
10
|
+
GoogleProvider,
|
|
11
|
+
IndexNowProvider,
|
|
12
|
+
)
|
|
13
|
+
from kliz.results import NotificationResult
|
|
14
|
+
|
|
15
|
+
__all__ = [
|
|
16
|
+
"BaseProvider",
|
|
17
|
+
"BatchProvider",
|
|
18
|
+
"GoogleProvider",
|
|
19
|
+
"IndexNowProvider",
|
|
20
|
+
"Kliz",
|
|
21
|
+
"KlizError",
|
|
22
|
+
"NotificationResult",
|
|
23
|
+
"ProviderError",
|
|
24
|
+
]
|
|
25
|
+
|
|
26
|
+
try:
|
|
27
|
+
__version__ = version("kliz")
|
|
28
|
+
except PackageNotFoundError: # pragma: no cover - source tree without installation
|
|
29
|
+
__version__ = "0.0.0"
|
kliz/_http.py
ADDED
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
"""Shared HTTP helpers for indexing providers."""
|
|
2
|
+
|
|
3
|
+
from collections.abc import Mapping
|
|
4
|
+
from typing import Any, Optional
|
|
5
|
+
|
|
6
|
+
import requests
|
|
7
|
+
|
|
8
|
+
from kliz.exceptions import ProviderError
|
|
9
|
+
|
|
10
|
+
DEFAULT_SUCCESS_STATUSES: set[int] = {200, 202}
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def create_session(session: Optional[requests.Session] = None) -> requests.Session:
|
|
14
|
+
"""Return *session* or a fresh :class:`requests.Session`."""
|
|
15
|
+
|
|
16
|
+
return session if session is not None else requests.Session()
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def post_json(
|
|
20
|
+
session: requests.Session,
|
|
21
|
+
url: str,
|
|
22
|
+
*,
|
|
23
|
+
payload: Mapping[str, Any],
|
|
24
|
+
timeout: float,
|
|
25
|
+
provider: str,
|
|
26
|
+
) -> requests.Response:
|
|
27
|
+
"""POST JSON and map transport failures to :class:`ProviderError`."""
|
|
28
|
+
|
|
29
|
+
try:
|
|
30
|
+
return session.post(url, json=dict(payload), timeout=timeout)
|
|
31
|
+
except (requests.Timeout, requests.ConnectionError) as exc:
|
|
32
|
+
raise ProviderError(
|
|
33
|
+
f"{provider} could not be reached",
|
|
34
|
+
provider=provider,
|
|
35
|
+
retryable=True,
|
|
36
|
+
) from exc
|
|
37
|
+
except requests.RequestException as exc:
|
|
38
|
+
raise ProviderError(
|
|
39
|
+
f"{provider} request failed",
|
|
40
|
+
provider=provider,
|
|
41
|
+
) from exc
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def raise_for_indexing_status(
|
|
45
|
+
response: requests.Response,
|
|
46
|
+
*,
|
|
47
|
+
provider: str,
|
|
48
|
+
success_statuses: Optional[set[int]] = None,
|
|
49
|
+
) -> bool:
|
|
50
|
+
"""Return ``True`` for accepted statuses or raise :class:`ProviderError`."""
|
|
51
|
+
|
|
52
|
+
accepted = (
|
|
53
|
+
success_statuses if success_statuses is not None else DEFAULT_SUCCESS_STATUSES
|
|
54
|
+
)
|
|
55
|
+
if response.status_code in accepted:
|
|
56
|
+
return True
|
|
57
|
+
|
|
58
|
+
retryable = response.status_code == 429 or response.status_code >= 500
|
|
59
|
+
raise ProviderError(
|
|
60
|
+
f"{provider} rejected the notification with HTTP {response.status_code}",
|
|
61
|
+
provider=provider,
|
|
62
|
+
retryable=retryable,
|
|
63
|
+
status_code=response.status_code,
|
|
64
|
+
)
|
kliz/_validation.py
ADDED
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
"""Shared input validation helpers."""
|
|
2
|
+
|
|
3
|
+
from urllib.parse import SplitResult, urlsplit
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
def parse_http_url(url: str, *, require_clean: bool = False) -> SplitResult:
|
|
7
|
+
"""Return a parsed absolute HTTP(S) URL or raise ``ValueError``.
|
|
8
|
+
|
|
9
|
+
A fragment (``#...``) is always rejected because it is never sent to the
|
|
10
|
+
server: ``page`` and ``page#top`` address the same resource, so a fragment
|
|
11
|
+
can never be meaningful for indexing and may even carry auth tokens.
|
|
12
|
+
|
|
13
|
+
A query string (``?...``) is rejected only when ``require_clean`` is true.
|
|
14
|
+
The server does receive the query, so it can address a real resource;
|
|
15
|
+
whether a clean canonical URL is mandatory is the caller's decision.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
if not isinstance(url, str) or not url.strip():
|
|
19
|
+
raise ValueError("url must be a non-empty string")
|
|
20
|
+
|
|
21
|
+
normalized_url = url.strip()
|
|
22
|
+
parsed_url = urlsplit(normalized_url)
|
|
23
|
+
if parsed_url.scheme.lower() not in {"http", "https"}:
|
|
24
|
+
raise ValueError("url must use the http or https scheme")
|
|
25
|
+
if not parsed_url.hostname:
|
|
26
|
+
raise ValueError("url must include a hostname")
|
|
27
|
+
if parsed_url.username is not None or parsed_url.password is not None:
|
|
28
|
+
raise ValueError("url must not contain credentials")
|
|
29
|
+
if parsed_url.fragment:
|
|
30
|
+
raise ValueError("url must not contain a fragment")
|
|
31
|
+
if require_clean and parsed_url.query:
|
|
32
|
+
raise ValueError("url must not contain a query string")
|
|
33
|
+
|
|
34
|
+
return parsed_url
|
kliz/cli.py
ADDED
|
@@ -0,0 +1,164 @@
|
|
|
1
|
+
"""CLI entry point for kliz."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
import os
|
|
7
|
+
import sys
|
|
8
|
+
from typing import Callable
|
|
9
|
+
|
|
10
|
+
from kliz import Kliz, __version__
|
|
11
|
+
from kliz.exceptions import KlizError
|
|
12
|
+
from kliz.providers import GoogleProvider, IndexNowProvider
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class ConfigurationError(ValueError):
|
|
16
|
+
"""Raised when the CLI is misconfigured."""
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def main(argv: list[str] | None = None) -> int:
|
|
20
|
+
parser = _build_parser()
|
|
21
|
+
try:
|
|
22
|
+
args = parser.parse_args(argv)
|
|
23
|
+
except SystemExit as exc:
|
|
24
|
+
code: int = exc.code if isinstance(exc.code, int) else 0
|
|
25
|
+
return code
|
|
26
|
+
try:
|
|
27
|
+
command: Callable[[argparse.Namespace], int] = args.command
|
|
28
|
+
return command(args)
|
|
29
|
+
except ConfigurationError as exc:
|
|
30
|
+
print(f"error: {exc}", file=sys.stderr)
|
|
31
|
+
return 2
|
|
32
|
+
except KlizError as exc:
|
|
33
|
+
print(f"error: {exc}", file=sys.stderr)
|
|
34
|
+
return 1
|
|
35
|
+
except KeyboardInterrupt:
|
|
36
|
+
print("\naborted", file=sys.stderr)
|
|
37
|
+
return 130
|
|
38
|
+
except Exception as exc: # noqa: BLE001
|
|
39
|
+
print(f"error: unexpected error: {exc}", file=sys.stderr)
|
|
40
|
+
return 1
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _build_parser() -> argparse.ArgumentParser:
|
|
44
|
+
parser = argparse.ArgumentParser(
|
|
45
|
+
prog="kliz",
|
|
46
|
+
description=(
|
|
47
|
+
"Bot d'indexation SEO agnostique pour notifier les moteurs de recherche."
|
|
48
|
+
),
|
|
49
|
+
)
|
|
50
|
+
parser.add_argument(
|
|
51
|
+
"--version",
|
|
52
|
+
action="version",
|
|
53
|
+
version=__version__,
|
|
54
|
+
)
|
|
55
|
+
parser.add_argument(
|
|
56
|
+
"--indexnow-api-key",
|
|
57
|
+
default=os.environ.get("KLIZ_INDEXNOW_API_KEY"),
|
|
58
|
+
help="IndexNow API key (KLIZ_INDEXNOW_API_KEY env var)",
|
|
59
|
+
)
|
|
60
|
+
parser.add_argument(
|
|
61
|
+
"--indexnow-key-location",
|
|
62
|
+
default=os.environ.get("KLIZ_INDEXNOW_KEY_LOCATION"),
|
|
63
|
+
help="URL where the IndexNow key file is hosted",
|
|
64
|
+
)
|
|
65
|
+
parser.add_argument(
|
|
66
|
+
"--google-service-account-file",
|
|
67
|
+
default=os.environ.get("KLIZ_GOOGLE_SERVICE_ACCOUNT_FILE"),
|
|
68
|
+
help="Path to the Google service account JSON file",
|
|
69
|
+
)
|
|
70
|
+
sub = parser.add_subparsers(dest="command", required=True)
|
|
71
|
+
p_notify = sub.add_parser("notify", help="Notify providers of a URL")
|
|
72
|
+
p_notify.add_argument("url", nargs="?", default=None, help="URL to notify")
|
|
73
|
+
p_notify.add_argument(
|
|
74
|
+
"--batch",
|
|
75
|
+
default=None,
|
|
76
|
+
help="File with one URL per line",
|
|
77
|
+
)
|
|
78
|
+
p_notify.set_defaults(command=_cmd_notify)
|
|
79
|
+
p_providers = sub.add_parser("providers", help="List configured providers")
|
|
80
|
+
p_providers.set_defaults(command=_cmd_providers)
|
|
81
|
+
return parser
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def _cmd_notify(args: argparse.Namespace) -> int:
|
|
85
|
+
urls = _resolve_urls(args)
|
|
86
|
+
indexer = _build_indexer(args)
|
|
87
|
+
all_ok = True
|
|
88
|
+
for url in urls:
|
|
89
|
+
results = indexer.notify_all_detailed(url)
|
|
90
|
+
failures = {
|
|
91
|
+
name: result.error for name, result in results.items() if not result.success
|
|
92
|
+
}
|
|
93
|
+
if failures:
|
|
94
|
+
all_ok = False
|
|
95
|
+
for name, error in failures.items():
|
|
96
|
+
print(
|
|
97
|
+
f"❌ {url} → {name}: {error}",
|
|
98
|
+
file=sys.stderr,
|
|
99
|
+
)
|
|
100
|
+
else:
|
|
101
|
+
names = ", ".join(results)
|
|
102
|
+
print(f"✅ {url} → {names}: OK")
|
|
103
|
+
return 0 if all_ok else 1
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def _cmd_providers(args: argparse.Namespace) -> int:
|
|
107
|
+
indexer = _build_indexer(args)
|
|
108
|
+
for provider in indexer.providers:
|
|
109
|
+
kind = provider.__class__.__name__
|
|
110
|
+
print(f"{kind} ({provider.name})")
|
|
111
|
+
return 0
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def _resolve_urls(args: argparse.Namespace) -> list[str]:
|
|
115
|
+
if args.batch:
|
|
116
|
+
return _read_urls_from_file(args.batch)
|
|
117
|
+
if args.url:
|
|
118
|
+
return [args.url]
|
|
119
|
+
raise ConfigurationError("provide a URL or --batch <file>")
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def _read_urls_from_file(path: str) -> list[str]:
|
|
123
|
+
try:
|
|
124
|
+
with open(path) as handle:
|
|
125
|
+
urls = [
|
|
126
|
+
line.strip()
|
|
127
|
+
for line in handle
|
|
128
|
+
if line.strip() and not line.startswith("#")
|
|
129
|
+
]
|
|
130
|
+
except OSError as exc:
|
|
131
|
+
raise ConfigurationError(f"cannot read {path}: {exc}") from exc
|
|
132
|
+
if not urls:
|
|
133
|
+
raise ConfigurationError(f"no URLs found in {path}")
|
|
134
|
+
return urls
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
def _build_indexer(args: argparse.Namespace) -> Kliz:
|
|
138
|
+
providers: list[IndexNowProvider | GoogleProvider] = []
|
|
139
|
+
if args.indexnow_api_key:
|
|
140
|
+
if not args.indexnow_key_location:
|
|
141
|
+
raise ConfigurationError(
|
|
142
|
+
"--indexnow-key-location is required when --indexnow-api-key is set",
|
|
143
|
+
)
|
|
144
|
+
providers.append(
|
|
145
|
+
IndexNowProvider(
|
|
146
|
+
api_key=args.indexnow_api_key,
|
|
147
|
+
key_location=args.indexnow_key_location,
|
|
148
|
+
),
|
|
149
|
+
)
|
|
150
|
+
if args.google_service_account_file:
|
|
151
|
+
if not os.path.exists(args.google_service_account_file):
|
|
152
|
+
raise ConfigurationError(
|
|
153
|
+
f"service account file not found: {args.google_service_account_file}",
|
|
154
|
+
)
|
|
155
|
+
providers.append(
|
|
156
|
+
GoogleProvider(args.google_service_account_file),
|
|
157
|
+
)
|
|
158
|
+
if not providers:
|
|
159
|
+
raise ConfigurationError(
|
|
160
|
+
"no providers configured: set --indexnow-api-key + "
|
|
161
|
+
"--indexnow-key-location or --google-service-account-file "
|
|
162
|
+
"(or the KLIZ_* env vars)",
|
|
163
|
+
)
|
|
164
|
+
return Kliz(providers)
|
kliz/core.py
ADDED
|
@@ -0,0 +1,177 @@
|
|
|
1
|
+
"""Provider orchestration for kliz."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import random
|
|
6
|
+
import time
|
|
7
|
+
from collections import Counter
|
|
8
|
+
from collections.abc import Callable, Iterable, Sequence
|
|
9
|
+
from typing import Any
|
|
10
|
+
|
|
11
|
+
from kliz.exceptions import ProviderError
|
|
12
|
+
from kliz.providers.base import BaseProvider
|
|
13
|
+
from kliz.results import NotificationResult
|
|
14
|
+
|
|
15
|
+
_RETRY_BASE_DELAY = 1.0
|
|
16
|
+
_JITTER_RANGE = 0.25
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class Kliz:
|
|
20
|
+
"""Dispatch indexing notifications to a collection of providers."""
|
|
21
|
+
|
|
22
|
+
def __init__(
|
|
23
|
+
self,
|
|
24
|
+
providers: Iterable[BaseProvider],
|
|
25
|
+
*,
|
|
26
|
+
max_attempts: int = 1,
|
|
27
|
+
sleep: Callable[[float], None] = time.sleep,
|
|
28
|
+
clock: Callable[[], float] = time.monotonic,
|
|
29
|
+
) -> None:
|
|
30
|
+
if isinstance(providers, (str, bytes)):
|
|
31
|
+
raise TypeError("providers must be an iterable of BaseProvider instances")
|
|
32
|
+
|
|
33
|
+
provider_list = list(providers)
|
|
34
|
+
if not all(isinstance(provider, BaseProvider) for provider in provider_list):
|
|
35
|
+
raise TypeError("every provider must inherit from BaseProvider")
|
|
36
|
+
|
|
37
|
+
if (
|
|
38
|
+
isinstance(max_attempts, bool)
|
|
39
|
+
or not isinstance(max_attempts, int)
|
|
40
|
+
or max_attempts < 1
|
|
41
|
+
):
|
|
42
|
+
raise ValueError("max_attempts must be a positive integer")
|
|
43
|
+
|
|
44
|
+
self.providers = tuple(provider_list)
|
|
45
|
+
self.max_attempts = max_attempts
|
|
46
|
+
self._sleep = sleep
|
|
47
|
+
self._clock = clock
|
|
48
|
+
|
|
49
|
+
def close(self) -> None:
|
|
50
|
+
"""Close every provider, releasing pooled connections."""
|
|
51
|
+
|
|
52
|
+
for provider in self.providers:
|
|
53
|
+
provider.close()
|
|
54
|
+
|
|
55
|
+
def __enter__(self) -> Kliz:
|
|
56
|
+
return self
|
|
57
|
+
|
|
58
|
+
def __exit__(self, *exc: Any) -> None:
|
|
59
|
+
self.close()
|
|
60
|
+
|
|
61
|
+
def notify_all(self, url: str) -> dict[str, bool]:
|
|
62
|
+
"""Notify every provider and return simple boolean statuses."""
|
|
63
|
+
|
|
64
|
+
return {
|
|
65
|
+
name: result.success
|
|
66
|
+
for name, result in self.notify_all_detailed(url).items()
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
def notify_all_detailed(self, url: str) -> dict[str, NotificationResult]:
|
|
70
|
+
"""Notify all providers without hiding error and retry information."""
|
|
71
|
+
|
|
72
|
+
return {
|
|
73
|
+
name: self._notify_provider(provider, url)
|
|
74
|
+
for provider, name in zip(self.providers, self._result_names())
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
def notify_many(self, urls: Sequence[str]) -> dict[str, bool]:
|
|
78
|
+
"""Notify every provider with a batch of URLs and return boolean statuses."""
|
|
79
|
+
|
|
80
|
+
return {
|
|
81
|
+
name: all(result.success for result in results)
|
|
82
|
+
for name, results in self.notify_many_detailed(urls).items()
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
def notify_many_detailed(
|
|
86
|
+
self, urls: Sequence[str]
|
|
87
|
+
) -> dict[str, list[NotificationResult]]:
|
|
88
|
+
"""Notify all providers with multiple URLs.
|
|
89
|
+
|
|
90
|
+
Preserves per-chunk outcome details without hiding errors.
|
|
91
|
+
"""
|
|
92
|
+
|
|
93
|
+
url_list = self._validate_urls(urls)
|
|
94
|
+
return {
|
|
95
|
+
name: self._notify_provider_many(provider, url_list)
|
|
96
|
+
for provider, name in zip(self.providers, self._result_names())
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
def _notify_provider(self, provider: BaseProvider, url: str) -> NotificationResult:
|
|
100
|
+
return self._execute_with_retry(provider, provider.notify, url)
|
|
101
|
+
|
|
102
|
+
def _notify_provider_many(
|
|
103
|
+
self, provider: BaseProvider, urls: list[str]
|
|
104
|
+
) -> list[NotificationResult]:
|
|
105
|
+
notify_many_fn = getattr(provider, "notify_many", None)
|
|
106
|
+
if callable(notify_many_fn):
|
|
107
|
+
max_urls = getattr(provider, "max_urls_per_request", None)
|
|
108
|
+
if isinstance(max_urls, int) and max_urls > 0:
|
|
109
|
+
chunks = [urls[i : i + max_urls] for i in range(0, len(urls), max_urls)]
|
|
110
|
+
else:
|
|
111
|
+
chunks = [urls]
|
|
112
|
+
|
|
113
|
+
return [
|
|
114
|
+
self._execute_with_retry(provider, notify_many_fn, chunk)
|
|
115
|
+
for chunk in chunks
|
|
116
|
+
]
|
|
117
|
+
|
|
118
|
+
return [
|
|
119
|
+
self._execute_with_retry(provider, provider.notify, url) for url in urls
|
|
120
|
+
]
|
|
121
|
+
|
|
122
|
+
def _execute_with_retry(
|
|
123
|
+
self, provider: BaseProvider, func: Callable[..., Any], *args: Any
|
|
124
|
+
) -> NotificationResult:
|
|
125
|
+
for attempt in range(1, self.max_attempts + 1):
|
|
126
|
+
try:
|
|
127
|
+
success = bool(func(*args))
|
|
128
|
+
return NotificationResult(
|
|
129
|
+
provider=provider.name,
|
|
130
|
+
success=success,
|
|
131
|
+
error=None if success else "provider returned False",
|
|
132
|
+
)
|
|
133
|
+
except ProviderError as exc:
|
|
134
|
+
if attempt < self.max_attempts and exc.retryable:
|
|
135
|
+
self._sleep_between_attempts(attempt)
|
|
136
|
+
continue
|
|
137
|
+
return NotificationResult(
|
|
138
|
+
provider=provider.name,
|
|
139
|
+
success=False,
|
|
140
|
+
retryable=exc.retryable,
|
|
141
|
+
error=str(exc),
|
|
142
|
+
status_code=exc.status_code,
|
|
143
|
+
)
|
|
144
|
+
except Exception as exc:
|
|
145
|
+
return NotificationResult(
|
|
146
|
+
provider=provider.name,
|
|
147
|
+
success=False,
|
|
148
|
+
error=str(exc) or exc.__class__.__name__,
|
|
149
|
+
)
|
|
150
|
+
raise AssertionError("unreachable") # pragma: no cover
|
|
151
|
+
|
|
152
|
+
def _validate_urls(self, urls: Sequence[str]) -> list[str]:
|
|
153
|
+
if isinstance(urls, (str, bytes)):
|
|
154
|
+
raise TypeError("urls must be a sequence of strings, not a string or bytes")
|
|
155
|
+
try:
|
|
156
|
+
url_list = list(urls)
|
|
157
|
+
except TypeError as exc:
|
|
158
|
+
raise TypeError("urls must be an iterable sequence of strings") from exc
|
|
159
|
+
|
|
160
|
+
if not url_list:
|
|
161
|
+
raise ValueError("urls must be a non-empty sequence")
|
|
162
|
+
return url_list
|
|
163
|
+
|
|
164
|
+
def _sleep_between_attempts(self, attempt: int) -> None:
|
|
165
|
+
delay = _RETRY_BASE_DELAY * (2 ** (attempt - 1))
|
|
166
|
+
delay += random.Random(self._clock()).uniform(0.0, _JITTER_RANGE)
|
|
167
|
+
self._sleep(delay)
|
|
168
|
+
|
|
169
|
+
def _result_names(self) -> list[str]:
|
|
170
|
+
counts: Counter[str] = Counter()
|
|
171
|
+
names: list[str] = []
|
|
172
|
+
for provider in self.providers:
|
|
173
|
+
counts[provider.name] += 1
|
|
174
|
+
occurrence = counts[provider.name]
|
|
175
|
+
suffix = "" if occurrence == 1 else f"#{occurrence}"
|
|
176
|
+
names.append(f"{provider.name}{suffix}")
|
|
177
|
+
return names
|
kliz/exceptions.py
ADDED
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
"""Exceptions exposed by kliz."""
|
|
2
|
+
|
|
3
|
+
from typing import Optional
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
class KlizError(Exception):
|
|
7
|
+
"""Base class for all kliz-specific errors."""
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class ProviderError(KlizError):
|
|
11
|
+
"""An indexing provider could not complete a notification."""
|
|
12
|
+
|
|
13
|
+
def __init__(
|
|
14
|
+
self,
|
|
15
|
+
message: str,
|
|
16
|
+
*,
|
|
17
|
+
provider: str,
|
|
18
|
+
retryable: bool = False,
|
|
19
|
+
status_code: Optional[int] = None,
|
|
20
|
+
) -> None:
|
|
21
|
+
super().__init__(message)
|
|
22
|
+
self.provider = provider
|
|
23
|
+
self.retryable = retryable
|
|
24
|
+
self.status_code = status_code
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
"""Built-in indexing providers."""
|
|
2
|
+
|
|
3
|
+
from kliz.providers.base import BaseProvider
|
|
4
|
+
from kliz.providers.batch import BatchProvider
|
|
5
|
+
from kliz.providers.google import GoogleProvider
|
|
6
|
+
from kliz.providers.indexnow import IndexNowProvider
|
|
7
|
+
|
|
8
|
+
__all__ = ["BaseProvider", "BatchProvider", "GoogleProvider", "IndexNowProvider"]
|
kliz/providers/base.py
ADDED
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
"""Contracts implemented by indexing providers."""
|
|
2
|
+
|
|
3
|
+
from abc import ABC, abstractmethod
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
class BaseProvider(ABC):
|
|
7
|
+
"""Strategy interface for a search-engine indexing provider."""
|
|
8
|
+
|
|
9
|
+
@property
|
|
10
|
+
def name(self) -> str:
|
|
11
|
+
"""Return the provider name used in orchestration results."""
|
|
12
|
+
|
|
13
|
+
return self.__class__.__name__
|
|
14
|
+
|
|
15
|
+
@abstractmethod
|
|
16
|
+
def notify(self, url: str) -> bool:
|
|
17
|
+
"""Notify the provider that *url* was updated.
|
|
18
|
+
|
|
19
|
+
Implementations return ``True`` after a successful notification and
|
|
20
|
+
raise :class:`kliz.exceptions.ProviderError` when the remote service
|
|
21
|
+
rejects the request or cannot be reached.
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
raise NotImplementedError
|
|
25
|
+
|
|
26
|
+
def close(self) -> None: # noqa: B027
|
|
27
|
+
"""Release resources held by this provider.
|
|
28
|
+
|
|
29
|
+
The default implementation is a no-op. Subclasses that hold
|
|
30
|
+
persistent connections (HTTP sessions, gRPC channels, etc.)
|
|
31
|
+
should override this method.
|
|
32
|
+
"""
|
kliz/providers/batch.py
ADDED
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
"""Reusable batch provider base for host-scoped indexing APIs."""
|
|
2
|
+
|
|
3
|
+
from abc import abstractmethod
|
|
4
|
+
from collections.abc import Sequence
|
|
5
|
+
from typing import Optional
|
|
6
|
+
from urllib.parse import SplitResult
|
|
7
|
+
|
|
8
|
+
import requests
|
|
9
|
+
|
|
10
|
+
from kliz._http import create_session
|
|
11
|
+
from kliz._validation import parse_http_url
|
|
12
|
+
from kliz.providers.base import BaseProvider
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class BatchProvider(BaseProvider):
|
|
16
|
+
"""Provider that notifies one or many URLs belonging to the same host.
|
|
17
|
+
|
|
18
|
+
Subclasses implement :meth:`_notify_many`. ``notify`` is implemented as a
|
|
19
|
+
single-URL batch so adapters only maintain one submission path.
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
max_urls_per_request: int = 1
|
|
23
|
+
|
|
24
|
+
def __init__(
|
|
25
|
+
self,
|
|
26
|
+
*,
|
|
27
|
+
timeout: float = 10.0,
|
|
28
|
+
session: Optional[requests.Session] = None,
|
|
29
|
+
) -> None:
|
|
30
|
+
if timeout <= 0:
|
|
31
|
+
raise ValueError("timeout must be greater than zero")
|
|
32
|
+
|
|
33
|
+
self.timeout = timeout
|
|
34
|
+
self._session = create_session(session)
|
|
35
|
+
|
|
36
|
+
def close(self) -> None:
|
|
37
|
+
"""Release the pooled connections held by this provider."""
|
|
38
|
+
|
|
39
|
+
self._session.close()
|
|
40
|
+
|
|
41
|
+
def notify(self, url: str) -> bool:
|
|
42
|
+
"""Notify the provider that *url* was updated."""
|
|
43
|
+
|
|
44
|
+
return self.notify_many([url])
|
|
45
|
+
|
|
46
|
+
def notify_many(self, urls: Sequence[str]) -> bool:
|
|
47
|
+
"""Submit up to ``max_urls_per_request`` URLs for the same host."""
|
|
48
|
+
|
|
49
|
+
normalized_urls, parsed_urls = self._validate_urls(urls)
|
|
50
|
+
return self._notify_many(normalized_urls, parsed_urls)
|
|
51
|
+
|
|
52
|
+
@abstractmethod
|
|
53
|
+
def _notify_many(
|
|
54
|
+
self,
|
|
55
|
+
urls: list[str],
|
|
56
|
+
parsed_urls: list[SplitResult],
|
|
57
|
+
) -> bool:
|
|
58
|
+
"""Submit an already validated same-host URL batch."""
|
|
59
|
+
|
|
60
|
+
raise NotImplementedError
|
|
61
|
+
|
|
62
|
+
def _validate_urls(
|
|
63
|
+
self, urls: Sequence[str]
|
|
64
|
+
) -> tuple[list[str], list[SplitResult]]:
|
|
65
|
+
if isinstance(urls, (str, bytes)) or not urls:
|
|
66
|
+
raise ValueError("urls must be a non-empty sequence")
|
|
67
|
+
if len(urls) > self.max_urls_per_request:
|
|
68
|
+
raise ValueError(
|
|
69
|
+
f"provider accepts at most {self.max_urls_per_request} URLs per request"
|
|
70
|
+
)
|
|
71
|
+
|
|
72
|
+
normalized_urls = [url.strip() for url in urls]
|
|
73
|
+
parsed_urls = [
|
|
74
|
+
parse_http_url(url, require_clean=True) for url in normalized_urls
|
|
75
|
+
]
|
|
76
|
+
hosts = {parsed.hostname.lower() for parsed in parsed_urls if parsed.hostname}
|
|
77
|
+
if len(hosts) != 1:
|
|
78
|
+
raise ValueError("all batch URLs must belong to the same host")
|
|
79
|
+
return normalized_urls, parsed_urls
|
kliz/providers/google.py
ADDED
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
"""Google Indexing API provider implementation."""
|
|
2
|
+
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
from typing import Any, Optional, Union
|
|
5
|
+
|
|
6
|
+
import google_auth_httplib2
|
|
7
|
+
import httplib2
|
|
8
|
+
from google.auth.exceptions import GoogleAuthError, TransportError
|
|
9
|
+
from google.oauth2 import service_account
|
|
10
|
+
from googleapiclient.discovery import build
|
|
11
|
+
from googleapiclient.errors import HttpError
|
|
12
|
+
|
|
13
|
+
from kliz._validation import parse_http_url
|
|
14
|
+
from kliz.exceptions import ProviderError
|
|
15
|
+
from kliz.providers.base import BaseProvider
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class GoogleProvider(BaseProvider):
|
|
19
|
+
"""Notify Google for eligible JobPosting or BroadcastEvent pages only."""
|
|
20
|
+
|
|
21
|
+
scope = "https://www.googleapis.com/auth/indexing"
|
|
22
|
+
|
|
23
|
+
def __init__(
|
|
24
|
+
self,
|
|
25
|
+
service_account_file: Union[str, Path],
|
|
26
|
+
*,
|
|
27
|
+
timeout: float = 60.0,
|
|
28
|
+
num_retries: int = 2,
|
|
29
|
+
) -> None:
|
|
30
|
+
if not str(service_account_file):
|
|
31
|
+
raise ValueError("service_account_file must not be empty")
|
|
32
|
+
if timeout <= 0:
|
|
33
|
+
raise ValueError("timeout must be greater than zero")
|
|
34
|
+
if num_retries < 0:
|
|
35
|
+
raise ValueError("num_retries must not be negative")
|
|
36
|
+
|
|
37
|
+
self.service_account_file = service_account_file
|
|
38
|
+
self.timeout = timeout
|
|
39
|
+
self.num_retries = num_retries
|
|
40
|
+
self._service: Optional[Any] = None
|
|
41
|
+
|
|
42
|
+
def notify(self, url: str) -> bool:
|
|
43
|
+
"""Publish a ``URL_UPDATED`` notification to Google."""
|
|
44
|
+
|
|
45
|
+
parse_http_url(url, require_clean=True)
|
|
46
|
+
normalized_url = url.strip()
|
|
47
|
+
service = self._get_service()
|
|
48
|
+
|
|
49
|
+
try:
|
|
50
|
+
(
|
|
51
|
+
service.urlNotifications()
|
|
52
|
+
.publish(body={"url": normalized_url, "type": "URL_UPDATED"})
|
|
53
|
+
.execute(num_retries=self.num_retries)
|
|
54
|
+
)
|
|
55
|
+
except HttpError as exc:
|
|
56
|
+
status_code = int(exc.resp.status)
|
|
57
|
+
retryable = status_code in {408, 429} or status_code >= 500
|
|
58
|
+
raise ProviderError(
|
|
59
|
+
f"Google rejected the notification with HTTP {status_code}",
|
|
60
|
+
provider=self.name,
|
|
61
|
+
retryable=retryable,
|
|
62
|
+
status_code=status_code,
|
|
63
|
+
) from exc
|
|
64
|
+
except (TransportError, httplib2.HttpLib2Error, OSError) as exc:
|
|
65
|
+
raise ProviderError(
|
|
66
|
+
"Google could not be reached",
|
|
67
|
+
provider=self.name,
|
|
68
|
+
retryable=True,
|
|
69
|
+
) from exc
|
|
70
|
+
return True
|
|
71
|
+
|
|
72
|
+
def _get_service(self) -> Any:
|
|
73
|
+
"""Return the Google API client, building it once on first use."""
|
|
74
|
+
|
|
75
|
+
if self._service is None:
|
|
76
|
+
self._service = self._build_service()
|
|
77
|
+
return self._service
|
|
78
|
+
|
|
79
|
+
def _build_service(self) -> Any:
|
|
80
|
+
"""Build the authenticated Google Indexing API client.
|
|
81
|
+
|
|
82
|
+
Only successes are cached: if building fails the provider retries on
|
|
83
|
+
a later notification instead of remaining broken forever.
|
|
84
|
+
"""
|
|
85
|
+
|
|
86
|
+
try:
|
|
87
|
+
credentials = service_account.Credentials.from_service_account_file( # type: ignore[no-untyped-call]
|
|
88
|
+
str(self.service_account_file),
|
|
89
|
+
scopes=[self.scope],
|
|
90
|
+
)
|
|
91
|
+
authorized_http = google_auth_httplib2.AuthorizedHttp(
|
|
92
|
+
credentials,
|
|
93
|
+
http=httplib2.Http(timeout=self.timeout),
|
|
94
|
+
)
|
|
95
|
+
return build(
|
|
96
|
+
"indexing",
|
|
97
|
+
"v3",
|
|
98
|
+
http=authorized_http,
|
|
99
|
+
cache_discovery=False,
|
|
100
|
+
)
|
|
101
|
+
except (OSError, ValueError, GoogleAuthError) as exc:
|
|
102
|
+
raise ProviderError(
|
|
103
|
+
"Google service account could not be loaded",
|
|
104
|
+
provider=self.name,
|
|
105
|
+
) from exc
|
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
"""IndexNow provider implementation."""
|
|
2
|
+
|
|
3
|
+
import re
|
|
4
|
+
from pathlib import PurePosixPath
|
|
5
|
+
from typing import Optional, Union
|
|
6
|
+
from urllib.parse import SplitResult
|
|
7
|
+
|
|
8
|
+
import requests
|
|
9
|
+
|
|
10
|
+
from kliz._http import post_json, raise_for_indexing_status
|
|
11
|
+
from kliz._validation import parse_http_url
|
|
12
|
+
from kliz.providers.batch import BatchProvider
|
|
13
|
+
|
|
14
|
+
PayloadValue = Union[str, list[str]]
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class IndexNowProvider(BatchProvider):
|
|
18
|
+
"""Notify search engines that support the IndexNow protocol."""
|
|
19
|
+
|
|
20
|
+
endpoint = "https://api.indexnow.org/indexnow"
|
|
21
|
+
max_urls_per_request = 10_000
|
|
22
|
+
_key_pattern = re.compile(r"^[A-Za-z0-9-]{8,128}$")
|
|
23
|
+
|
|
24
|
+
def __init__(
|
|
25
|
+
self,
|
|
26
|
+
api_key: str,
|
|
27
|
+
key_location: Optional[str] = None,
|
|
28
|
+
timeout: float = 10.0,
|
|
29
|
+
*,
|
|
30
|
+
session: Optional[requests.Session] = None,
|
|
31
|
+
) -> None:
|
|
32
|
+
if not isinstance(api_key, str) or not self._key_pattern.fullmatch(api_key):
|
|
33
|
+
raise ValueError(
|
|
34
|
+
"api_key must contain 8 to 128 letters, numbers, or dashes"
|
|
35
|
+
)
|
|
36
|
+
if key_location is not None:
|
|
37
|
+
parse_http_url(key_location, require_clean=True)
|
|
38
|
+
|
|
39
|
+
super().__init__(timeout=timeout, session=session)
|
|
40
|
+
self.api_key = api_key
|
|
41
|
+
self.key_location = key_location
|
|
42
|
+
|
|
43
|
+
def _notify_many(
|
|
44
|
+
self,
|
|
45
|
+
urls: list[str],
|
|
46
|
+
parsed_urls: list[SplitResult],
|
|
47
|
+
) -> bool:
|
|
48
|
+
host = parsed_urls[0].hostname
|
|
49
|
+
if host is None: # Defensive: parse_http_url already enforces this.
|
|
50
|
+
raise ValueError("url must include a hostname")
|
|
51
|
+
|
|
52
|
+
self._validate_key_location(parsed_urls[0])
|
|
53
|
+
payload: dict[str, PayloadValue] = {
|
|
54
|
+
"host": host,
|
|
55
|
+
"key": self.api_key,
|
|
56
|
+
"urlList": urls,
|
|
57
|
+
}
|
|
58
|
+
if self.key_location:
|
|
59
|
+
payload["keyLocation"] = self.key_location
|
|
60
|
+
|
|
61
|
+
response = post_json(
|
|
62
|
+
self._session,
|
|
63
|
+
self.endpoint,
|
|
64
|
+
payload=payload,
|
|
65
|
+
timeout=self.timeout,
|
|
66
|
+
provider=self.name,
|
|
67
|
+
)
|
|
68
|
+
return raise_for_indexing_status(response, provider=self.name)
|
|
69
|
+
|
|
70
|
+
def _validate_key_location(self, submitted_url: SplitResult) -> None:
|
|
71
|
+
if self.key_location is None:
|
|
72
|
+
return
|
|
73
|
+
|
|
74
|
+
key_url = parse_http_url(self.key_location, require_clean=True)
|
|
75
|
+
if key_url.hostname is None or submitted_url.hostname is None:
|
|
76
|
+
raise ValueError("key_location and url must include a hostname")
|
|
77
|
+
if key_url.hostname.lower() != submitted_url.hostname.lower():
|
|
78
|
+
raise ValueError("key_location must use the same host as the submitted URL")
|
|
79
|
+
|
|
80
|
+
key_directory = str(PurePosixPath(key_url.path).parent)
|
|
81
|
+
if key_directory == ".":
|
|
82
|
+
key_directory = "/"
|
|
83
|
+
normalized_directory = key_directory.rstrip("/") + "/"
|
|
84
|
+
submitted_path = submitted_url.path or "/"
|
|
85
|
+
if normalized_directory != "/" and not submitted_path.startswith(
|
|
86
|
+
normalized_directory
|
|
87
|
+
):
|
|
88
|
+
raise ValueError("url must be within the path covered by key_location")
|
kliz/py.typed
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
kliz/results.py
ADDED
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
"""Result models returned by the kliz orchestrator."""
|
|
2
|
+
|
|
3
|
+
from dataclasses import dataclass
|
|
4
|
+
from typing import Optional
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
@dataclass(frozen=True)
|
|
8
|
+
class NotificationResult:
|
|
9
|
+
"""Detailed outcome of one provider notification."""
|
|
10
|
+
|
|
11
|
+
provider: str
|
|
12
|
+
success: bool
|
|
13
|
+
retryable: bool = False
|
|
14
|
+
error: Optional[str] = None
|
|
15
|
+
status_code: Optional[int] = None
|
|
@@ -0,0 +1,436 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: kliz
|
|
3
|
+
Version: 0.2.0
|
|
4
|
+
Summary: Bot d'indexation SEO agnostique pour notifier les moteurs de recherche.
|
|
5
|
+
Author: Freddy Choudja
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/freddychoudja/kliz-
|
|
8
|
+
Project-URL: Repository, https://github.com/freddychoudja/kliz-.git
|
|
9
|
+
Project-URL: Issues, https://github.com/freddychoudja/kliz-/issues
|
|
10
|
+
Project-URL: Documentation, https://github.com/freddychoudja/kliz-#readme
|
|
11
|
+
Keywords: seo,indexing,indexnow,google
|
|
12
|
+
Classifier: Development Status :: 3 - Alpha
|
|
13
|
+
Classifier: Intended Audience :: Developers
|
|
14
|
+
Classifier: Programming Language :: Python :: 3
|
|
15
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
22
|
+
Classifier: Topic :: Internet :: WWW/HTTP
|
|
23
|
+
Requires-Python: >=3.9
|
|
24
|
+
Description-Content-Type: text/markdown
|
|
25
|
+
License-File: LICENSE
|
|
26
|
+
Requires-Dist: requests>=2.28.0
|
|
27
|
+
Requires-Dist: google-auth>=2.0.0
|
|
28
|
+
Requires-Dist: google-auth-httplib2>=0.1.0
|
|
29
|
+
Requires-Dist: google-api-python-client>=2.0.0
|
|
30
|
+
Requires-Dist: httplib2<1.0.0,>=0.19.0
|
|
31
|
+
Provides-Extra: test
|
|
32
|
+
Requires-Dist: pytest<10.0,>=8.0; extra == "test"
|
|
33
|
+
Requires-Dist: pytest-cov<8.0,>=5.0; extra == "test"
|
|
34
|
+
Provides-Extra: dev
|
|
35
|
+
Requires-Dist: build<2.0,>=1.2; extra == "dev"
|
|
36
|
+
Requires-Dist: mypy<3.0,>=1.11; extra == "dev"
|
|
37
|
+
Requires-Dist: pip-audit<3.0,>=2.7; extra == "dev"
|
|
38
|
+
Requires-Dist: pytest<10.0,>=8.0; extra == "dev"
|
|
39
|
+
Requires-Dist: pytest-cov<8.0,>=5.0; extra == "dev"
|
|
40
|
+
Requires-Dist: ruff<1.0,>=0.9; extra == "dev"
|
|
41
|
+
Requires-Dist: twine<8.0,>=5.1; extra == "dev"
|
|
42
|
+
Requires-Dist: types-requests>=2.28.0; extra == "dev"
|
|
43
|
+
Dynamic: license-file
|
|
44
|
+
|
|
45
|
+
# kliz
|
|
46
|
+
|
|
47
|
+
[](https://github.com/freddychoudja/kliz-/actions/workflows/ci.yml)
|
|
48
|
+
[](https://pypi.org/project/kliz/)
|
|
49
|
+
[](https://github.com/freddychoudja/kliz-/issues)
|
|
50
|
+
[](LICENSE)
|
|
51
|
+
|
|
52
|
+
`kliz` est un bot d'indexation SEO agnostique. Il permet à une application de
|
|
53
|
+
notifier plusieurs moteurs de recherche dès qu'une URL est créée ou mise à
|
|
54
|
+
jour.
|
|
55
|
+
|
|
56
|
+
Le package ne dépend ni de Django, ni de Celery, ni de Redis. Il expose une API
|
|
57
|
+
Python synchrone que l'application appelante peut exécuter directement ou
|
|
58
|
+
encapsuler dans le système de tâches de son choix.
|
|
59
|
+
|
|
60
|
+
## Installation
|
|
61
|
+
|
|
62
|
+
```bash
|
|
63
|
+
pip install kliz
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
Une documentation web statique est disponible dans
|
|
67
|
+
[`docs/index.html`](docs/index.html). Elle peut aussi être publiée via GitHub
|
|
68
|
+
Pages avec le workflow fourni.
|
|
69
|
+
|
|
70
|
+
Une traduction anglaise est disponible dans
|
|
71
|
+
[`README.en.md`](README.en.md).
|
|
72
|
+
|
|
73
|
+
Pour contribuer et exécuter les tests :
|
|
74
|
+
|
|
75
|
+
```bash
|
|
76
|
+
python -m pip install -e ".[dev]"
|
|
77
|
+
pytest --cov=kliz
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
## Démarrage rapide
|
|
81
|
+
|
|
82
|
+
```python
|
|
83
|
+
from kliz import GoogleProvider, IndexNowProvider, Kliz
|
|
84
|
+
|
|
85
|
+
indexer = Kliz(
|
|
86
|
+
[
|
|
87
|
+
IndexNowProvider(
|
|
88
|
+
api_key="votre-cle-indexnow",
|
|
89
|
+
key_location="https://example.com/votre-cle-indexnow.txt",
|
|
90
|
+
),
|
|
91
|
+
GoogleProvider("/run/secrets/google-service-account.json"),
|
|
92
|
+
]
|
|
93
|
+
)
|
|
94
|
+
|
|
95
|
+
statuses = indexer.notify_all("https://example.com/articles/nouvel-article")
|
|
96
|
+
# {
|
|
97
|
+
# "IndexNowProvider": True,
|
|
98
|
+
# "GoogleProvider": True,
|
|
99
|
+
# }
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
Pour soumettre plusieurs URL d'un coup, `notify_many` découpe selon
|
|
103
|
+
`max_urls_per_request` (lots IndexNow) et retombe sur une boucle `notify` pour
|
|
104
|
+
les autres providers :
|
|
105
|
+
|
|
106
|
+
```python
|
|
107
|
+
statuses = indexer.notify_many(
|
|
108
|
+
[
|
|
109
|
+
"https://example.com/articles/a",
|
|
110
|
+
"https://example.com/articles/b",
|
|
111
|
+
]
|
|
112
|
+
)
|
|
113
|
+
```
|
|
114
|
+
|
|
115
|
+
Le retry intégré est **désactivé par défaut** (`max_attempts=1`). Pour l'activer
|
|
116
|
+
avec backoff exponentiel et jitter :
|
|
117
|
+
|
|
118
|
+
```python
|
|
119
|
+
indexer = Kliz(
|
|
120
|
+
[IndexNowProvider(api_key="votre-cle-indexnow")],
|
|
121
|
+
max_attempts=3,
|
|
122
|
+
)
|
|
123
|
+
```
|
|
124
|
+
|
|
125
|
+
`notify_all` continue d'appeler les autres fournisseurs lorsqu'un fournisseur
|
|
126
|
+
échoue. Son statut vaut alors `False`. Un appel direct à `provider.notify(url)`
|
|
127
|
+
laisse en revanche remonter une `ProviderError` afin que l'application puisse
|
|
128
|
+
appliquer sa propre politique de retry.
|
|
129
|
+
|
|
130
|
+
Pour obtenir la cause, le statut HTTP et l'indication de retry :
|
|
131
|
+
|
|
132
|
+
```python
|
|
133
|
+
results = indexer.notify_all_detailed(
|
|
134
|
+
"https://example.com/articles/nouvel-article"
|
|
135
|
+
)
|
|
136
|
+
|
|
137
|
+
for name, result in results.items():
|
|
138
|
+
print(name, result.success, result.retryable, result.error)
|
|
139
|
+
```
|
|
140
|
+
|
|
141
|
+
Si plusieurs instances ont le même nom, leurs clés sont suffixées :
|
|
142
|
+
`IndexNowProvider`, `IndexNowProvider#2`, etc.
|
|
143
|
+
|
|
144
|
+
## Architecture agnostique
|
|
145
|
+
|
|
146
|
+
`BaseProvider` définit une stratégie minimale : `notify(url) -> bool`. Chaque
|
|
147
|
+
adaptateur traduit ce contrat vers l'API distante concernée :
|
|
148
|
+
|
|
149
|
+
- `IndexNowProvider` envoie une requête HTTP à l'API IndexNow ;
|
|
150
|
+
- `GoogleProvider` publie une notification `URL_UPDATED` via l'API Google
|
|
151
|
+
Indexing ;
|
|
152
|
+
- `Kliz` orchestre les stratégies injectées dans son constructeur.
|
|
153
|
+
|
|
154
|
+
Cette séparation permet d'ajouter un moteur sans modifier l'orchestrateur et
|
|
155
|
+
laisse l'application libre de choisir son framework web, sa file d'attente et
|
|
156
|
+
sa politique de retry.
|
|
157
|
+
|
|
158
|
+
Un fournisseur personnalisé doit uniquement hériter de `BaseProvider` :
|
|
159
|
+
|
|
160
|
+
```python
|
|
161
|
+
from kliz import BaseProvider
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
class CustomProvider(BaseProvider):
|
|
165
|
+
def notify(self, url: str) -> bool:
|
|
166
|
+
# Appel vers l'API du moteur concerné
|
|
167
|
+
return True
|
|
168
|
+
```
|
|
169
|
+
|
|
170
|
+
Pour un moteur qui accepte des lots d'URL sur le même hôte, héritez de
|
|
171
|
+
`BatchProvider` : `notify` et la validation (hôte commun, taille max, URL
|
|
172
|
+
propres) sont fournis ; il reste à implémenter `_notify_many`.
|
|
173
|
+
|
|
174
|
+
```python
|
|
175
|
+
from urllib.parse import SplitResult
|
|
176
|
+
|
|
177
|
+
from kliz import BatchProvider
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
class CustomBatchProvider(BatchProvider):
|
|
181
|
+
max_urls_per_request = 100
|
|
182
|
+
|
|
183
|
+
def _notify_many(
|
|
184
|
+
self, urls: list[str], parsed_urls: list[SplitResult]
|
|
185
|
+
) -> bool:
|
|
186
|
+
# Appel HTTP groupé vers le moteur
|
|
187
|
+
return True
|
|
188
|
+
```
|
|
189
|
+
|
|
190
|
+
## Validation des URL
|
|
191
|
+
|
|
192
|
+
Toutes les URL soumises à un provider sont contrôlées avant tout envoi :
|
|
193
|
+
|
|
194
|
+
- le schéma doit être `http` ou `https` et l'hôte doit être présent ;
|
|
195
|
+
- les identifiants (`https://user:pass@...`) sont interdits ;
|
|
196
|
+
- les fragments (`#...`) sont toujours rejetés : ils ne sont jamais transmis au
|
|
197
|
+
serveur et ne peuvent donc désigner un contenu distinct ;
|
|
198
|
+
- les chaînes de requête (`?...`) sont rejetées pour les notifications : seule
|
|
199
|
+
une URL canonique propre est soumise aux moteurs.
|
|
200
|
+
|
|
201
|
+
La fonction partagée `parse_http_url(url, require_clean=True)` applique ces
|
|
202
|
+
règles. `require_clean` vaut `False` par défaut afin de ne pas casser les
|
|
203
|
+
usages existants ; seules les notifications exigent une URL propre.
|
|
204
|
+
|
|
205
|
+
## Configuration des fournisseurs
|
|
206
|
+
|
|
207
|
+
### IndexNow
|
|
208
|
+
|
|
209
|
+
La clé doit être publiée conformément aux règles d'IndexNow. Si
|
|
210
|
+
`key_location` est fourni, il est transmis dans le champ `keyLocation`.
|
|
211
|
+
|
|
212
|
+
```python
|
|
213
|
+
from kliz import IndexNowProvider
|
|
214
|
+
|
|
215
|
+
provider = IndexNowProvider(
|
|
216
|
+
api_key="votre-cle-valide",
|
|
217
|
+
key_location="https://example.com/votre-cle-valide.txt", # optionnel
|
|
218
|
+
timeout=10.0,
|
|
219
|
+
)
|
|
220
|
+
provider.notify("https://example.com/page")
|
|
221
|
+
```
|
|
222
|
+
|
|
223
|
+
Le provider réutilise une connexion HTTP persistante (`requests.Session`) entre
|
|
224
|
+
les notifications, afin de ne pas reconstruire une connexion et une poignée de
|
|
225
|
+
main TLS à chaque appel. Vous pouvez injecter votre propre session (tests,
|
|
226
|
+
configuration réseau partagée, proxies) :
|
|
227
|
+
|
|
228
|
+
```python
|
|
229
|
+
import requests
|
|
230
|
+
|
|
231
|
+
provider = IndexNowProvider(
|
|
232
|
+
api_key="votre-cle-valide",
|
|
233
|
+
session=requests.Session(),
|
|
234
|
+
)
|
|
235
|
+
```
|
|
236
|
+
|
|
237
|
+
La session interne garde les connexions ouvertes ; appelez `provider.close()` à
|
|
238
|
+
l'arrêt de votre application pour les libérer proprement.
|
|
239
|
+
|
|
240
|
+
Pour soumettre plusieurs URL du même hôte dans un seul appel :
|
|
241
|
+
|
|
242
|
+
```python
|
|
243
|
+
provider.notify_many(
|
|
244
|
+
[
|
|
245
|
+
"https://example.com/page-1",
|
|
246
|
+
"https://example.com/page-2",
|
|
247
|
+
]
|
|
248
|
+
)
|
|
249
|
+
```
|
|
250
|
+
|
|
251
|
+
IndexNow accepte jusqu'à 10 000 URL par requête. `kliz` classe les erreurs
|
|
252
|
+
`429` et `5xx` comme retentables.
|
|
253
|
+
|
|
254
|
+
### Google
|
|
255
|
+
|
|
256
|
+
Activez l'API Google Indexing pour votre projet, créez un compte de service et
|
|
257
|
+
autorisez-le sur la propriété concernée. Ne versionnez jamais le fichier JSON
|
|
258
|
+
du compte de service.
|
|
259
|
+
|
|
260
|
+
> **Restriction importante :** l'API Google Indexing est officiellement
|
|
261
|
+
> réservée aux pages contenant un `JobPosting` ou un `BroadcastEvent` intégré
|
|
262
|
+
> dans un `VideoObject`. N'utilisez pas ce provider comme API d'indexation
|
|
263
|
+
> générique pour les autres contenus ; utilisez notamment un sitemap pour leur
|
|
264
|
+
> couverture.
|
|
265
|
+
|
|
266
|
+
```python
|
|
267
|
+
from kliz import GoogleProvider
|
|
268
|
+
|
|
269
|
+
provider = GoogleProvider(
|
|
270
|
+
"/run/secrets/google-service-account.json",
|
|
271
|
+
timeout=60.0,
|
|
272
|
+
num_retries=2,
|
|
273
|
+
)
|
|
274
|
+
provider.notify("https://example.com/jobs/backend-python")
|
|
275
|
+
```
|
|
276
|
+
|
|
277
|
+
L'API Google Indexing est soumise aux règles d'éligibilité et aux quotas de
|
|
278
|
+
Google. Une notification ne garantit pas l'indexation de l'URL.
|
|
279
|
+
|
|
280
|
+
Le client Indexing est construit de manière paresseuse : le fichier de compte
|
|
281
|
+
de service n'est lu qu'au premier appel de `notify`, puis réutilisé pour les
|
|
282
|
+
appels suivants. La création du provider ne déclenche donc aucune lecture de
|
|
283
|
+
fichier. Les erreurs de configuration (fichier absent, JSON invalide)
|
|
284
|
+
remontent au moment de la notification, sont marquées comme non retentables, et
|
|
285
|
+
le provider se rétablit dès que le fichier est corrigé.
|
|
286
|
+
|
|
287
|
+
## Recettes / Intégration Asynchrone
|
|
288
|
+
|
|
289
|
+
`kliz` reste volontairement synchrone. Pour une exécution asynchrone, placez
|
|
290
|
+
l'appel dans un worker, une tâche ou un job appartenant à votre application.
|
|
291
|
+
Ainsi, les dépendances d'infrastructure ne contaminent pas le package.
|
|
292
|
+
|
|
293
|
+
### Tâche Celery (Python/Django)
|
|
294
|
+
|
|
295
|
+
Dans un projet Django utilisant déjà Celery, la tâche peut lire sa
|
|
296
|
+
configuration depuis les settings et laisser Celery gérer les retries :
|
|
297
|
+
|
|
298
|
+
```python
|
|
299
|
+
# myapp/tasks.py — ce code appartient à l'application, pas à kliz
|
|
300
|
+
from dataclasses import asdict
|
|
301
|
+
|
|
302
|
+
from celery import shared_task
|
|
303
|
+
from django.conf import settings
|
|
304
|
+
|
|
305
|
+
from kliz import IndexNowProvider, Kliz
|
|
306
|
+
|
|
307
|
+
|
|
308
|
+
@shared_task(bind=True, max_retries=5)
|
|
309
|
+
def notify_search_engines(self, url: str) -> dict[str, dict[str, object]]:
|
|
310
|
+
indexer = Kliz(
|
|
311
|
+
[
|
|
312
|
+
IndexNowProvider(
|
|
313
|
+
api_key=settings.INDEXNOW_API_KEY,
|
|
314
|
+
key_location=settings.INDEXNOW_KEY_LOCATION,
|
|
315
|
+
),
|
|
316
|
+
]
|
|
317
|
+
)
|
|
318
|
+
results = indexer.notify_all_detailed(url)
|
|
319
|
+
retryable = [result for result in results.values() if result.retryable]
|
|
320
|
+
|
|
321
|
+
if retryable:
|
|
322
|
+
raise self.retry(
|
|
323
|
+
exc=RuntimeError("temporary indexing provider failure"),
|
|
324
|
+
countdown=min(60 * (2**self.request.retries), 3600),
|
|
325
|
+
)
|
|
326
|
+
|
|
327
|
+
return {name: asdict(result) for name, result in results.items()}
|
|
328
|
+
```
|
|
329
|
+
|
|
330
|
+
Depuis une vue, un signal ou un service Django :
|
|
331
|
+
|
|
332
|
+
```python
|
|
333
|
+
from myapp.tasks import notify_search_engines
|
|
334
|
+
|
|
335
|
+
notify_search_engines.delay("https://example.com/articles/nouveau")
|
|
336
|
+
```
|
|
337
|
+
|
|
338
|
+
Pour isoler les retries et quotas de chaque moteur, utilisez idéalement une
|
|
339
|
+
tâche par provider. Le provider Google ne doit être ajouté que pour les pages
|
|
340
|
+
officiellement éligibles.
|
|
341
|
+
|
|
342
|
+
### Job générique
|
|
343
|
+
|
|
344
|
+
Le même principe fonctionne avec un scheduler, un worker maison, RQ, Dramatiq,
|
|
345
|
+
une fonction serverless ou un cron. Le job ne connaît que l'API publique de
|
|
346
|
+
`kliz` :
|
|
347
|
+
|
|
348
|
+
```python
|
|
349
|
+
from kliz import IndexNowProvider, Kliz
|
|
350
|
+
|
|
351
|
+
|
|
352
|
+
class ContentIndexingJob:
|
|
353
|
+
def __init__(self, api_key: str) -> None:
|
|
354
|
+
self.indexer = Kliz([IndexNowProvider(api_key=api_key)])
|
|
355
|
+
|
|
356
|
+
def run(self, payload: dict[str, str]) -> dict[str, bool]:
|
|
357
|
+
return self.indexer.notify_all(payload["url"])
|
|
358
|
+
|
|
359
|
+
|
|
360
|
+
# Le système de jobs choisi sérialise ce payload et appelle job.run(payload).
|
|
361
|
+
job = ContentIndexingJob(api_key="votre-cle")
|
|
362
|
+
result = job.run({"url": "https://example.com/page-modifiee"})
|
|
363
|
+
```
|
|
364
|
+
|
|
365
|
+
## Interface en ligne de commande
|
|
366
|
+
|
|
367
|
+
L'installation fournit aussi une commande `kliz` :
|
|
368
|
+
|
|
369
|
+
```bash
|
|
370
|
+
export KLIZ_INDEXNOW_API_KEY="votre-cle"
|
|
371
|
+
export KLIZ_INDEXNOW_KEY_LOCATION="https://example.com/votre-cle.txt"
|
|
372
|
+
|
|
373
|
+
kliz notify https://example.com/page # une URL
|
|
374
|
+
kliz notify --batch urls.txt # une URL par ligne, `#` pour un commentaire
|
|
375
|
+
kliz providers # liste des providers configurés
|
|
376
|
+
kliz --version
|
|
377
|
+
```
|
|
378
|
+
|
|
379
|
+
Les crédits se passent aussi en options (`--indexnow-api-key`,
|
|
380
|
+
`--indexnow-key-location`, `--google-service-account-file`). Le processus
|
|
381
|
+
termine avec le code `0` si tout a réussi, `1` en cas d'échec de notification
|
|
382
|
+
et `2` en cas de configuration invalide.
|
|
383
|
+
|
|
384
|
+
## Tests
|
|
385
|
+
|
|
386
|
+
Les tests mockent les appels `requests` et le client Google. Ils ne nécessitent
|
|
387
|
+
donc ni accès réseau, ni clé IndexNow, ni compte de service Google.
|
|
388
|
+
|
|
389
|
+
La validation complète locale est :
|
|
390
|
+
|
|
391
|
+
```bash
|
|
392
|
+
ruff format --check src tests
|
|
393
|
+
ruff check src tests
|
|
394
|
+
mypy src
|
|
395
|
+
pytest --cov=kliz
|
|
396
|
+
python -m build
|
|
397
|
+
twine check --strict dist/*
|
|
398
|
+
pip-audit . --strict
|
|
399
|
+
```
|
|
400
|
+
|
|
401
|
+
## Exploitation en production
|
|
402
|
+
|
|
403
|
+
Le package ne stocke aucun secret et n'impose aucun système de tâches. Dans
|
|
404
|
+
l'application qui l'utilise :
|
|
405
|
+
|
|
406
|
+
- injectez les clés par un gestionnaire de secrets ;
|
|
407
|
+
- activez le retry opt-in de `Kliz` (`max_attempts`) ou appliquez un backoff
|
|
408
|
+
applicatif aux résultats `retryable=True` ;
|
|
409
|
+
- placez les échecs définitifs dans une dead-letter queue ;
|
|
410
|
+
- mesurez latence, taux de succès, codes HTTP et quotas par provider ;
|
|
411
|
+
- ne partagez pas une même instance `GoogleProvider` entre plusieurs threads ;
|
|
412
|
+
- conservez un sitemap à jour : une notification ne garantit jamais
|
|
413
|
+
l'indexation.
|
|
414
|
+
|
|
415
|
+
## Publication
|
|
416
|
+
|
|
417
|
+
Les tags `vX.Y.Z` déclenchent le workflow de release. Le tag doit correspondre
|
|
418
|
+
exactement à la version de `pyproject.toml`. La publication utilise le Trusted
|
|
419
|
+
Publishing PyPI et ne nécessite aucun token PyPI permanent dans GitHub.
|
|
420
|
+
|
|
421
|
+
Avant la première release, configurez sur PyPI un publisher avec le dépôt
|
|
422
|
+
`freddychoudja/kliz-`, le workflow `release.yml` et l'environnement `pypi`.
|
|
423
|
+
|
|
424
|
+
## Contribuer
|
|
425
|
+
|
|
426
|
+
Les contributions sont les bienvenues. Consultez
|
|
427
|
+
[CONTRIBUTING.md](CONTRIBUTING.md) avant d'ouvrir une issue ou une pull
|
|
428
|
+
request.
|
|
429
|
+
|
|
430
|
+
Le code source et le suivi du projet sont disponibles sur
|
|
431
|
+
[GitHub](https://github.com/freddychoudja/kliz-).
|
|
432
|
+
|
|
433
|
+
## Licence
|
|
434
|
+
|
|
435
|
+
`kliz` est distribué sous la [licence MIT](LICENSE). Copyright © 2026 Freddy
|
|
436
|
+
Choudja.
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
kliz/__init__.py,sha256=91H-r1VX8Otn-s9NR28Y8fB8Q4tuaPe0oNYNFyN4Wrk,652
|
|
2
|
+
kliz/_http.py,sha256=-qNHQ2YsrCwFeyYrHB60wq4ss9Fp1S-637-m9nB7DqU,1842
|
|
3
|
+
kliz/_validation.py,sha256=NDBpi7Xzv1yZ1y7z6uhnn6CAG7TsbkhsexQYsMqqbJM,1455
|
|
4
|
+
kliz/cli.py,sha256=GFhLhLsTE5iFmy_CBkjqaxCWcwrxVoOpPIPK-0pFlfY,5289
|
|
5
|
+
kliz/core.py,sha256=AQ2aamSnUss5hF_8I7Ok1KbbD8VLGhfCW-WMagYKW1Q,6276
|
|
6
|
+
kliz/exceptions.py,sha256=9AafUFFDVOien0lpBwSLItZAwKHUWrRZ0v2JpRdgHs0,565
|
|
7
|
+
kliz/py.typed,sha256=AbpHGcgLb-kRsJGnwFEktk7uzpZOCcBY74-YBdrKVGs,1
|
|
8
|
+
kliz/results.py,sha256=RcEsm_pVb93mMUUwb9A5WI0S_YCaCeld5ODp06LHNoc,362
|
|
9
|
+
kliz/providers/__init__.py,sha256=v_dYpfOxclltln99eMw4g0IqQ5biiim-5eqeb2hiHa4,313
|
|
10
|
+
kliz/providers/base.py,sha256=3hfRXkGTgwOqKhODM1vSThZCv0AVk58pQZw-xGna2i0,985
|
|
11
|
+
kliz/providers/batch.py,sha256=P5DFtPo9OsZPUEApquYyeQOYDGKPRsi_fvA3zoO9e_M,2539
|
|
12
|
+
kliz/providers/google.py,sha256=gz9rSXceoVcWGrpCf6ZS_9Pcy8FL-OH7yQqBUDyiz-E,3671
|
|
13
|
+
kliz/providers/indexnow.py,sha256=1g7q9FN4RwxbPkNTr1ch8uicOVfzLJxWzs1YrZyihY4,3052
|
|
14
|
+
kliz-0.2.0.dist-info/licenses/LICENSE,sha256=eOlskYaawwtAU6sGhMd7o6xD1iPyjaQUpQY_bMrUCRg,1071
|
|
15
|
+
kliz-0.2.0.dist-info/METADATA,sha256=gK_Cm0pjH36iW0911soOOu9htEVovbRibxHCZWc4yVY,14407
|
|
16
|
+
kliz-0.2.0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
|
|
17
|
+
kliz-0.2.0.dist-info/entry_points.txt,sha256=-69WW759zTomRAmrXJRTWIpaLy1jCqMjRB3jwAPpTi8,39
|
|
18
|
+
kliz-0.2.0.dist-info/top_level.txt,sha256=GZNKut60zID1o86dl8rdWDt-5Yo9KepGFX63ey8983c,5
|
|
19
|
+
kliz-0.2.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Freddy Choudja
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
kliz
|