spn-client 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- spn_client/__init__.py +43 -0
- spn_client/_http.py +22 -0
- spn_client/_redact.py +51 -0
- spn_client/client.py +957 -0
- spn_client-0.1.0.dist-info/METADATA +79 -0
- spn_client-0.1.0.dist-info/RECORD +8 -0
- spn_client-0.1.0.dist-info/WHEEL +4 -0
- spn_client-0.1.0.dist-info/licenses/LICENSE +21 -0
spn_client/__init__.py
ADDED
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
"""spn-client: a hardened client for archive.org's availability and Save Page Now APIs."""
|
|
2
|
+
|
|
3
|
+
from spn_client.client import (
|
|
4
|
+
ARCHIVE_ARCHIVED,
|
|
5
|
+
ARCHIVE_CAPTURE_FAILED,
|
|
6
|
+
ARCHIVE_NOT_ATTEMPTED,
|
|
7
|
+
ARCHIVE_OUTCOME_LABELS,
|
|
8
|
+
ARCHIVE_PENDING,
|
|
9
|
+
ARCHIVE_SUBMIT_FAILED,
|
|
10
|
+
ARCHIVE_SUBMITTED,
|
|
11
|
+
DEFAULT_STALE_DAYS,
|
|
12
|
+
capture_capacity,
|
|
13
|
+
check,
|
|
14
|
+
check_job_status,
|
|
15
|
+
rate_limited_out,
|
|
16
|
+
reset_rate_limit_state,
|
|
17
|
+
service_health_note,
|
|
18
|
+
snapshot_raw_url,
|
|
19
|
+
submit,
|
|
20
|
+
system_status,
|
|
21
|
+
)
|
|
22
|
+
|
|
23
|
+
__version__ = "0.1.0"
|
|
24
|
+
|
|
25
|
+
__all__ = [
|
|
26
|
+
"ARCHIVE_ARCHIVED",
|
|
27
|
+
"ARCHIVE_CAPTURE_FAILED",
|
|
28
|
+
"ARCHIVE_NOT_ATTEMPTED",
|
|
29
|
+
"ARCHIVE_OUTCOME_LABELS",
|
|
30
|
+
"ARCHIVE_PENDING",
|
|
31
|
+
"ARCHIVE_SUBMIT_FAILED",
|
|
32
|
+
"ARCHIVE_SUBMITTED",
|
|
33
|
+
"DEFAULT_STALE_DAYS",
|
|
34
|
+
"capture_capacity",
|
|
35
|
+
"check",
|
|
36
|
+
"check_job_status",
|
|
37
|
+
"rate_limited_out",
|
|
38
|
+
"reset_rate_limit_state",
|
|
39
|
+
"service_health_note",
|
|
40
|
+
"snapshot_raw_url",
|
|
41
|
+
"submit",
|
|
42
|
+
"system_status",
|
|
43
|
+
]
|
spn_client/_http.py
ADDED
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
"""Outbound-HTTP identity for spn-client.
|
|
2
|
+
|
|
3
|
+
A dedicated User-Agent, separate from any consuming project's own — this
|
|
4
|
+
package is meant to be embedded in several unrelated projects, and its
|
|
5
|
+
requests to archive.org should identify the library making them, not whatever
|
|
6
|
+
application happens to import it.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from importlib import metadata
|
|
10
|
+
|
|
11
|
+
try:
|
|
12
|
+
_VERSION = metadata.version("spn-client")
|
|
13
|
+
except metadata.PackageNotFoundError: # pragma: no cover - source/dev checkout
|
|
14
|
+
_VERSION = "0.0.0"
|
|
15
|
+
|
|
16
|
+
USER_AGENT = f"spn-client/{_VERSION}"
|
|
17
|
+
|
|
18
|
+
DEFAULT_HEADERS = {
|
|
19
|
+
"User-Agent": USER_AGENT,
|
|
20
|
+
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
|
|
21
|
+
"Accept-Language": "en-US,en;q=0.9",
|
|
22
|
+
}
|
spn_client/_redact.py
ADDED
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
"""Secret-redaction helpers.
|
|
2
|
+
|
|
3
|
+
Vendored from a sister project's ``ci_core.redact`` module rather than taken
|
|
4
|
+
as a dependency on it — this package is deliberately decoupled from any
|
|
5
|
+
project-specific shared library, and these two functions are the only pieces
|
|
6
|
+
``client.py`` needs.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
import re
|
|
10
|
+
|
|
11
|
+
# The sensitive word a credential parameter ends with. Hyphenated spellings are
|
|
12
|
+
# not hypothetical: Azure OpenAI names its parameter `api-key` and Google sends
|
|
13
|
+
# `X-Goog-Api-Key`. An underscore-only list matched neither, nor
|
|
14
|
+
# `client_secret`, `refresh_token`, `subscription-key`, `password`, `auth` or
|
|
15
|
+
# `sig` — eight of twelve real-world spellings tested on 2026-09-04 passed
|
|
16
|
+
# straight through into whatever log or report the error text reached.
|
|
17
|
+
_SENSITIVE_WORD = (
|
|
18
|
+
r"(?:api[-_]?key|access[-_]?token|refresh[-_]?token|id[-_]?token"
|
|
19
|
+
r"|client[-_]?secret|subscription[-_]?key|session[-_]?key"
|
|
20
|
+
r"|key|token|secret|password|passwd|credentials?|auth|signature|sig)"
|
|
21
|
+
)
|
|
22
|
+
|
|
23
|
+
# Any dash/underscore-separated prefix, then that word, then `=`. Anchoring the
|
|
24
|
+
# end on `=` is what leaves innocent parameters alone: `author=` cannot match,
|
|
25
|
+
# because `auth` would have to consume the whole name and `or` is left over.
|
|
26
|
+
# `keywords=` survives for the same reason.
|
|
27
|
+
_PARAM_NAME = rf"(?:[A-Za-z0-9]+[-_])*{_SENSITIVE_WORD}"
|
|
28
|
+
|
|
29
|
+
# Matches ?key=..., &apiKey=..., &api-key=... in a URL and replaces the value.
|
|
30
|
+
# Stops at the next &, whitespace, quote, or closing bracket so we don't eat
|
|
31
|
+
# the rest of the message.
|
|
32
|
+
_KEY_QUERY_RE = re.compile(
|
|
33
|
+
rf"([?&]{_PARAM_NAME}=)[^&\s'\"<>)\]]+",
|
|
34
|
+
re.IGNORECASE,
|
|
35
|
+
)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def redact_url_keys(text):
|
|
39
|
+
"""Replace key/token query-parameter values in any string with [REDACTED].
|
|
40
|
+
|
|
41
|
+
Provider-agnostic: works even when the key value isn't available to compare
|
|
42
|
+
against, because it matches on the parameter name.
|
|
43
|
+
"""
|
|
44
|
+
return _KEY_QUERY_RE.sub(r"\1[REDACTED]", str(text))
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def redact_value(text, secret):
|
|
48
|
+
"""Replace a known secret value with [REDACTED] wherever it appears."""
|
|
49
|
+
if secret and secret in str(text):
|
|
50
|
+
return str(text).replace(secret, "[REDACTED]")
|
|
51
|
+
return str(text)
|
spn_client/client.py
ADDED
|
@@ -0,0 +1,957 @@
|
|
|
1
|
+
"""A hardened client for archive.org's availability and Save Page Now (SPN2) APIs.
|
|
2
|
+
|
|
3
|
+
Extracted from a citation-verification pipeline where this logic was built and
|
|
4
|
+
incident-tested against real archive.org behavior — process-wide rate pacing
|
|
5
|
+
with a circuit breaker, dual-mode (anonymous + S3-key-authenticated) capture
|
|
6
|
+
submission with async job-status polling, and an outcome vocabulary that
|
|
7
|
+
distinguishes "we asked archive.org" from "archive.org confirmed it archived
|
|
8
|
+
this." Every non-obvious behavior below is a direct response to a dated,
|
|
9
|
+
measured incident, not a speculative design.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
import logging
|
|
13
|
+
import threading
|
|
14
|
+
import time
|
|
15
|
+
import re
|
|
16
|
+
from datetime import datetime, timezone
|
|
17
|
+
|
|
18
|
+
import requests
|
|
19
|
+
|
|
20
|
+
from spn_client._http import DEFAULT_HEADERS
|
|
21
|
+
from spn_client import _redact as redact
|
|
22
|
+
|
|
23
|
+
log = logging.getLogger(__name__)
|
|
24
|
+
|
|
25
|
+
_AVAILABILITY_API = "https://archive.org/wayback/available"
|
|
26
|
+
_SAVE_API = "https://web.archive.org/save"
|
|
27
|
+
_SAVE_STATUS_API = "https://web.archive.org/save/status"
|
|
28
|
+
_SAVE_USER_STATUS_API = "https://web.archive.org/save/status/user"
|
|
29
|
+
_SAVE_SYSTEM_STATUS_API = "https://web.archive.org/save/status/system"
|
|
30
|
+
|
|
31
|
+
#: Fallback staleness threshold (days) used only when a caller does not pass
|
|
32
|
+
#: its own ``stale_days``. Callers with an actual staleness policy should
|
|
33
|
+
#: always pass one explicitly — this exists so ``check()``/``submit()``/
|
|
34
|
+
#: ``check_job_status()`` have *a* number to compare against rather than
|
|
35
|
+
#: raising when none is given.
|
|
36
|
+
DEFAULT_STALE_DAYS = 30
|
|
37
|
+
|
|
38
|
+
# Matches a Wayback Machine snapshot URL and captures its embedded timestamp:
|
|
39
|
+
# https://web.archive.org/web/20250101000000/https://example.com/...
|
|
40
|
+
_ARCHIVE_URL_RE = re.compile(
|
|
41
|
+
r"https?://web\.archive\.org/web/(\d{4,14})(?:[a-z_]*)?/", re.IGNORECASE
|
|
42
|
+
)
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
#: What a run actually established about archiving a URL, as opposed to what
|
|
46
|
+
#: it *asked* for. The distinction is the whole point of this vocabulary: it
|
|
47
|
+
#: is tempting to record only ``submitted: True`` and report "submitted for
|
|
48
|
+
#: archiving", which a reader reasonably finishes as "so it is archived".
|
|
49
|
+
#: Nothing establishes that on its own. A capture archive.org silently
|
|
50
|
+
#: dropped looks exactly like one it completed.
|
|
51
|
+
#:
|
|
52
|
+
#: ``ARCHIVED`` is the only value that asserts a snapshot exists, and it is
|
|
53
|
+
#: only ever set alongside a ``snapshot_url`` that came back from archive.org.
|
|
54
|
+
ARCHIVE_SUBMITTED = "submitted" # accepted; outcome not established (see below)
|
|
55
|
+
ARCHIVE_ARCHIVED = "archived" # a snapshot URL came back — this one is real
|
|
56
|
+
ARCHIVE_PENDING = "pending" # job accepted, capture still running at report time
|
|
57
|
+
ARCHIVE_CAPTURE_FAILED = "capture_failed" # accepted, then the capture failed
|
|
58
|
+
ARCHIVE_SUBMIT_FAILED = "submit_failed" # archive.org refused the request itself
|
|
59
|
+
ARCHIVE_NOT_ATTEMPTED = "not_attempted" # we never asked; the reason is recorded
|
|
60
|
+
|
|
61
|
+
#: Human-readable phrasing for each outcome, for report output. Kept here so
|
|
62
|
+
#: two different renderers cannot drift into describing the same state two
|
|
63
|
+
#: different ways.
|
|
64
|
+
ARCHIVE_OUTCOME_LABELS = {
|
|
65
|
+
ARCHIVE_SUBMITTED: "submitted; capture outcome not established",
|
|
66
|
+
ARCHIVE_ARCHIVED: "archived",
|
|
67
|
+
ARCHIVE_PENDING: "submitted; capture still pending",
|
|
68
|
+
ARCHIVE_CAPTURE_FAILED: "capture failed",
|
|
69
|
+
ARCHIVE_SUBMIT_FAILED: "submission failed",
|
|
70
|
+
ARCHIVE_NOT_ATTEMPTED: "not submitted",
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def _age_days_from_timestamp(ts):
|
|
75
|
+
"""Days since a Wayback timestamp (YYYYMMDD...), or None if unparseable."""
|
|
76
|
+
if len(ts) >= 8:
|
|
77
|
+
try:
|
|
78
|
+
snap_dt = datetime.strptime(ts[:8], "%Y%m%d").replace(tzinfo=timezone.utc)
|
|
79
|
+
return (datetime.now(timezone.utc) - snap_dt).days
|
|
80
|
+
except ValueError:
|
|
81
|
+
pass
|
|
82
|
+
return None
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
# archive.org throttles the availability API at IP level with a long window, not
|
|
86
|
+
# per-second. Measured 2026-08-12: 12 consecutive requests all returned 429 —
|
|
87
|
+
# including the first — and it was still 429 after a 45s cooldown at one request
|
|
88
|
+
# every 6 seconds. A pipeline with no pacing and no backoff at all ended up with
|
|
89
|
+
# an archive thread that had been dead for at least two runs: 49 resolved items,
|
|
90
|
+
# 0 archived, ~49 rate-limited.
|
|
91
|
+
#
|
|
92
|
+
# (Re-probed 2026-08-15: six back-to-back lookups all returned 200. The throttle
|
|
93
|
+
# is episodic, so the guard has to be always-on rather than tuned to one window.)
|
|
94
|
+
#
|
|
95
|
+
# Two mechanisms, because one is not enough:
|
|
96
|
+
# * a process-wide minimum interval between calls, serialised on a lock. If
|
|
97
|
+
# callers run this from a thread pool, without this the pool's width
|
|
98
|
+
# decides the request rate.
|
|
99
|
+
# * retry with backoff that honours Retry-After, and a circuit breaker: once
|
|
100
|
+
# archive.org has said 429 repeatedly, further calls in the same run are
|
|
101
|
+
# skipped rather than spending the run's time collecting more 429s.
|
|
102
|
+
#
|
|
103
|
+
# Every piece of state below is process-wide, and that is the whole design
|
|
104
|
+
# constraint: ``check()`` may be called from several worker threads at once.
|
|
105
|
+
# Two consequences the first version of this code got wrong, both of which
|
|
106
|
+
# silently disabled the protection they were meant to provide:
|
|
107
|
+
#
|
|
108
|
+
# * **Per-thread backoff is not backoff.** Sleeping inside the failing thread
|
|
109
|
+
# leaves the other workers hammering archive.org at the full pace. A 429 has
|
|
110
|
+
# to move a clock every thread waits on, so the *process* slows down.
|
|
111
|
+
# * **"Consecutive" is not well defined across interleaved threads.** Resetting
|
|
112
|
+
# a shared counter on success let one worker's 200 erase four other workers'
|
|
113
|
+
# refusals, so a run being throttled 4-in-5 would never trip the breaker —
|
|
114
|
+
# precisely the case it exists for. The count is now a per-run budget that
|
|
115
|
+
# only moves up.
|
|
116
|
+
_MIN_INTERVAL_SECONDS = 3.0
|
|
117
|
+
_MAX_ATTEMPTS = 3
|
|
118
|
+
_BACKOFF_BASE_SECONDS = 5.0
|
|
119
|
+
|
|
120
|
+
#: Refused *lookups* tolerated in a run before we stop asking. Counted once per
|
|
121
|
+
#: lookup, not once per attempt, so the number means what it says: the
|
|
122
|
+
#: ``_MAX_ATTEMPTS`` retries inside one lookup are a single refusal. (Counting
|
|
123
|
+
#: attempts made this trip after two lookups rather than five.)
|
|
124
|
+
_CIRCUIT_TRIP_AFTER = 5
|
|
125
|
+
|
|
126
|
+
#: Queue depth at which archive.org counts as busy rather than merely up.
|
|
127
|
+
#: Measured healthy 2026-09-06 with every one of its thirteen capture queues
|
|
128
|
+
#: at zero, so any sustained backlog is worth telling the caller about — it
|
|
129
|
+
#: is the difference between "your submission failed" and "everyone's did".
|
|
130
|
+
_BUSY_QUEUE_DEPTH = 50
|
|
131
|
+
|
|
132
|
+
_pace_lock = threading.Lock()
|
|
133
|
+
_last_call_at = 0.0
|
|
134
|
+
|
|
135
|
+
#: Absolute ``time.monotonic()`` before which no thread may call. A 429 pushes
|
|
136
|
+
#: this forward, which is what makes the backoff process-wide.
|
|
137
|
+
_blocked_until = 0.0
|
|
138
|
+
|
|
139
|
+
#: Lookups this run that exhausted every attempt against a 429. Never reset on
|
|
140
|
+
#: success — see the note above on why "consecutive" cannot work here.
|
|
141
|
+
_rate_limited_lookups = 0
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
def reset_rate_limit_state():
|
|
145
|
+
"""Clear the pacing clock, the backoff, and the circuit breaker.
|
|
146
|
+
|
|
147
|
+
Call this at the start of every run. The state is process-wide, so without
|
|
148
|
+
a reset a breaker tripped by one run would skip every archive.org call in
|
|
149
|
+
the next run inside the same process — a silently degraded run whose
|
|
150
|
+
submissions all report "skipped" for a limit that expired long ago.
|
|
151
|
+
"""
|
|
152
|
+
global _last_call_at, _blocked_until, _rate_limited_lookups
|
|
153
|
+
with _pace_lock:
|
|
154
|
+
_last_call_at = 0.0
|
|
155
|
+
_blocked_until = 0.0
|
|
156
|
+
_rate_limited_lookups = 0
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def rate_limited_out():
|
|
160
|
+
"""True once the circuit breaker has tripped for this run."""
|
|
161
|
+
with _pace_lock:
|
|
162
|
+
return _rate_limited_lookups >= _CIRCUIT_TRIP_AFTER
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def _pace():
|
|
166
|
+
"""Block until the shared clock allows another call.
|
|
167
|
+
|
|
168
|
+
Waits for whichever is later: ``_MIN_INTERVAL_SECONDS`` since the last call,
|
|
169
|
+
or the end of a backoff that a 429 imposed on every thread. Sleeping while
|
|
170
|
+
holding the lock is deliberate — it is exactly what makes the interval
|
|
171
|
+
process-wide instead of per-thread.
|
|
172
|
+
"""
|
|
173
|
+
global _last_call_at
|
|
174
|
+
with _pace_lock:
|
|
175
|
+
now = time.monotonic()
|
|
176
|
+
target = max(_last_call_at + _MIN_INTERVAL_SECONDS, _blocked_until)
|
|
177
|
+
if target > now:
|
|
178
|
+
time.sleep(target - now)
|
|
179
|
+
_last_call_at = time.monotonic()
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
def _note_rate_limited(retry_after):
|
|
183
|
+
"""Back every thread off after a 429, not just the one that hit it."""
|
|
184
|
+
global _blocked_until
|
|
185
|
+
with _pace_lock:
|
|
186
|
+
_blocked_until = max(_blocked_until, time.monotonic() + retry_after)
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def _note_lookup_refused():
|
|
190
|
+
"""Count one lookup that never got past a 429."""
|
|
191
|
+
global _rate_limited_lookups
|
|
192
|
+
with _pace_lock:
|
|
193
|
+
_rate_limited_lookups += 1
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def _retry_after_seconds(resp, attempt):
|
|
197
|
+
"""Seconds to wait before retrying, preferring the server's own answer."""
|
|
198
|
+
header = (resp.headers or {}).get("Retry-After") if resp is not None else None
|
|
199
|
+
if header:
|
|
200
|
+
try:
|
|
201
|
+
return min(float(header), 60.0)
|
|
202
|
+
except (TypeError, ValueError):
|
|
203
|
+
pass
|
|
204
|
+
return _BACKOFF_BASE_SECONDS * (2**attempt)
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
def _paced_get(endpoint, timeout, params=None, headers=None):
|
|
208
|
+
"""GET an archive.org endpoint with pacing, backoff, and a circuit breaker.
|
|
209
|
+
|
|
210
|
+
Raises the last exception if every attempt fails, so the caller's existing
|
|
211
|
+
error handling is unchanged.
|
|
212
|
+
|
|
213
|
+
There is no local sleep between attempts: a 429 pushes the shared clock out
|
|
214
|
+
and the wait then happens in the next ``_pace()``, which every thread goes
|
|
215
|
+
through. That is the difference between the process backing off and one
|
|
216
|
+
thread backing off while seven others keep the pressure on.
|
|
217
|
+
|
|
218
|
+
Endpoint-agnostic on purpose. The availability API and the Save Page Now
|
|
219
|
+
job-status API are the same host, the same IP-level throttle, and the same
|
|
220
|
+
consequence for ignoring it, so they share one pacing clock and one breaker
|
|
221
|
+
budget rather than each getting its own. Two schemes would each be pacing
|
|
222
|
+
against half the real request rate, which is how you end up rate-limited by
|
|
223
|
+
a system that believes it is being polite.
|
|
224
|
+
"""
|
|
225
|
+
last_exc = None
|
|
226
|
+
rate_limited = False
|
|
227
|
+
for attempt in range(_MAX_ATTEMPTS):
|
|
228
|
+
_pace()
|
|
229
|
+
try:
|
|
230
|
+
resp = requests.get(
|
|
231
|
+
endpoint,
|
|
232
|
+
params=params,
|
|
233
|
+
timeout=timeout,
|
|
234
|
+
headers=headers or DEFAULT_HEADERS,
|
|
235
|
+
)
|
|
236
|
+
except Exception as exc:
|
|
237
|
+
last_exc = exc
|
|
238
|
+
continue
|
|
239
|
+
if resp.status_code == 429:
|
|
240
|
+
rate_limited = True
|
|
241
|
+
last_exc = requests.HTTPError(
|
|
242
|
+
f"429 Client Error: Too Many Requests for url: {resp.url}",
|
|
243
|
+
response=resp,
|
|
244
|
+
)
|
|
245
|
+
_note_rate_limited(_retry_after_seconds(resp, attempt))
|
|
246
|
+
continue
|
|
247
|
+
resp.raise_for_status()
|
|
248
|
+
return resp
|
|
249
|
+
# One refused lookup, however many attempts it took to establish that.
|
|
250
|
+
if rate_limited:
|
|
251
|
+
_note_lookup_refused()
|
|
252
|
+
raise last_exc
|
|
253
|
+
|
|
254
|
+
|
|
255
|
+
def _get_availability(url, timeout):
|
|
256
|
+
"""GET the availability API through the shared pacing/backoff/breaker path."""
|
|
257
|
+
return _paced_get(_AVAILABILITY_API, timeout, params={"url": url})
|
|
258
|
+
|
|
259
|
+
|
|
260
|
+
def _transport_failure_summary(exc, what):
|
|
261
|
+
"""One reader-facing sentence for an archive.org call that did not complete.
|
|
262
|
+
|
|
263
|
+
The raw exception is for the log, not for whoever reads a report built on
|
|
264
|
+
this. Left unfiltered it reaches the caller as
|
|
265
|
+
``HTTPSConnectionPool(host='web.archive.org', port=443): Max retries
|
|
266
|
+
exceeded with url: /save/status/... (Caused by NewConnectionError(...
|
|
267
|
+
[WinError 10061] ...))`` — observed verbatim in a real run 2026-09-06. That
|
|
268
|
+
is a debugger's string in a message meant for someone deciding what to do
|
|
269
|
+
next, and it invites them to debug the networking instead of telling them
|
|
270
|
+
what it means for their submission.
|
|
271
|
+
|
|
272
|
+
Callers keep the raw text under a separate key so nothing is lost.
|
|
273
|
+
"""
|
|
274
|
+
if isinstance(exc, requests.exceptions.Timeout):
|
|
275
|
+
return f"archive.org did not answer {what} within the timeout"
|
|
276
|
+
if isinstance(exc, requests.exceptions.ConnectionError):
|
|
277
|
+
return (
|
|
278
|
+
f"could not reach archive.org {what} — the connection was refused, "
|
|
279
|
+
f"dropped, or the host did not resolve"
|
|
280
|
+
)
|
|
281
|
+
status = getattr(getattr(exc, "response", None), "status_code", None)
|
|
282
|
+
if status:
|
|
283
|
+
return f"archive.org answered HTTP {status} {what}"
|
|
284
|
+
return f"the request to archive.org {what} failed"
|
|
285
|
+
|
|
286
|
+
|
|
287
|
+
def snapshot_raw_url(snapshot_url):
|
|
288
|
+
"""The ``id_`` form of a snapshot URL: the original captured bytes.
|
|
289
|
+
|
|
290
|
+
``https://web.archive.org/web/<ts>/<url>`` serves the capture with
|
|
291
|
+
archive.org's own banner and its ``wombat.js`` URL-rewriting shim injected.
|
|
292
|
+
Appending ``id_`` to the timestamp — ``/web/<ts>id_/<url>`` — serves what was
|
|
293
|
+
actually captured, unmodified.
|
|
294
|
+
|
|
295
|
+
That difference is what makes an archived copy checkable against the live
|
|
296
|
+
page at all. Measured 2026-09-06 across three URLs, one of them archived a
|
|
297
|
+
week earlier: the ``id_`` body was **byte-identical** to the live page
|
|
298
|
+
(SHA-256 equal, 6639/10923/9569 bytes), while the ordinary form was nearly
|
|
299
|
+
three times the size for the same document. Comparing the injected form
|
|
300
|
+
would be comparing archive.org's chrome, and would report every page as
|
|
301
|
+
divergent from itself.
|
|
302
|
+
|
|
303
|
+
Returns ``None`` if ``snapshot_url`` is not a snapshot URL.
|
|
304
|
+
"""
|
|
305
|
+
if not snapshot_url:
|
|
306
|
+
return None
|
|
307
|
+
m = _ARCHIVE_URL_RE.search(snapshot_url)
|
|
308
|
+
if not m:
|
|
309
|
+
return None
|
|
310
|
+
# Replace the matched "/web/<ts><flags>/" with "/web/<ts>id_/".
|
|
311
|
+
return (
|
|
312
|
+
snapshot_url[: m.start()]
|
|
313
|
+
+ f"https://web.archive.org/web/{m.group(1)}id_/"
|
|
314
|
+
+ snapshot_url[m.end() :]
|
|
315
|
+
)
|
|
316
|
+
|
|
317
|
+
|
|
318
|
+
def _snapshot_state(snapshot_url, ts, stale_days=None):
|
|
319
|
+
"""The four snapshot fields every caller reports, from a URL and timestamp.
|
|
320
|
+
|
|
321
|
+
One implementation because ``check()``, ``submit()`` and
|
|
322
|
+
``check_job_status()`` all have to answer "how old is this snapshot, and
|
|
323
|
+
is that too old" the same way. They previously could disagree because only
|
|
324
|
+
one of them ever answered it; now that a submission can come back with a
|
|
325
|
+
real snapshot, they share the arithmetic instead of risking that.
|
|
326
|
+
"""
|
|
327
|
+
threshold = stale_days if stale_days is not None else DEFAULT_STALE_DAYS
|
|
328
|
+
age = _age_days_from_timestamp(ts) if ts else None
|
|
329
|
+
return {
|
|
330
|
+
"snapshot_url": snapshot_url,
|
|
331
|
+
"snapshot_ts": ts,
|
|
332
|
+
"snapshot_age_days": age,
|
|
333
|
+
"snapshot_stale": age is not None and age > threshold,
|
|
334
|
+
}
|
|
335
|
+
|
|
336
|
+
|
|
337
|
+
def check(url, timeout=10, stale_days=None):
|
|
338
|
+
"""Check if a URL has a Wayback Machine snapshot and how fresh it is.
|
|
339
|
+
|
|
340
|
+
If ``url`` is itself a Wayback Machine snapshot link, it is recognized as
|
|
341
|
+
already-archived: the snapshot date is read from the URL's embedded
|
|
342
|
+
timestamp (no archive-of-an-archive lookup), and ``is_archive_url`` is set.
|
|
343
|
+
Whether that archive link actually resolves is a separate question this
|
|
344
|
+
function does not answer.
|
|
345
|
+
|
|
346
|
+
Returns a dict with:
|
|
347
|
+
archived bool | None — True/False, or None on network error
|
|
348
|
+
is_archive_url bool — True when the link itself is a web.archive.org URL
|
|
349
|
+
snapshot_url str — direct https://web.archive.org/web/... URL
|
|
350
|
+
snapshot_ts str — raw Wayback timestamp (YYYYMMDDHHMMSS)
|
|
351
|
+
snapshot_age_days int — days since the snapshot was taken
|
|
352
|
+
snapshot_stale bool — True when older than the stale threshold
|
|
353
|
+
error str — set only on network/parse failure
|
|
354
|
+
"""
|
|
355
|
+
# The link is already a Wayback snapshot — read its date from the URL itself.
|
|
356
|
+
m = _ARCHIVE_URL_RE.search(url)
|
|
357
|
+
if m:
|
|
358
|
+
result = {"url": url, "archived": True, "is_archive_url": True}
|
|
359
|
+
result.update(_snapshot_state(url, m.group(1), stale_days))
|
|
360
|
+
return result
|
|
361
|
+
|
|
362
|
+
# Once archive.org has refused repeatedly, stop asking: every further call
|
|
363
|
+
# costs the run several seconds of pacing to collect another 429.
|
|
364
|
+
if rate_limited_out():
|
|
365
|
+
return {
|
|
366
|
+
"url": url,
|
|
367
|
+
"archived": None,
|
|
368
|
+
"error": (
|
|
369
|
+
"skipped: archive.org rate limit tripped earlier this run "
|
|
370
|
+
"(no snapshot lookup attempted)"
|
|
371
|
+
),
|
|
372
|
+
}
|
|
373
|
+
try:
|
|
374
|
+
resp = _get_availability(url, timeout)
|
|
375
|
+
except Exception as exc:
|
|
376
|
+
log.debug("Wayback availability check failed for %s: %s", url, exc)
|
|
377
|
+
return {"url": url, "archived": None, "error": str(exc)}
|
|
378
|
+
|
|
379
|
+
try:
|
|
380
|
+
data = resp.json()
|
|
381
|
+
except ValueError as exc:
|
|
382
|
+
# A 200 that isn't JSON means archive.org served something other than an
|
|
383
|
+
# availability answer — an error or challenge page. Reporting the raw
|
|
384
|
+
# decoder message ("Expecting value: line 1 column 1") sends the reader
|
|
385
|
+
# hunting for a parser bug that isn't there; it is the same misdirection
|
|
386
|
+
# upstream reported as akamhy/waybackpy#200, where a throttled lookup
|
|
387
|
+
# surfaces as invalid JSON. Say what actually arrived instead.
|
|
388
|
+
log.debug("Wayback availability returned non-JSON for %s: %s", url, exc)
|
|
389
|
+
return {
|
|
390
|
+
"url": url,
|
|
391
|
+
"archived": None,
|
|
392
|
+
"error": (
|
|
393
|
+
f"archive.org returned a non-JSON {resp.status_code} response "
|
|
394
|
+
f"({len(resp.content)} bytes) from the availability API"
|
|
395
|
+
),
|
|
396
|
+
}
|
|
397
|
+
|
|
398
|
+
closest = data.get("archived_snapshots", {}).get("closest", {})
|
|
399
|
+
if not closest.get("available"):
|
|
400
|
+
return {"url": url, "archived": False}
|
|
401
|
+
|
|
402
|
+
result = {"url": url, "archived": True}
|
|
403
|
+
# The availability API reports the captured response's own HTTP status, and
|
|
404
|
+
# it's tempting to discard it. It matters: a snapshot can be a capture of a
|
|
405
|
+
# 403 block page or a 404, and "a snapshot exists" then means the opposite
|
|
406
|
+
# of what a caller might assume. A capture made through ``submit()`` here
|
|
407
|
+
# always sends ``capture_all=0`` so it never makes one, but a *pre-existing*
|
|
408
|
+
# snapshot is outside this package's control.
|
|
409
|
+
status = closest.get("status")
|
|
410
|
+
if status:
|
|
411
|
+
result["snapshot_status"] = str(status)
|
|
412
|
+
result["snapshot_is_error_capture"] = not str(status).startswith("2")
|
|
413
|
+
result.update(
|
|
414
|
+
_snapshot_state(
|
|
415
|
+
closest.get("url", ""), closest.get("timestamp", ""), stale_days
|
|
416
|
+
)
|
|
417
|
+
)
|
|
418
|
+
return result
|
|
419
|
+
|
|
420
|
+
|
|
421
|
+
#: Save Page Now capture options this client sets deliberately.
|
|
422
|
+
#:
|
|
423
|
+
#: Read off the live ``/save`` form 2026-09-06 — its control names *are* the API
|
|
424
|
+
#: parameter names: ``capture_outlinks``, ``capture_all``, ``capture_screenshot``,
|
|
425
|
+
#: ``disable_adblocker``, ``wm-save-mywebarchive``, ``email_result``, ``wacz``.
|
|
426
|
+
#: All of them work on the authenticated endpoint. Only one is set, and it's a
|
|
427
|
+
#: correctness fix rather than a feature:
|
|
428
|
+
#:
|
|
429
|
+
#: ``capture_all=0`` — **the form defaults this ON**, and on means "archive the
|
|
430
|
+
#: page even if it answers 4xx/5xx". For most durability use cases that is a
|
|
431
|
+
#: way to manufacture a lie: a source that blocks the capture with a 403
|
|
432
|
+
#: would get its block page archived, a later ``check()`` would find a
|
|
433
|
+
#: snapshot, and a caller would report the URL as archived when what's
|
|
434
|
+
#: archived is an error page. It bites hardest on exactly the sources that
|
|
435
|
+
#: refuse automated requests — the ones most needing an archive. A capture
|
|
436
|
+
#: that fails should be *reported* as failed, which is what this module does.
|
|
437
|
+
#:
|
|
438
|
+
#: Deliberately NOT set, and why:
|
|
439
|
+
#: ``email_result`` / ``wacz`` — archive.org emails the account owner, once
|
|
440
|
+
#: per capture. A run submits many URLs; enabling either turns a batch job
|
|
441
|
+
#: into an inbox full of mail nobody asked for.
|
|
442
|
+
#: ``wm-save-mywebarchive`` — writes to the operator's personal archive. Their
|
|
443
|
+
#: account, their choice, not a side effect of calling this library.
|
|
444
|
+
#: ``capture_outlinks`` — every outlink of every submitted URL is an enormous
|
|
445
|
+
#: load increase on a service that already throttles callers, for pages
|
|
446
|
+
#: nothing asked to capture.
|
|
447
|
+
#: ``capture_screenshot`` — a second form of evidence, and a real option worth
|
|
448
|
+
#: revisiting, but this package renders nothing from it today.
|
|
449
|
+
#: ``force_get`` — trades the headless browser for a plain GET, which captures
|
|
450
|
+
#: JavaScript-rendered pages worse. The point is a faithful copy.
|
|
451
|
+
_CAPTURE_OPTIONS = {"capture_all": "0"}
|
|
452
|
+
|
|
453
|
+
|
|
454
|
+
def _capture_params(url, stale_days=None):
|
|
455
|
+
"""Form fields for one authenticated capture request.
|
|
456
|
+
|
|
457
|
+
``if_not_archived_within`` is deliberately NOT sent, and it is worth saying
|
|
458
|
+
why because it looks like an obvious fit. It asks archive.org to skip the
|
|
459
|
+
capture when a snapshot newer than N already exists — seemingly the same
|
|
460
|
+
idea as a caller's own staleness policy.
|
|
461
|
+
|
|
462
|
+
Tried against the live API 2026-09-06. Sending ``if_not_archived_within``
|
|
463
|
+
for a page with a recent snapshot returns::
|
|
464
|
+
|
|
465
|
+
{"url": "...", "job_id": null,
|
|
466
|
+
"message": "The same snapshot had been made 177 hours, 9 minutes ago.
|
|
467
|
+
You can make new capture of this URL after 4320 hours."}
|
|
468
|
+
|
|
469
|
+
while the identical request without it captures normally. So the skip is
|
|
470
|
+
caused entirely by the parameter — archive.org imposes no such restriction
|
|
471
|
+
of its own — and it leaves the caller with no job id and no snapshot URL,
|
|
472
|
+
i.e. less information than before.
|
|
473
|
+
|
|
474
|
+
More importantly it is a *second gate on the same decision*. This module
|
|
475
|
+
only submits a URL when ``check()`` already said it has no snapshot or a
|
|
476
|
+
stale one; asking archive.org to independently re-decide that can only
|
|
477
|
+
produce disagreement between two rules meant to answer one question.
|
|
478
|
+
|
|
479
|
+
``stale_days`` is accepted so callers need not care which options apply.
|
|
480
|
+
"""
|
|
481
|
+
return {"url": url, **_CAPTURE_OPTIONS}
|
|
482
|
+
|
|
483
|
+
|
|
484
|
+
def submit(url, timeout=30, access_key=None, secret_key=None, stale_days=None):
|
|
485
|
+
"""Request that archive.org capture and archive ``url`` (Save Page Now / SPN2).
|
|
486
|
+
|
|
487
|
+
Returns what was *established*, not merely what was asked for. The two
|
|
488
|
+
paths establish different amounts, and the difference is not a detail:
|
|
489
|
+
|
|
490
|
+
**Unauthenticated** (``GET /save/<url>``, no credentials): archive.org runs
|
|
491
|
+
the capture inline and answers with a ``302`` to the resulting snapshot.
|
|
492
|
+
Measured 2026-09-05 against a real URL: one redirect hop to a snapshot with
|
|
493
|
+
a timestamp minted seconds earlier. So on this path the snapshot URL is in
|
|
494
|
+
hand *before this function returns* — no job id, no polling, nothing to
|
|
495
|
+
wait for. A naive implementation that follows the redirect and discards
|
|
496
|
+
the final URL, reporting only ``submitted: True``, throws away an answer
|
|
497
|
+
archive.org has already given.
|
|
498
|
+
|
|
499
|
+
**Authenticated** (``POST /save`` with an S3-style key pair from
|
|
500
|
+
https://archive.org/account/s3.php): the capture is queued and the response
|
|
501
|
+
carries a ``job_id`` instead of a snapshot. That id is the only handle on
|
|
502
|
+
the outcome, and ``check_job_status`` is the only thing that can read it.
|
|
503
|
+
|
|
504
|
+
Either way this does not block waiting for a capture — callers decide who
|
|
505
|
+
waits, how long, and why (see ``check_job_status``).
|
|
506
|
+
|
|
507
|
+
Returns a dict with:
|
|
508
|
+
url str
|
|
509
|
+
submitted bool — archive.org accepted the request
|
|
510
|
+
job_id str | None — SPN2 job id (authenticated submission only)
|
|
511
|
+
archived bool — set True only when a snapshot URL came back
|
|
512
|
+
snapshot_url str — present only when ``archived``
|
|
513
|
+
snapshot_ts str — raw Wayback timestamp (YYYYMMDDHHMMSS)
|
|
514
|
+
snapshot_age_days int | None
|
|
515
|
+
snapshot_stale bool — a redirect can land on a pre-existing
|
|
516
|
+
snapshot rather than a fresh capture, so
|
|
517
|
+
freshness is measured, never assumed
|
|
518
|
+
outcome_unknown bool — set with ``error`` when the request went
|
|
519
|
+
out and we stopped listening before an
|
|
520
|
+
answer came back. Not the same fact as a
|
|
521
|
+
refusal: archive.org may well have run the
|
|
522
|
+
capture anyway, so this must not be
|
|
523
|
+
reported as a failed submission.
|
|
524
|
+
error str — set only on failure
|
|
525
|
+
"""
|
|
526
|
+
# The breaker exists because archive.org throttles per IP across endpoints.
|
|
527
|
+
# A naive submission path ignores it: once five lookups had been refused,
|
|
528
|
+
# the availability API goes quiet while Save Page Now — the *more*
|
|
529
|
+
# expensive call, since it starts a real capture — keeps firing at full
|
|
530
|
+
# pace. Honouring it here is reuse of the one scheme, not a second one.
|
|
531
|
+
if rate_limited_out():
|
|
532
|
+
return {
|
|
533
|
+
"url": url,
|
|
534
|
+
"submitted": False,
|
|
535
|
+
"job_id": None,
|
|
536
|
+
"archived": False,
|
|
537
|
+
"error": (
|
|
538
|
+
"skipped: archive.org rate limit tripped earlier this run "
|
|
539
|
+
"(no capture requested)"
|
|
540
|
+
),
|
|
541
|
+
"error_summary": (
|
|
542
|
+
"not requested — archive.org had already rate-limited this run"
|
|
543
|
+
),
|
|
544
|
+
"rate_limited": True,
|
|
545
|
+
}
|
|
546
|
+
|
|
547
|
+
headers = dict(DEFAULT_HEADERS)
|
|
548
|
+
try:
|
|
549
|
+
if access_key and secret_key:
|
|
550
|
+
headers["Authorization"] = f"LOW {access_key}:{secret_key}"
|
|
551
|
+
headers["Accept"] = "application/json"
|
|
552
|
+
resp = requests.post(
|
|
553
|
+
_SAVE_API,
|
|
554
|
+
data=_capture_params(url, stale_days),
|
|
555
|
+
headers=headers,
|
|
556
|
+
timeout=timeout,
|
|
557
|
+
)
|
|
558
|
+
resp.raise_for_status()
|
|
559
|
+
payload = resp.json()
|
|
560
|
+
result = {
|
|
561
|
+
"url": url,
|
|
562
|
+
"submitted": True,
|
|
563
|
+
"job_id": payload.get("job_id"),
|
|
564
|
+
"archived": False,
|
|
565
|
+
}
|
|
566
|
+
if not result["job_id"]:
|
|
567
|
+
# Accepted, but no capture started. archive.org explains itself
|
|
568
|
+
# in ``message`` (e.g. "The same snapshot had been made 177
|
|
569
|
+
# hours ago"). Carrying that through is the difference between
|
|
570
|
+
# the caller knowing why nothing happened and knowing nothing.
|
|
571
|
+
message = str(payload.get("message") or "").strip()
|
|
572
|
+
result["error_summary"] = (
|
|
573
|
+
f"archive.org accepted the request without starting a "
|
|
574
|
+
f"capture: {message}"
|
|
575
|
+
if message
|
|
576
|
+
else "archive.org accepted the request but started no capture"
|
|
577
|
+
)
|
|
578
|
+
return result
|
|
579
|
+
|
|
580
|
+
resp = requests.get(
|
|
581
|
+
f"{_SAVE_API}/{url}",
|
|
582
|
+
headers=headers,
|
|
583
|
+
timeout=timeout,
|
|
584
|
+
)
|
|
585
|
+
resp.raise_for_status()
|
|
586
|
+
result = {"url": url, "submitted": True, "job_id": None, "archived": False}
|
|
587
|
+
# The capture archive.org just ran, named by the URL it redirected us to.
|
|
588
|
+
m = _ARCHIVE_URL_RE.search(resp.url or "")
|
|
589
|
+
if m:
|
|
590
|
+
result["archived"] = True
|
|
591
|
+
result.update(_snapshot_state(resp.url, m.group(1), stale_days))
|
|
592
|
+
return result
|
|
593
|
+
except Exception as exc:
|
|
594
|
+
err = redact.redact_value(redact.redact_url_keys(str(exc)), secret_key)
|
|
595
|
+
# A read timeout is not a refusal. The request reached archive.org and
|
|
596
|
+
# we gave up waiting for the answer; the capture may have run to
|
|
597
|
+
# completion regardless. Observed live 2026-09-05 — a 30s read timeout
|
|
598
|
+
# on a save that archive.org had almost certainly accepted. Calling
|
|
599
|
+
# that "submission failed" overstates it in the other direction, the
|
|
600
|
+
# same way calling it "archived" would.
|
|
601
|
+
timed_out = isinstance(exc, requests.exceptions.Timeout)
|
|
602
|
+
log.warning(
|
|
603
|
+
"Wayback submission %s for %s: %s",
|
|
604
|
+
"timed out" if timed_out else "failed",
|
|
605
|
+
url,
|
|
606
|
+
err,
|
|
607
|
+
)
|
|
608
|
+
result = {
|
|
609
|
+
"url": url,
|
|
610
|
+
"submitted": False,
|
|
611
|
+
"job_id": None,
|
|
612
|
+
"error": err,
|
|
613
|
+
# The sentence a caller is safe to surface to an end user.
|
|
614
|
+
# ``error`` stays raw (redacted) for logging.
|
|
615
|
+
"error_summary": _transport_failure_summary(exc, "when asked to capture"),
|
|
616
|
+
}
|
|
617
|
+
if timed_out:
|
|
618
|
+
result["outcome_unknown"] = True
|
|
619
|
+
return result
|
|
620
|
+
|
|
621
|
+
|
|
622
|
+
def check_job_status(
|
|
623
|
+
job_id, timeout=15, access_key=None, secret_key=None, stale_days=None
|
|
624
|
+
):
|
|
625
|
+
"""Read the outcome of an SPN2 capture job. Never raises.
|
|
626
|
+
|
|
627
|
+
**This endpoint is credential-only.** Probed 2026-09-05: both
|
|
628
|
+
``GET /save/status/<job_id>`` and the bare ``GET /save/status`` answer
|
|
629
|
+
``401 {"message": "You need to be logged in to use Save Page Now."}``
|
|
630
|
+
without an ``Authorization`` header, and answer the same to a well-formed
|
|
631
|
+
but wrong ``LOW`` credential. There is therefore no unauthenticated way to
|
|
632
|
+
find out how a capture went, which is exactly why the unauthenticated
|
|
633
|
+
submission path reads its answer off the redirect instead.
|
|
634
|
+
|
|
635
|
+
Goes through the same ``_paced_get`` as the availability lookup, so it
|
|
636
|
+
shares one pacing clock, one backoff and one breaker budget with every
|
|
637
|
+
other archive.org call in the process — per this module's rate-limit
|
|
638
|
+
design, which exists because archive.org throttles per IP, not per
|
|
639
|
+
endpoint.
|
|
640
|
+
|
|
641
|
+
Returns a dict with:
|
|
642
|
+
job_id str
|
|
643
|
+
state str — one of:
|
|
644
|
+
"success" capture completed; snapshot fields set
|
|
645
|
+
"pending" archive.org is still working on it
|
|
646
|
+
"failed" archive.org tried and could not
|
|
647
|
+
"not_checked" we did not ask (no creds, breaker
|
|
648
|
+
tripped) — asserts nothing either way
|
|
649
|
+
"unknown" we asked and could not read the answer
|
|
650
|
+
reason str | None — why, for every state except "success"
|
|
651
|
+
snapshot_url str — present only on "success"
|
|
652
|
+
snapshot_ts str
|
|
653
|
+
snapshot_age_days int | None
|
|
654
|
+
snapshot_stale bool
|
|
655
|
+
"""
|
|
656
|
+
if not job_id:
|
|
657
|
+
return {
|
|
658
|
+
"job_id": job_id,
|
|
659
|
+
"state": "not_checked",
|
|
660
|
+
"reason": "no SPN2 job id was recorded for this submission",
|
|
661
|
+
}
|
|
662
|
+
if not (access_key and secret_key):
|
|
663
|
+
return {
|
|
664
|
+
"job_id": job_id,
|
|
665
|
+
"state": "not_checked",
|
|
666
|
+
"reason": (
|
|
667
|
+
"archive.org's job-status endpoint requires credentials (it "
|
|
668
|
+
"answers 401 without them) — pass access_key/secret_key to "
|
|
669
|
+
"have capture outcomes verified"
|
|
670
|
+
),
|
|
671
|
+
}
|
|
672
|
+
# Same reasoning as check(): once archive.org has refused repeatedly, every
|
|
673
|
+
# further call costs the run seconds of pacing to collect another 429.
|
|
674
|
+
if rate_limited_out():
|
|
675
|
+
return {
|
|
676
|
+
"job_id": job_id,
|
|
677
|
+
"state": "not_checked",
|
|
678
|
+
"reason": (
|
|
679
|
+
"skipped: archive.org rate limit tripped earlier this run "
|
|
680
|
+
"(no job-status lookup attempted)"
|
|
681
|
+
),
|
|
682
|
+
}
|
|
683
|
+
|
|
684
|
+
headers = dict(DEFAULT_HEADERS)
|
|
685
|
+
headers["Authorization"] = f"LOW {access_key}:{secret_key}"
|
|
686
|
+
headers["Accept"] = "application/json"
|
|
687
|
+
try:
|
|
688
|
+
resp = _paced_get(f"{_SAVE_STATUS_API}/{job_id}", timeout, headers=headers)
|
|
689
|
+
except Exception as exc:
|
|
690
|
+
err = redact.redact_value(redact.redact_url_keys(str(exc)), secret_key)
|
|
691
|
+
status_code = getattr(getattr(exc, "response", None), "status_code", None)
|
|
692
|
+
if status_code == 401:
|
|
693
|
+
# Verified shape, not a guess — see this function's docstring.
|
|
694
|
+
err = (
|
|
695
|
+
"archive.org rejected the credentials for the job-status "
|
|
696
|
+
"endpoint (401). The capture outcome is unknown, not failed."
|
|
697
|
+
)
|
|
698
|
+
log.debug("Wayback job status lookup failed for %s: %s", job_id, err)
|
|
699
|
+
if status_code == 401:
|
|
700
|
+
return {"job_id": job_id, "state": "unknown", "reason": err}
|
|
701
|
+
return {
|
|
702
|
+
"job_id": job_id,
|
|
703
|
+
"state": "unknown",
|
|
704
|
+
"reason": _transport_failure_summary(exc, "about this capture"),
|
|
705
|
+
"raw_error": err,
|
|
706
|
+
}
|
|
707
|
+
|
|
708
|
+
try:
|
|
709
|
+
data = resp.json()
|
|
710
|
+
except ValueError:
|
|
711
|
+
# Same misdirection guard as check(): a 200 that isn't JSON is an error
|
|
712
|
+
# or challenge page, and reporting the decoder's complaint sends the
|
|
713
|
+
# reader hunting for a parser bug that isn't there.
|
|
714
|
+
return {
|
|
715
|
+
"job_id": job_id,
|
|
716
|
+
"state": "unknown",
|
|
717
|
+
"reason": (
|
|
718
|
+
f"archive.org returned a non-JSON {resp.status_code} response "
|
|
719
|
+
f"({len(resp.content)} bytes) from the job-status API"
|
|
720
|
+
),
|
|
721
|
+
}
|
|
722
|
+
|
|
723
|
+
status = str(data.get("status") or "").strip().lower()
|
|
724
|
+
if status == "success":
|
|
725
|
+
ts = str(data.get("timestamp") or "")
|
|
726
|
+
original = str(data.get("original_url") or "")
|
|
727
|
+
if not ts or not original:
|
|
728
|
+
# Reported success without naming what it captured. Don't invent a
|
|
729
|
+
# snapshot URL out of half an answer.
|
|
730
|
+
return {
|
|
731
|
+
"job_id": job_id,
|
|
732
|
+
"state": "unknown",
|
|
733
|
+
"reason": (
|
|
734
|
+
"archive.org reported the capture succeeded but named no "
|
|
735
|
+
"timestamp/original_url, so no snapshot URL can be given"
|
|
736
|
+
),
|
|
737
|
+
}
|
|
738
|
+
result = {"job_id": job_id, "state": "success", "reason": None}
|
|
739
|
+
result.update(
|
|
740
|
+
_snapshot_state(
|
|
741
|
+
f"https://web.archive.org/web/{ts}/{original}", ts, stale_days
|
|
742
|
+
)
|
|
743
|
+
)
|
|
744
|
+
return result
|
|
745
|
+
if status == "pending":
|
|
746
|
+
return {
|
|
747
|
+
"job_id": job_id,
|
|
748
|
+
"state": "pending",
|
|
749
|
+
"reason": "archive.org has not finished this capture yet",
|
|
750
|
+
}
|
|
751
|
+
if status == "error":
|
|
752
|
+
# SPN2 spreads the explanation over three optional fields; take the most
|
|
753
|
+
# human one present rather than whichever happens to be first.
|
|
754
|
+
detail = (
|
|
755
|
+
data.get("message") or data.get("status_ext") or data.get("exception") or ""
|
|
756
|
+
)
|
|
757
|
+
return {
|
|
758
|
+
"job_id": job_id,
|
|
759
|
+
"state": "failed",
|
|
760
|
+
"reason": str(detail).strip()
|
|
761
|
+
or "archive.org reported an unspecified error",
|
|
762
|
+
}
|
|
763
|
+
return {
|
|
764
|
+
"job_id": job_id,
|
|
765
|
+
"state": "unknown",
|
|
766
|
+
"reason": (
|
|
767
|
+
f"archive.org reported an unrecognized job status {status!r}"
|
|
768
|
+
if status
|
|
769
|
+
else "archive.org's job-status response carried no status field"
|
|
770
|
+
),
|
|
771
|
+
}
|
|
772
|
+
|
|
773
|
+
|
|
774
|
+
def capture_capacity(timeout=15, access_key=None, secret_key=None):
|
|
775
|
+
"""How much Save Page Now capacity this account has right now. Never raises.
|
|
776
|
+
|
|
777
|
+
``GET /save/status/user``, credential-only like the job-status endpoint.
|
|
778
|
+
Measured live 2026-09-06::
|
|
779
|
+
|
|
780
|
+
{"processing":0,"available":3,"daily_captures":49,"daily_captures_limit":30000}
|
|
781
|
+
|
|
782
|
+
Why this is worth a request: every concurrency number governing submission
|
|
783
|
+
is otherwise invented, picked by watching archive.org get upset, and static
|
|
784
|
+
— it cannot tell a run with three free capture slots from a run with none.
|
|
785
|
+
This endpoint answers the question directly, and it answers it *before*
|
|
786
|
+
the requests are spent rather than after, which is the difference between
|
|
787
|
+
pacing and apologising. The circuit breaker stays exactly where it is;
|
|
788
|
+
this only narrows what a caller attempts in the first place.
|
|
789
|
+
|
|
790
|
+
Deliberately never *raises* the concurrency ceiling a caller might derive
|
|
791
|
+
from it — a reading can only make a run more cautious, so a wrong or stale
|
|
792
|
+
answer cannot make things worse.
|
|
793
|
+
|
|
794
|
+
Returns a dict with:
|
|
795
|
+
available int | None — concurrent capture slots free now
|
|
796
|
+
processing int | None — captures this account has in flight
|
|
797
|
+
daily_captures int | None
|
|
798
|
+
daily_captures_limit int | None
|
|
799
|
+
daily_exhausted bool — quota is used up; submitting is pointless
|
|
800
|
+
known bool — False when we could not find out
|
|
801
|
+
reason str | None — why not, when ``known`` is False
|
|
802
|
+
"""
|
|
803
|
+
unknown = {
|
|
804
|
+
"available": None,
|
|
805
|
+
"processing": None,
|
|
806
|
+
"daily_captures": None,
|
|
807
|
+
"daily_captures_limit": None,
|
|
808
|
+
"daily_exhausted": False,
|
|
809
|
+
"known": False,
|
|
810
|
+
}
|
|
811
|
+
if not (access_key and secret_key):
|
|
812
|
+
return {
|
|
813
|
+
**unknown,
|
|
814
|
+
"reason": (
|
|
815
|
+
"archive.org reports capture capacity only to an authenticated "
|
|
816
|
+
"account; pass access_key/secret_key"
|
|
817
|
+
),
|
|
818
|
+
}
|
|
819
|
+
if rate_limited_out():
|
|
820
|
+
return {
|
|
821
|
+
**unknown,
|
|
822
|
+
"reason": (
|
|
823
|
+
"skipped: archive.org rate limit tripped earlier this run "
|
|
824
|
+
"(no capacity lookup attempted)"
|
|
825
|
+
),
|
|
826
|
+
}
|
|
827
|
+
|
|
828
|
+
headers = dict(DEFAULT_HEADERS)
|
|
829
|
+
headers["Authorization"] = f"LOW {access_key}:{secret_key}"
|
|
830
|
+
headers["Accept"] = "application/json"
|
|
831
|
+
try:
|
|
832
|
+
resp = _paced_get(_SAVE_USER_STATUS_API, timeout, headers=headers)
|
|
833
|
+
data = resp.json()
|
|
834
|
+
except Exception as exc:
|
|
835
|
+
log.debug("Wayback capture-capacity lookup failed: %s", exc)
|
|
836
|
+
return {
|
|
837
|
+
**unknown,
|
|
838
|
+
"reason": _transport_failure_summary(exc, "for capture capacity"),
|
|
839
|
+
}
|
|
840
|
+
|
|
841
|
+
def _int(key):
|
|
842
|
+
value = data.get(key)
|
|
843
|
+
return value if isinstance(value, int) else None
|
|
844
|
+
|
|
845
|
+
used, limit = _int("daily_captures"), _int("daily_captures_limit")
|
|
846
|
+
return {
|
|
847
|
+
"available": _int("available"),
|
|
848
|
+
"processing": _int("processing"),
|
|
849
|
+
"daily_captures": used,
|
|
850
|
+
"daily_captures_limit": limit,
|
|
851
|
+
# Only assert exhaustion when both numbers are real. "Unknown" must not
|
|
852
|
+
# collapse into "you are out of quota" and stop a run submitting.
|
|
853
|
+
"daily_exhausted": bool(
|
|
854
|
+
used is not None and limit is not None and used >= limit
|
|
855
|
+
),
|
|
856
|
+
"known": True,
|
|
857
|
+
"reason": None,
|
|
858
|
+
}
|
|
859
|
+
|
|
860
|
+
|
|
861
|
+
def system_status(timeout=15, access_key=None, secret_key=None):
|
|
862
|
+
"""Is archive.org's capture system healthy, or are we the problem? Never raises.
|
|
863
|
+
|
|
864
|
+
``GET /save/status/system``. Measured live 2026-09-06::
|
|
865
|
+
|
|
866
|
+
{"recent_captures":941,"status":"ok","queues":{"spn2-captures":0, ...13 queues}}
|
|
867
|
+
|
|
868
|
+
Why this is worth asking. When submission degrades, everything a caller
|
|
869
|
+
can see looks the same from the inside: a 520, a read timeout, a refused
|
|
870
|
+
connection. Those are equally consistent with "we asked too often" and with
|
|
871
|
+
"the service is having a bad afternoon" — the same misdiagnosis as treating
|
|
872
|
+
a User-Agent block as rate limiting. The fix for one is to back off, and
|
|
873
|
+
the fix for the other is to wait and stop blaming yourself.
|
|
874
|
+
|
|
875
|
+
Meant to be asked once per run and only when something has already gone
|
|
876
|
+
wrong, so a healthy run pays nothing for it.
|
|
877
|
+
|
|
878
|
+
Returns a dict with:
|
|
879
|
+
ok bool | None — archive.org's own health verdict
|
|
880
|
+
status str — the raw status string it reported
|
|
881
|
+
recent_captures int | None
|
|
882
|
+
busiest_queue (name, depth) | None — the deepest non-empty queue
|
|
883
|
+
known bool — False when we could not find out
|
|
884
|
+
reason str | None
|
|
885
|
+
"""
|
|
886
|
+
unknown = {
|
|
887
|
+
"ok": None,
|
|
888
|
+
"status": "",
|
|
889
|
+
"recent_captures": None,
|
|
890
|
+
"busiest_queue": None,
|
|
891
|
+
"known": False,
|
|
892
|
+
}
|
|
893
|
+
if rate_limited_out():
|
|
894
|
+
# Deliberately still asks nothing: the breaker exists because further
|
|
895
|
+
# calls cost the run pacing budget, and that applies to diagnosis too.
|
|
896
|
+
return {
|
|
897
|
+
**unknown,
|
|
898
|
+
"reason": (
|
|
899
|
+
"skipped: archive.org rate limit tripped earlier this run "
|
|
900
|
+
"(no service-status lookup attempted)"
|
|
901
|
+
),
|
|
902
|
+
}
|
|
903
|
+
|
|
904
|
+
headers = dict(DEFAULT_HEADERS)
|
|
905
|
+
headers["Accept"] = "application/json"
|
|
906
|
+
if access_key and secret_key:
|
|
907
|
+
headers["Authorization"] = f"LOW {access_key}:{secret_key}"
|
|
908
|
+
try:
|
|
909
|
+
resp = _paced_get(_SAVE_SYSTEM_STATUS_API, timeout, headers=headers)
|
|
910
|
+
data = resp.json()
|
|
911
|
+
except Exception as exc:
|
|
912
|
+
log.debug("Wayback system-status lookup failed: %s", exc)
|
|
913
|
+
return {
|
|
914
|
+
**unknown,
|
|
915
|
+
"reason": _transport_failure_summary(exc, "for its service status"),
|
|
916
|
+
}
|
|
917
|
+
|
|
918
|
+
status = str(data.get("status") or "").strip()
|
|
919
|
+
queues = data.get("queues")
|
|
920
|
+
busiest = None
|
|
921
|
+
if isinstance(queues, dict):
|
|
922
|
+
depths = [(n, d) for n, d in queues.items() if isinstance(d, int) and d > 0]
|
|
923
|
+
if depths:
|
|
924
|
+
busiest = max(depths, key=lambda kv: kv[1])
|
|
925
|
+
recent = data.get("recent_captures")
|
|
926
|
+
return {
|
|
927
|
+
"ok": status.lower() == "ok",
|
|
928
|
+
"status": status,
|
|
929
|
+
"recent_captures": recent if isinstance(recent, int) else None,
|
|
930
|
+
"busiest_queue": busiest,
|
|
931
|
+
"known": True,
|
|
932
|
+
"reason": None,
|
|
933
|
+
}
|
|
934
|
+
|
|
935
|
+
|
|
936
|
+
def service_health_note(status):
|
|
937
|
+
"""One clause naming who was at fault, or None when it adds nothing.
|
|
938
|
+
|
|
939
|
+
Returns None for a healthy service *and* for an unknown one: appending
|
|
940
|
+
"we could not tell" to every failed submission would be noise on top of a
|
|
941
|
+
failure the caller is already looking at.
|
|
942
|
+
"""
|
|
943
|
+
if not status or not status.get("known"):
|
|
944
|
+
return None
|
|
945
|
+
if status.get("ok"):
|
|
946
|
+
busiest = status.get("busiest_queue")
|
|
947
|
+
if busiest and busiest[1] >= _BUSY_QUEUE_DEPTH:
|
|
948
|
+
return (
|
|
949
|
+
f"archive.org reported itself healthy but busy at the time "
|
|
950
|
+
f"({busiest[0]} queue {busiest[1]} deep)"
|
|
951
|
+
)
|
|
952
|
+
return "archive.org reported its capture system healthy at the time"
|
|
953
|
+
return (
|
|
954
|
+
f"archive.org reported its capture system as "
|
|
955
|
+
f"{status.get('status') or 'not ok'} at the time — this was the service, "
|
|
956
|
+
f"not the caller"
|
|
957
|
+
)
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: spn-client
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: A hardened client for archive.org's availability and Save Page Now (SPN2) APIs — process-wide rate pacing with a circuit breaker, dual-mode anonymous/S3-authenticated submission, and an explicit archive-outcome vocabulary.
|
|
5
|
+
Project-URL: Homepage, https://github.com/MHammett/spn-client
|
|
6
|
+
Project-URL: Issues, https://github.com/MHammett/spn-client/issues
|
|
7
|
+
License-Expression: MIT
|
|
8
|
+
License-File: LICENSE
|
|
9
|
+
Requires-Python: >=3.10
|
|
10
|
+
Requires-Dist: requests<3.0,>=2.31.0
|
|
11
|
+
Description-Content-Type: text/markdown
|
|
12
|
+
|
|
13
|
+
# spn-client
|
|
14
|
+
|
|
15
|
+
A hardened Python client for archive.org's availability API and Save Page Now
|
|
16
|
+
(SPN2) — the pieces that are easy to get wrong when you actually run this
|
|
17
|
+
against archive.org at any volume:
|
|
18
|
+
|
|
19
|
+
- **Process-wide rate pacing with a circuit breaker.** archive.org throttles
|
|
20
|
+
per IP, not per thread or per endpoint. A naive per-thread backoff (or none
|
|
21
|
+
at all) looks fine in testing and then silently stops archiving anything
|
|
22
|
+
the first time it's run concurrently or against a busy queue.
|
|
23
|
+
- **Both submission modes.** Anonymous `GET /save/<url>` (capture runs inline,
|
|
24
|
+
answer comes back as a redirect) and S3-key-authenticated `POST /save`
|
|
25
|
+
(capture is queued, answer comes back as a `job_id` you poll for).
|
|
26
|
+
- **An explicit outcome vocabulary.** `submitted` is not `archived`. This
|
|
27
|
+
library only ever reports `archived: True` alongside a real snapshot URL
|
|
28
|
+
that came back from archive.org — never as a synonym for "we asked."
|
|
29
|
+
- **Staleness checking**, so callers can skip re-archiving a URL that already
|
|
30
|
+
has a recent-enough snapshot.
|
|
31
|
+
|
|
32
|
+
Every non-obvious piece of behavior in `client.py` is a documented response to
|
|
33
|
+
a specific, dated, measured incident against the real archive.org API — not
|
|
34
|
+
a guess.
|
|
35
|
+
|
|
36
|
+
## Install
|
|
37
|
+
|
|
38
|
+
```bash
|
|
39
|
+
pip install spn-client
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
## Usage
|
|
43
|
+
|
|
44
|
+
```python
|
|
45
|
+
import spn_client
|
|
46
|
+
|
|
47
|
+
result = spn_client.check("https://example.com/some-page")
|
|
48
|
+
if result["archived"] is False or result.get("snapshot_stale"):
|
|
49
|
+
submission = spn_client.submit(
|
|
50
|
+
"https://example.com/some-page",
|
|
51
|
+
access_key=ACCESS_KEY, # optional — omit for anonymous, lower-rate submission
|
|
52
|
+
secret_key=SECRET_KEY,
|
|
53
|
+
)
|
|
54
|
+
if submission["job_id"]:
|
|
55
|
+
# authenticated path: poll for the real outcome
|
|
56
|
+
status = spn_client.check_job_status(
|
|
57
|
+
submission["job_id"], access_key=ACCESS_KEY, secret_key=SECRET_KEY
|
|
58
|
+
)
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
Call `spn_client.reset_rate_limit_state()` once at the start of each
|
|
62
|
+
independent run/process if you're running this as a long-lived worker —
|
|
63
|
+
the pacing/breaker state is process-wide and intentionally does not reset
|
|
64
|
+
itself, so a breaker tripped by one run would otherwise silently degrade
|
|
65
|
+
the next.
|
|
66
|
+
|
|
67
|
+
## What this doesn't do
|
|
68
|
+
|
|
69
|
+
This is the archive.org client only — it has no opinion about:
|
|
70
|
+
- What staleness policy is right for your use case (`stale_days` is always a
|
|
71
|
+
caller-supplied parameter)
|
|
72
|
+
- How you track which URLs you've already archived (that's a caller-side
|
|
73
|
+
ledger/cache concern)
|
|
74
|
+
- Batch-loop concerns like a progress heartbeat or a wall-clock budget across
|
|
75
|
+
many submissions — those depend on your own operational needs
|
|
76
|
+
|
|
77
|
+
## License
|
|
78
|
+
|
|
79
|
+
MIT
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
spn_client/__init__.py,sha256=o7XRJeXD823itejqfcy9iXig-DeKJx_3eSvB1Px76lI,964
|
|
2
|
+
spn_client/_http.py,sha256=Hzb5Qy56e-g3kvzmTt5ngTEgUs3xlQFARUg5zt_pz5I,706
|
|
3
|
+
spn_client/_redact.py,sha256=ZGoN5V8_Z5m0sW__h2kmj_AVelX5x-zlHc9WSKFmN-U,2125
|
|
4
|
+
spn_client/client.py,sha256=7PZLVsnuK97NHq66PTDzEkhQGEibXRLCCsh6RYJNRo4,41913
|
|
5
|
+
spn_client-0.1.0.dist-info/METADATA,sha256=7QPQ2YcuvlKpIQPiol5isTkY_YoSrQC56diICsnlgsM,3178
|
|
6
|
+
spn_client-0.1.0.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
|
|
7
|
+
spn_client-0.1.0.dist-info/licenses/LICENSE,sha256=e2rCCsFHz2bduyCacklOEqV5DICqfjL9jX6gzpA1PY4,1069
|
|
8
|
+
spn_client-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Mike Hammett
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|