spn-client 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
spn_client/__init__.py ADDED
@@ -0,0 +1,43 @@
1
+ """spn-client: a hardened client for archive.org's availability and Save Page Now APIs."""
2
+
3
+ from spn_client.client import (
4
+ ARCHIVE_ARCHIVED,
5
+ ARCHIVE_CAPTURE_FAILED,
6
+ ARCHIVE_NOT_ATTEMPTED,
7
+ ARCHIVE_OUTCOME_LABELS,
8
+ ARCHIVE_PENDING,
9
+ ARCHIVE_SUBMIT_FAILED,
10
+ ARCHIVE_SUBMITTED,
11
+ DEFAULT_STALE_DAYS,
12
+ capture_capacity,
13
+ check,
14
+ check_job_status,
15
+ rate_limited_out,
16
+ reset_rate_limit_state,
17
+ service_health_note,
18
+ snapshot_raw_url,
19
+ submit,
20
+ system_status,
21
+ )
22
+
23
+ __version__ = "0.1.0"
24
+
25
+ __all__ = [
26
+ "ARCHIVE_ARCHIVED",
27
+ "ARCHIVE_CAPTURE_FAILED",
28
+ "ARCHIVE_NOT_ATTEMPTED",
29
+ "ARCHIVE_OUTCOME_LABELS",
30
+ "ARCHIVE_PENDING",
31
+ "ARCHIVE_SUBMIT_FAILED",
32
+ "ARCHIVE_SUBMITTED",
33
+ "DEFAULT_STALE_DAYS",
34
+ "capture_capacity",
35
+ "check",
36
+ "check_job_status",
37
+ "rate_limited_out",
38
+ "reset_rate_limit_state",
39
+ "service_health_note",
40
+ "snapshot_raw_url",
41
+ "submit",
42
+ "system_status",
43
+ ]
spn_client/_http.py ADDED
@@ -0,0 +1,22 @@
1
+ """Outbound-HTTP identity for spn-client.
2
+
3
+ A dedicated User-Agent, separate from any consuming project's own — this
4
+ package is meant to be embedded in several unrelated projects, and its
5
+ requests to archive.org should identify the library making them, not whatever
6
+ application happens to import it.
7
+ """
8
+
9
+ from importlib import metadata
10
+
11
+ try:
12
+ _VERSION = metadata.version("spn-client")
13
+ except metadata.PackageNotFoundError: # pragma: no cover - source/dev checkout
14
+ _VERSION = "0.0.0"
15
+
16
+ USER_AGENT = f"spn-client/{_VERSION}"
17
+
18
+ DEFAULT_HEADERS = {
19
+ "User-Agent": USER_AGENT,
20
+ "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
21
+ "Accept-Language": "en-US,en;q=0.9",
22
+ }
spn_client/_redact.py ADDED
@@ -0,0 +1,51 @@
1
+ """Secret-redaction helpers.
2
+
3
+ Vendored from a sister project's ``ci_core.redact`` module rather than taken
4
+ as a dependency on it — this package is deliberately decoupled from any
5
+ project-specific shared library, and these two functions are the only pieces
6
+ ``client.py`` needs.
7
+ """
8
+
9
+ import re
10
+
11
+ # The sensitive word a credential parameter ends with. Hyphenated spellings are
12
+ # not hypothetical: Azure OpenAI names its parameter `api-key` and Google sends
13
+ # `X-Goog-Api-Key`. An underscore-only list matched neither, nor
14
+ # `client_secret`, `refresh_token`, `subscription-key`, `password`, `auth` or
15
+ # `sig` — eight of twelve real-world spellings tested on 2026-09-04 passed
16
+ # straight through into whatever log or report the error text reached.
17
+ _SENSITIVE_WORD = (
18
+ r"(?:api[-_]?key|access[-_]?token|refresh[-_]?token|id[-_]?token"
19
+ r"|client[-_]?secret|subscription[-_]?key|session[-_]?key"
20
+ r"|key|token|secret|password|passwd|credentials?|auth|signature|sig)"
21
+ )
22
+
23
+ # Any dash/underscore-separated prefix, then that word, then `=`. Anchoring the
24
+ # end on `=` is what leaves innocent parameters alone: `author=` cannot match,
25
+ # because `auth` would have to consume the whole name and `or` is left over.
26
+ # `keywords=` survives for the same reason.
27
+ _PARAM_NAME = rf"(?:[A-Za-z0-9]+[-_])*{_SENSITIVE_WORD}"
28
+
29
+ # Matches ?key=..., &apiKey=..., &api-key=... in a URL and replaces the value.
30
+ # Stops at the next &, whitespace, quote, or closing bracket so we don't eat
31
+ # the rest of the message.
32
+ _KEY_QUERY_RE = re.compile(
33
+ rf"([?&]{_PARAM_NAME}=)[^&\s'\"<>)\]]+",
34
+ re.IGNORECASE,
35
+ )
36
+
37
+
38
+ def redact_url_keys(text):
39
+ """Replace key/token query-parameter values in any string with [REDACTED].
40
+
41
+ Provider-agnostic: works even when the key value isn't available to compare
42
+ against, because it matches on the parameter name.
43
+ """
44
+ return _KEY_QUERY_RE.sub(r"\1[REDACTED]", str(text))
45
+
46
+
47
+ def redact_value(text, secret):
48
+ """Replace a known secret value with [REDACTED] wherever it appears."""
49
+ if secret and secret in str(text):
50
+ return str(text).replace(secret, "[REDACTED]")
51
+ return str(text)
spn_client/client.py ADDED
@@ -0,0 +1,957 @@
1
+ """A hardened client for archive.org's availability and Save Page Now (SPN2) APIs.
2
+
3
+ Extracted from a citation-verification pipeline where this logic was built and
4
+ incident-tested against real archive.org behavior — process-wide rate pacing
5
+ with a circuit breaker, dual-mode (anonymous + S3-key-authenticated) capture
6
+ submission with async job-status polling, and an outcome vocabulary that
7
+ distinguishes "we asked archive.org" from "archive.org confirmed it archived
8
+ this." Every non-obvious behavior below is a direct response to a dated,
9
+ measured incident, not a speculative design.
10
+ """
11
+
12
+ import logging
13
+ import threading
14
+ import time
15
+ import re
16
+ from datetime import datetime, timezone
17
+
18
+ import requests
19
+
20
+ from spn_client._http import DEFAULT_HEADERS
21
+ from spn_client import _redact as redact
22
+
23
+ log = logging.getLogger(__name__)
24
+
25
+ _AVAILABILITY_API = "https://archive.org/wayback/available"
26
+ _SAVE_API = "https://web.archive.org/save"
27
+ _SAVE_STATUS_API = "https://web.archive.org/save/status"
28
+ _SAVE_USER_STATUS_API = "https://web.archive.org/save/status/user"
29
+ _SAVE_SYSTEM_STATUS_API = "https://web.archive.org/save/status/system"
30
+
31
+ #: Fallback staleness threshold (days) used only when a caller does not pass
32
+ #: its own ``stale_days``. Callers with an actual staleness policy should
33
+ #: always pass one explicitly — this exists so ``check()``/``submit()``/
34
+ #: ``check_job_status()`` have *a* number to compare against rather than
35
+ #: raising when none is given.
36
+ DEFAULT_STALE_DAYS = 30
37
+
38
+ # Matches a Wayback Machine snapshot URL and captures its embedded timestamp:
39
+ # https://web.archive.org/web/20250101000000/https://example.com/...
40
+ _ARCHIVE_URL_RE = re.compile(
41
+ r"https?://web\.archive\.org/web/(\d{4,14})(?:[a-z_]*)?/", re.IGNORECASE
42
+ )
43
+
44
+
45
+ #: What a run actually established about archiving a URL, as opposed to what
46
+ #: it *asked* for. The distinction is the whole point of this vocabulary: it
47
+ #: is tempting to record only ``submitted: True`` and report "submitted for
48
+ #: archiving", which a reader reasonably finishes as "so it is archived".
49
+ #: Nothing establishes that on its own. A capture archive.org silently
50
+ #: dropped looks exactly like one it completed.
51
+ #:
52
+ #: ``ARCHIVED`` is the only value that asserts a snapshot exists, and it is
53
+ #: only ever set alongside a ``snapshot_url`` that came back from archive.org.
54
+ ARCHIVE_SUBMITTED = "submitted" # accepted; outcome not established (see below)
55
+ ARCHIVE_ARCHIVED = "archived" # a snapshot URL came back — this one is real
56
+ ARCHIVE_PENDING = "pending" # job accepted, capture still running at report time
57
+ ARCHIVE_CAPTURE_FAILED = "capture_failed" # accepted, then the capture failed
58
+ ARCHIVE_SUBMIT_FAILED = "submit_failed" # archive.org refused the request itself
59
+ ARCHIVE_NOT_ATTEMPTED = "not_attempted" # we never asked; the reason is recorded
60
+
61
+ #: Human-readable phrasing for each outcome, for report output. Kept here so
62
+ #: two different renderers cannot drift into describing the same state two
63
+ #: different ways.
64
+ ARCHIVE_OUTCOME_LABELS = {
65
+ ARCHIVE_SUBMITTED: "submitted; capture outcome not established",
66
+ ARCHIVE_ARCHIVED: "archived",
67
+ ARCHIVE_PENDING: "submitted; capture still pending",
68
+ ARCHIVE_CAPTURE_FAILED: "capture failed",
69
+ ARCHIVE_SUBMIT_FAILED: "submission failed",
70
+ ARCHIVE_NOT_ATTEMPTED: "not submitted",
71
+ }
72
+
73
+
74
+ def _age_days_from_timestamp(ts):
75
+ """Days since a Wayback timestamp (YYYYMMDD...), or None if unparseable."""
76
+ if len(ts) >= 8:
77
+ try:
78
+ snap_dt = datetime.strptime(ts[:8], "%Y%m%d").replace(tzinfo=timezone.utc)
79
+ return (datetime.now(timezone.utc) - snap_dt).days
80
+ except ValueError:
81
+ pass
82
+ return None
83
+
84
+
85
+ # archive.org throttles the availability API at IP level with a long window, not
86
+ # per-second. Measured 2026-08-12: 12 consecutive requests all returned 429 —
87
+ # including the first — and it was still 429 after a 45s cooldown at one request
88
+ # every 6 seconds. A pipeline with no pacing and no backoff at all ended up with
89
+ # an archive thread that had been dead for at least two runs: 49 resolved items,
90
+ # 0 archived, ~49 rate-limited.
91
+ #
92
+ # (Re-probed 2026-08-15: six back-to-back lookups all returned 200. The throttle
93
+ # is episodic, so the guard has to be always-on rather than tuned to one window.)
94
+ #
95
+ # Two mechanisms, because one is not enough:
96
+ # * a process-wide minimum interval between calls, serialised on a lock. If
97
+ # callers run this from a thread pool, without this the pool's width
98
+ # decides the request rate.
99
+ # * retry with backoff that honours Retry-After, and a circuit breaker: once
100
+ # archive.org has said 429 repeatedly, further calls in the same run are
101
+ # skipped rather than spending the run's time collecting more 429s.
102
+ #
103
+ # Every piece of state below is process-wide, and that is the whole design
104
+ # constraint: ``check()`` may be called from several worker threads at once.
105
+ # Two consequences the first version of this code got wrong, both of which
106
+ # silently disabled the protection they were meant to provide:
107
+ #
108
+ # * **Per-thread backoff is not backoff.** Sleeping inside the failing thread
109
+ # leaves the other workers hammering archive.org at the full pace. A 429 has
110
+ # to move a clock every thread waits on, so the *process* slows down.
111
+ # * **"Consecutive" is not well defined across interleaved threads.** Resetting
112
+ # a shared counter on success let one worker's 200 erase four other workers'
113
+ # refusals, so a run being throttled 4-in-5 would never trip the breaker —
114
+ # precisely the case it exists for. The count is now a per-run budget that
115
+ # only moves up.
116
+ _MIN_INTERVAL_SECONDS = 3.0
117
+ _MAX_ATTEMPTS = 3
118
+ _BACKOFF_BASE_SECONDS = 5.0
119
+
120
+ #: Refused *lookups* tolerated in a run before we stop asking. Counted once per
121
+ #: lookup, not once per attempt, so the number means what it says: the
122
+ #: ``_MAX_ATTEMPTS`` retries inside one lookup are a single refusal. (Counting
123
+ #: attempts made this trip after two lookups rather than five.)
124
+ _CIRCUIT_TRIP_AFTER = 5
125
+
126
+ #: Queue depth at which archive.org counts as busy rather than merely up.
127
+ #: Measured healthy 2026-09-06 with every one of its thirteen capture queues
128
+ #: at zero, so any sustained backlog is worth telling the caller about — it
129
+ #: is the difference between "your submission failed" and "everyone's did".
130
+ _BUSY_QUEUE_DEPTH = 50
131
+
132
+ _pace_lock = threading.Lock()
133
+ _last_call_at = 0.0
134
+
135
+ #: Absolute ``time.monotonic()`` before which no thread may call. A 429 pushes
136
+ #: this forward, which is what makes the backoff process-wide.
137
+ _blocked_until = 0.0
138
+
139
+ #: Lookups this run that exhausted every attempt against a 429. Never reset on
140
+ #: success — see the note above on why "consecutive" cannot work here.
141
+ _rate_limited_lookups = 0
142
+
143
+
144
+ def reset_rate_limit_state():
145
+ """Clear the pacing clock, the backoff, and the circuit breaker.
146
+
147
+ Call this at the start of every run. The state is process-wide, so without
148
+ a reset a breaker tripped by one run would skip every archive.org call in
149
+ the next run inside the same process — a silently degraded run whose
150
+ submissions all report "skipped" for a limit that expired long ago.
151
+ """
152
+ global _last_call_at, _blocked_until, _rate_limited_lookups
153
+ with _pace_lock:
154
+ _last_call_at = 0.0
155
+ _blocked_until = 0.0
156
+ _rate_limited_lookups = 0
157
+
158
+
159
+ def rate_limited_out():
160
+ """True once the circuit breaker has tripped for this run."""
161
+ with _pace_lock:
162
+ return _rate_limited_lookups >= _CIRCUIT_TRIP_AFTER
163
+
164
+
165
+ def _pace():
166
+ """Block until the shared clock allows another call.
167
+
168
+ Waits for whichever is later: ``_MIN_INTERVAL_SECONDS`` since the last call,
169
+ or the end of a backoff that a 429 imposed on every thread. Sleeping while
170
+ holding the lock is deliberate — it is exactly what makes the interval
171
+ process-wide instead of per-thread.
172
+ """
173
+ global _last_call_at
174
+ with _pace_lock:
175
+ now = time.monotonic()
176
+ target = max(_last_call_at + _MIN_INTERVAL_SECONDS, _blocked_until)
177
+ if target > now:
178
+ time.sleep(target - now)
179
+ _last_call_at = time.monotonic()
180
+
181
+
182
+ def _note_rate_limited(retry_after):
183
+ """Back every thread off after a 429, not just the one that hit it."""
184
+ global _blocked_until
185
+ with _pace_lock:
186
+ _blocked_until = max(_blocked_until, time.monotonic() + retry_after)
187
+
188
+
189
+ def _note_lookup_refused():
190
+ """Count one lookup that never got past a 429."""
191
+ global _rate_limited_lookups
192
+ with _pace_lock:
193
+ _rate_limited_lookups += 1
194
+
195
+
196
+ def _retry_after_seconds(resp, attempt):
197
+ """Seconds to wait before retrying, preferring the server's own answer."""
198
+ header = (resp.headers or {}).get("Retry-After") if resp is not None else None
199
+ if header:
200
+ try:
201
+ return min(float(header), 60.0)
202
+ except (TypeError, ValueError):
203
+ pass
204
+ return _BACKOFF_BASE_SECONDS * (2**attempt)
205
+
206
+
207
+ def _paced_get(endpoint, timeout, params=None, headers=None):
208
+ """GET an archive.org endpoint with pacing, backoff, and a circuit breaker.
209
+
210
+ Raises the last exception if every attempt fails, so the caller's existing
211
+ error handling is unchanged.
212
+
213
+ There is no local sleep between attempts: a 429 pushes the shared clock out
214
+ and the wait then happens in the next ``_pace()``, which every thread goes
215
+ through. That is the difference between the process backing off and one
216
+ thread backing off while seven others keep the pressure on.
217
+
218
+ Endpoint-agnostic on purpose. The availability API and the Save Page Now
219
+ job-status API are the same host, the same IP-level throttle, and the same
220
+ consequence for ignoring it, so they share one pacing clock and one breaker
221
+ budget rather than each getting its own. Two schemes would each be pacing
222
+ against half the real request rate, which is how you end up rate-limited by
223
+ a system that believes it is being polite.
224
+ """
225
+ last_exc = None
226
+ rate_limited = False
227
+ for attempt in range(_MAX_ATTEMPTS):
228
+ _pace()
229
+ try:
230
+ resp = requests.get(
231
+ endpoint,
232
+ params=params,
233
+ timeout=timeout,
234
+ headers=headers or DEFAULT_HEADERS,
235
+ )
236
+ except Exception as exc:
237
+ last_exc = exc
238
+ continue
239
+ if resp.status_code == 429:
240
+ rate_limited = True
241
+ last_exc = requests.HTTPError(
242
+ f"429 Client Error: Too Many Requests for url: {resp.url}",
243
+ response=resp,
244
+ )
245
+ _note_rate_limited(_retry_after_seconds(resp, attempt))
246
+ continue
247
+ resp.raise_for_status()
248
+ return resp
249
+ # One refused lookup, however many attempts it took to establish that.
250
+ if rate_limited:
251
+ _note_lookup_refused()
252
+ raise last_exc
253
+
254
+
255
+ def _get_availability(url, timeout):
256
+ """GET the availability API through the shared pacing/backoff/breaker path."""
257
+ return _paced_get(_AVAILABILITY_API, timeout, params={"url": url})
258
+
259
+
260
+ def _transport_failure_summary(exc, what):
261
+ """One reader-facing sentence for an archive.org call that did not complete.
262
+
263
+ The raw exception is for the log, not for whoever reads a report built on
264
+ this. Left unfiltered it reaches the caller as
265
+ ``HTTPSConnectionPool(host='web.archive.org', port=443): Max retries
266
+ exceeded with url: /save/status/... (Caused by NewConnectionError(...
267
+ [WinError 10061] ...))`` — observed verbatim in a real run 2026-09-06. That
268
+ is a debugger's string in a message meant for someone deciding what to do
269
+ next, and it invites them to debug the networking instead of telling them
270
+ what it means for their submission.
271
+
272
+ Callers keep the raw text under a separate key so nothing is lost.
273
+ """
274
+ if isinstance(exc, requests.exceptions.Timeout):
275
+ return f"archive.org did not answer {what} within the timeout"
276
+ if isinstance(exc, requests.exceptions.ConnectionError):
277
+ return (
278
+ f"could not reach archive.org {what} — the connection was refused, "
279
+ f"dropped, or the host did not resolve"
280
+ )
281
+ status = getattr(getattr(exc, "response", None), "status_code", None)
282
+ if status:
283
+ return f"archive.org answered HTTP {status} {what}"
284
+ return f"the request to archive.org {what} failed"
285
+
286
+
287
+ def snapshot_raw_url(snapshot_url):
288
+ """The ``id_`` form of a snapshot URL: the original captured bytes.
289
+
290
+ ``https://web.archive.org/web/<ts>/<url>`` serves the capture with
291
+ archive.org's own banner and its ``wombat.js`` URL-rewriting shim injected.
292
+ Appending ``id_`` to the timestamp — ``/web/<ts>id_/<url>`` — serves what was
293
+ actually captured, unmodified.
294
+
295
+ That difference is what makes an archived copy checkable against the live
296
+ page at all. Measured 2026-09-06 across three URLs, one of them archived a
297
+ week earlier: the ``id_`` body was **byte-identical** to the live page
298
+ (SHA-256 equal, 6639/10923/9569 bytes), while the ordinary form was nearly
299
+ three times the size for the same document. Comparing the injected form
300
+ would be comparing archive.org's chrome, and would report every page as
301
+ divergent from itself.
302
+
303
+ Returns ``None`` if ``snapshot_url`` is not a snapshot URL.
304
+ """
305
+ if not snapshot_url:
306
+ return None
307
+ m = _ARCHIVE_URL_RE.search(snapshot_url)
308
+ if not m:
309
+ return None
310
+ # Replace the matched "/web/<ts><flags>/" with "/web/<ts>id_/".
311
+ return (
312
+ snapshot_url[: m.start()]
313
+ + f"https://web.archive.org/web/{m.group(1)}id_/"
314
+ + snapshot_url[m.end() :]
315
+ )
316
+
317
+
318
+ def _snapshot_state(snapshot_url, ts, stale_days=None):
319
+ """The four snapshot fields every caller reports, from a URL and timestamp.
320
+
321
+ One implementation because ``check()``, ``submit()`` and
322
+ ``check_job_status()`` all have to answer "how old is this snapshot, and
323
+ is that too old" the same way. They previously could disagree because only
324
+ one of them ever answered it; now that a submission can come back with a
325
+ real snapshot, they share the arithmetic instead of risking that.
326
+ """
327
+ threshold = stale_days if stale_days is not None else DEFAULT_STALE_DAYS
328
+ age = _age_days_from_timestamp(ts) if ts else None
329
+ return {
330
+ "snapshot_url": snapshot_url,
331
+ "snapshot_ts": ts,
332
+ "snapshot_age_days": age,
333
+ "snapshot_stale": age is not None and age > threshold,
334
+ }
335
+
336
+
337
+ def check(url, timeout=10, stale_days=None):
338
+ """Check if a URL has a Wayback Machine snapshot and how fresh it is.
339
+
340
+ If ``url`` is itself a Wayback Machine snapshot link, it is recognized as
341
+ already-archived: the snapshot date is read from the URL's embedded
342
+ timestamp (no archive-of-an-archive lookup), and ``is_archive_url`` is set.
343
+ Whether that archive link actually resolves is a separate question this
344
+ function does not answer.
345
+
346
+ Returns a dict with:
347
+ archived bool | None — True/False, or None on network error
348
+ is_archive_url bool — True when the link itself is a web.archive.org URL
349
+ snapshot_url str — direct https://web.archive.org/web/... URL
350
+ snapshot_ts str — raw Wayback timestamp (YYYYMMDDHHMMSS)
351
+ snapshot_age_days int — days since the snapshot was taken
352
+ snapshot_stale bool — True when older than the stale threshold
353
+ error str — set only on network/parse failure
354
+ """
355
+ # The link is already a Wayback snapshot — read its date from the URL itself.
356
+ m = _ARCHIVE_URL_RE.search(url)
357
+ if m:
358
+ result = {"url": url, "archived": True, "is_archive_url": True}
359
+ result.update(_snapshot_state(url, m.group(1), stale_days))
360
+ return result
361
+
362
+ # Once archive.org has refused repeatedly, stop asking: every further call
363
+ # costs the run several seconds of pacing to collect another 429.
364
+ if rate_limited_out():
365
+ return {
366
+ "url": url,
367
+ "archived": None,
368
+ "error": (
369
+ "skipped: archive.org rate limit tripped earlier this run "
370
+ "(no snapshot lookup attempted)"
371
+ ),
372
+ }
373
+ try:
374
+ resp = _get_availability(url, timeout)
375
+ except Exception as exc:
376
+ log.debug("Wayback availability check failed for %s: %s", url, exc)
377
+ return {"url": url, "archived": None, "error": str(exc)}
378
+
379
+ try:
380
+ data = resp.json()
381
+ except ValueError as exc:
382
+ # A 200 that isn't JSON means archive.org served something other than an
383
+ # availability answer — an error or challenge page. Reporting the raw
384
+ # decoder message ("Expecting value: line 1 column 1") sends the reader
385
+ # hunting for a parser bug that isn't there; it is the same misdirection
386
+ # upstream reported as akamhy/waybackpy#200, where a throttled lookup
387
+ # surfaces as invalid JSON. Say what actually arrived instead.
388
+ log.debug("Wayback availability returned non-JSON for %s: %s", url, exc)
389
+ return {
390
+ "url": url,
391
+ "archived": None,
392
+ "error": (
393
+ f"archive.org returned a non-JSON {resp.status_code} response "
394
+ f"({len(resp.content)} bytes) from the availability API"
395
+ ),
396
+ }
397
+
398
+ closest = data.get("archived_snapshots", {}).get("closest", {})
399
+ if not closest.get("available"):
400
+ return {"url": url, "archived": False}
401
+
402
+ result = {"url": url, "archived": True}
403
+ # The availability API reports the captured response's own HTTP status, and
404
+ # it's tempting to discard it. It matters: a snapshot can be a capture of a
405
+ # 403 block page or a 404, and "a snapshot exists" then means the opposite
406
+ # of what a caller might assume. A capture made through ``submit()`` here
407
+ # always sends ``capture_all=0`` so it never makes one, but a *pre-existing*
408
+ # snapshot is outside this package's control.
409
+ status = closest.get("status")
410
+ if status:
411
+ result["snapshot_status"] = str(status)
412
+ result["snapshot_is_error_capture"] = not str(status).startswith("2")
413
+ result.update(
414
+ _snapshot_state(
415
+ closest.get("url", ""), closest.get("timestamp", ""), stale_days
416
+ )
417
+ )
418
+ return result
419
+
420
+
421
+ #: Save Page Now capture options this client sets deliberately.
422
+ #:
423
+ #: Read off the live ``/save`` form 2026-09-06 — its control names *are* the API
424
+ #: parameter names: ``capture_outlinks``, ``capture_all``, ``capture_screenshot``,
425
+ #: ``disable_adblocker``, ``wm-save-mywebarchive``, ``email_result``, ``wacz``.
426
+ #: All of them work on the authenticated endpoint. Only one is set, and it's a
427
+ #: correctness fix rather than a feature:
428
+ #:
429
+ #: ``capture_all=0`` — **the form defaults this ON**, and on means "archive the
430
+ #: page even if it answers 4xx/5xx". For most durability use cases that is a
431
+ #: way to manufacture a lie: a source that blocks the capture with a 403
432
+ #: would get its block page archived, a later ``check()`` would find a
433
+ #: snapshot, and a caller would report the URL as archived when what's
434
+ #: archived is an error page. It bites hardest on exactly the sources that
435
+ #: refuse automated requests — the ones most needing an archive. A capture
436
+ #: that fails should be *reported* as failed, which is what this module does.
437
+ #:
438
+ #: Deliberately NOT set, and why:
439
+ #: ``email_result`` / ``wacz`` — archive.org emails the account owner, once
440
+ #: per capture. A run submits many URLs; enabling either turns a batch job
441
+ #: into an inbox full of mail nobody asked for.
442
+ #: ``wm-save-mywebarchive`` — writes to the operator's personal archive. Their
443
+ #: account, their choice, not a side effect of calling this library.
444
+ #: ``capture_outlinks`` — every outlink of every submitted URL is an enormous
445
+ #: load increase on a service that already throttles callers, for pages
446
+ #: nothing asked to capture.
447
+ #: ``capture_screenshot`` — a second form of evidence, and a real option worth
448
+ #: revisiting, but this package renders nothing from it today.
449
+ #: ``force_get`` — trades the headless browser for a plain GET, which captures
450
+ #: JavaScript-rendered pages worse. The point is a faithful copy.
451
+ _CAPTURE_OPTIONS = {"capture_all": "0"}
452
+
453
+
454
+ def _capture_params(url, stale_days=None):
455
+ """Form fields for one authenticated capture request.
456
+
457
+ ``if_not_archived_within`` is deliberately NOT sent, and it is worth saying
458
+ why because it looks like an obvious fit. It asks archive.org to skip the
459
+ capture when a snapshot newer than N already exists — seemingly the same
460
+ idea as a caller's own staleness policy.
461
+
462
+ Tried against the live API 2026-09-06. Sending ``if_not_archived_within``
463
+ for a page with a recent snapshot returns::
464
+
465
+ {"url": "...", "job_id": null,
466
+ "message": "The same snapshot had been made 177 hours, 9 minutes ago.
467
+ You can make new capture of this URL after 4320 hours."}
468
+
469
+ while the identical request without it captures normally. So the skip is
470
+ caused entirely by the parameter — archive.org imposes no such restriction
471
+ of its own — and it leaves the caller with no job id and no snapshot URL,
472
+ i.e. less information than before.
473
+
474
+ More importantly it is a *second gate on the same decision*. This module
475
+ only submits a URL when ``check()`` already said it has no snapshot or a
476
+ stale one; asking archive.org to independently re-decide that can only
477
+ produce disagreement between two rules meant to answer one question.
478
+
479
+ ``stale_days`` is accepted so callers need not care which options apply.
480
+ """
481
+ return {"url": url, **_CAPTURE_OPTIONS}
482
+
483
+
484
+ def submit(url, timeout=30, access_key=None, secret_key=None, stale_days=None):
485
+ """Request that archive.org capture and archive ``url`` (Save Page Now / SPN2).
486
+
487
+ Returns what was *established*, not merely what was asked for. The two
488
+ paths establish different amounts, and the difference is not a detail:
489
+
490
+ **Unauthenticated** (``GET /save/<url>``, no credentials): archive.org runs
491
+ the capture inline and answers with a ``302`` to the resulting snapshot.
492
+ Measured 2026-09-05 against a real URL: one redirect hop to a snapshot with
493
+ a timestamp minted seconds earlier. So on this path the snapshot URL is in
494
+ hand *before this function returns* — no job id, no polling, nothing to
495
+ wait for. A naive implementation that follows the redirect and discards
496
+ the final URL, reporting only ``submitted: True``, throws away an answer
497
+ archive.org has already given.
498
+
499
+ **Authenticated** (``POST /save`` with an S3-style key pair from
500
+ https://archive.org/account/s3.php): the capture is queued and the response
501
+ carries a ``job_id`` instead of a snapshot. That id is the only handle on
502
+ the outcome, and ``check_job_status`` is the only thing that can read it.
503
+
504
+ Either way this does not block waiting for a capture — callers decide who
505
+ waits, how long, and why (see ``check_job_status``).
506
+
507
+ Returns a dict with:
508
+ url str
509
+ submitted bool — archive.org accepted the request
510
+ job_id str | None — SPN2 job id (authenticated submission only)
511
+ archived bool — set True only when a snapshot URL came back
512
+ snapshot_url str — present only when ``archived``
513
+ snapshot_ts str — raw Wayback timestamp (YYYYMMDDHHMMSS)
514
+ snapshot_age_days int | None
515
+ snapshot_stale bool — a redirect can land on a pre-existing
516
+ snapshot rather than a fresh capture, so
517
+ freshness is measured, never assumed
518
+ outcome_unknown bool — set with ``error`` when the request went
519
+ out and we stopped listening before an
520
+ answer came back. Not the same fact as a
521
+ refusal: archive.org may well have run the
522
+ capture anyway, so this must not be
523
+ reported as a failed submission.
524
+ error str — set only on failure
525
+ """
526
+ # The breaker exists because archive.org throttles per IP across endpoints.
527
+ # A naive submission path ignores it: once five lookups had been refused,
528
+ # the availability API goes quiet while Save Page Now — the *more*
529
+ # expensive call, since it starts a real capture — keeps firing at full
530
+ # pace. Honouring it here is reuse of the one scheme, not a second one.
531
+ if rate_limited_out():
532
+ return {
533
+ "url": url,
534
+ "submitted": False,
535
+ "job_id": None,
536
+ "archived": False,
537
+ "error": (
538
+ "skipped: archive.org rate limit tripped earlier this run "
539
+ "(no capture requested)"
540
+ ),
541
+ "error_summary": (
542
+ "not requested — archive.org had already rate-limited this run"
543
+ ),
544
+ "rate_limited": True,
545
+ }
546
+
547
+ headers = dict(DEFAULT_HEADERS)
548
+ try:
549
+ if access_key and secret_key:
550
+ headers["Authorization"] = f"LOW {access_key}:{secret_key}"
551
+ headers["Accept"] = "application/json"
552
+ resp = requests.post(
553
+ _SAVE_API,
554
+ data=_capture_params(url, stale_days),
555
+ headers=headers,
556
+ timeout=timeout,
557
+ )
558
+ resp.raise_for_status()
559
+ payload = resp.json()
560
+ result = {
561
+ "url": url,
562
+ "submitted": True,
563
+ "job_id": payload.get("job_id"),
564
+ "archived": False,
565
+ }
566
+ if not result["job_id"]:
567
+ # Accepted, but no capture started. archive.org explains itself
568
+ # in ``message`` (e.g. "The same snapshot had been made 177
569
+ # hours ago"). Carrying that through is the difference between
570
+ # the caller knowing why nothing happened and knowing nothing.
571
+ message = str(payload.get("message") or "").strip()
572
+ result["error_summary"] = (
573
+ f"archive.org accepted the request without starting a "
574
+ f"capture: {message}"
575
+ if message
576
+ else "archive.org accepted the request but started no capture"
577
+ )
578
+ return result
579
+
580
+ resp = requests.get(
581
+ f"{_SAVE_API}/{url}",
582
+ headers=headers,
583
+ timeout=timeout,
584
+ )
585
+ resp.raise_for_status()
586
+ result = {"url": url, "submitted": True, "job_id": None, "archived": False}
587
+ # The capture archive.org just ran, named by the URL it redirected us to.
588
+ m = _ARCHIVE_URL_RE.search(resp.url or "")
589
+ if m:
590
+ result["archived"] = True
591
+ result.update(_snapshot_state(resp.url, m.group(1), stale_days))
592
+ return result
593
+ except Exception as exc:
594
+ err = redact.redact_value(redact.redact_url_keys(str(exc)), secret_key)
595
+ # A read timeout is not a refusal. The request reached archive.org and
596
+ # we gave up waiting for the answer; the capture may have run to
597
+ # completion regardless. Observed live 2026-09-05 — a 30s read timeout
598
+ # on a save that archive.org had almost certainly accepted. Calling
599
+ # that "submission failed" overstates it in the other direction, the
600
+ # same way calling it "archived" would.
601
+ timed_out = isinstance(exc, requests.exceptions.Timeout)
602
+ log.warning(
603
+ "Wayback submission %s for %s: %s",
604
+ "timed out" if timed_out else "failed",
605
+ url,
606
+ err,
607
+ )
608
+ result = {
609
+ "url": url,
610
+ "submitted": False,
611
+ "job_id": None,
612
+ "error": err,
613
+ # The sentence a caller is safe to surface to an end user.
614
+ # ``error`` stays raw (redacted) for logging.
615
+ "error_summary": _transport_failure_summary(exc, "when asked to capture"),
616
+ }
617
+ if timed_out:
618
+ result["outcome_unknown"] = True
619
+ return result
620
+
621
+
622
+ def check_job_status(
623
+ job_id, timeout=15, access_key=None, secret_key=None, stale_days=None
624
+ ):
625
+ """Read the outcome of an SPN2 capture job. Never raises.
626
+
627
+ **This endpoint is credential-only.** Probed 2026-09-05: both
628
+ ``GET /save/status/<job_id>`` and the bare ``GET /save/status`` answer
629
+ ``401 {"message": "You need to be logged in to use Save Page Now."}``
630
+ without an ``Authorization`` header, and answer the same to a well-formed
631
+ but wrong ``LOW`` credential. There is therefore no unauthenticated way to
632
+ find out how a capture went, which is exactly why the unauthenticated
633
+ submission path reads its answer off the redirect instead.
634
+
635
+ Goes through the same ``_paced_get`` as the availability lookup, so it
636
+ shares one pacing clock, one backoff and one breaker budget with every
637
+ other archive.org call in the process — per this module's rate-limit
638
+ design, which exists because archive.org throttles per IP, not per
639
+ endpoint.
640
+
641
+ Returns a dict with:
642
+ job_id str
643
+ state str — one of:
644
+ "success" capture completed; snapshot fields set
645
+ "pending" archive.org is still working on it
646
+ "failed" archive.org tried and could not
647
+ "not_checked" we did not ask (no creds, breaker
648
+ tripped) — asserts nothing either way
649
+ "unknown" we asked and could not read the answer
650
+ reason str | None — why, for every state except "success"
651
+ snapshot_url str — present only on "success"
652
+ snapshot_ts str
653
+ snapshot_age_days int | None
654
+ snapshot_stale bool
655
+ """
656
+ if not job_id:
657
+ return {
658
+ "job_id": job_id,
659
+ "state": "not_checked",
660
+ "reason": "no SPN2 job id was recorded for this submission",
661
+ }
662
+ if not (access_key and secret_key):
663
+ return {
664
+ "job_id": job_id,
665
+ "state": "not_checked",
666
+ "reason": (
667
+ "archive.org's job-status endpoint requires credentials (it "
668
+ "answers 401 without them) — pass access_key/secret_key to "
669
+ "have capture outcomes verified"
670
+ ),
671
+ }
672
+ # Same reasoning as check(): once archive.org has refused repeatedly, every
673
+ # further call costs the run seconds of pacing to collect another 429.
674
+ if rate_limited_out():
675
+ return {
676
+ "job_id": job_id,
677
+ "state": "not_checked",
678
+ "reason": (
679
+ "skipped: archive.org rate limit tripped earlier this run "
680
+ "(no job-status lookup attempted)"
681
+ ),
682
+ }
683
+
684
+ headers = dict(DEFAULT_HEADERS)
685
+ headers["Authorization"] = f"LOW {access_key}:{secret_key}"
686
+ headers["Accept"] = "application/json"
687
+ try:
688
+ resp = _paced_get(f"{_SAVE_STATUS_API}/{job_id}", timeout, headers=headers)
689
+ except Exception as exc:
690
+ err = redact.redact_value(redact.redact_url_keys(str(exc)), secret_key)
691
+ status_code = getattr(getattr(exc, "response", None), "status_code", None)
692
+ if status_code == 401:
693
+ # Verified shape, not a guess — see this function's docstring.
694
+ err = (
695
+ "archive.org rejected the credentials for the job-status "
696
+ "endpoint (401). The capture outcome is unknown, not failed."
697
+ )
698
+ log.debug("Wayback job status lookup failed for %s: %s", job_id, err)
699
+ if status_code == 401:
700
+ return {"job_id": job_id, "state": "unknown", "reason": err}
701
+ return {
702
+ "job_id": job_id,
703
+ "state": "unknown",
704
+ "reason": _transport_failure_summary(exc, "about this capture"),
705
+ "raw_error": err,
706
+ }
707
+
708
+ try:
709
+ data = resp.json()
710
+ except ValueError:
711
+ # Same misdirection guard as check(): a 200 that isn't JSON is an error
712
+ # or challenge page, and reporting the decoder's complaint sends the
713
+ # reader hunting for a parser bug that isn't there.
714
+ return {
715
+ "job_id": job_id,
716
+ "state": "unknown",
717
+ "reason": (
718
+ f"archive.org returned a non-JSON {resp.status_code} response "
719
+ f"({len(resp.content)} bytes) from the job-status API"
720
+ ),
721
+ }
722
+
723
+ status = str(data.get("status") or "").strip().lower()
724
+ if status == "success":
725
+ ts = str(data.get("timestamp") or "")
726
+ original = str(data.get("original_url") or "")
727
+ if not ts or not original:
728
+ # Reported success without naming what it captured. Don't invent a
729
+ # snapshot URL out of half an answer.
730
+ return {
731
+ "job_id": job_id,
732
+ "state": "unknown",
733
+ "reason": (
734
+ "archive.org reported the capture succeeded but named no "
735
+ "timestamp/original_url, so no snapshot URL can be given"
736
+ ),
737
+ }
738
+ result = {"job_id": job_id, "state": "success", "reason": None}
739
+ result.update(
740
+ _snapshot_state(
741
+ f"https://web.archive.org/web/{ts}/{original}", ts, stale_days
742
+ )
743
+ )
744
+ return result
745
+ if status == "pending":
746
+ return {
747
+ "job_id": job_id,
748
+ "state": "pending",
749
+ "reason": "archive.org has not finished this capture yet",
750
+ }
751
+ if status == "error":
752
+ # SPN2 spreads the explanation over three optional fields; take the most
753
+ # human one present rather than whichever happens to be first.
754
+ detail = (
755
+ data.get("message") or data.get("status_ext") or data.get("exception") or ""
756
+ )
757
+ return {
758
+ "job_id": job_id,
759
+ "state": "failed",
760
+ "reason": str(detail).strip()
761
+ or "archive.org reported an unspecified error",
762
+ }
763
+ return {
764
+ "job_id": job_id,
765
+ "state": "unknown",
766
+ "reason": (
767
+ f"archive.org reported an unrecognized job status {status!r}"
768
+ if status
769
+ else "archive.org's job-status response carried no status field"
770
+ ),
771
+ }
772
+
773
+
774
+ def capture_capacity(timeout=15, access_key=None, secret_key=None):
775
+ """How much Save Page Now capacity this account has right now. Never raises.
776
+
777
+ ``GET /save/status/user``, credential-only like the job-status endpoint.
778
+ Measured live 2026-09-06::
779
+
780
+ {"processing":0,"available":3,"daily_captures":49,"daily_captures_limit":30000}
781
+
782
+ Why this is worth a request: every concurrency number governing submission
783
+ is otherwise invented, picked by watching archive.org get upset, and static
784
+ — it cannot tell a run with three free capture slots from a run with none.
785
+ This endpoint answers the question directly, and it answers it *before*
786
+ the requests are spent rather than after, which is the difference between
787
+ pacing and apologising. The circuit breaker stays exactly where it is;
788
+ this only narrows what a caller attempts in the first place.
789
+
790
+ Deliberately never *raises* the concurrency ceiling a caller might derive
791
+ from it — a reading can only make a run more cautious, so a wrong or stale
792
+ answer cannot make things worse.
793
+
794
+ Returns a dict with:
795
+ available int | None — concurrent capture slots free now
796
+ processing int | None — captures this account has in flight
797
+ daily_captures int | None
798
+ daily_captures_limit int | None
799
+ daily_exhausted bool — quota is used up; submitting is pointless
800
+ known bool — False when we could not find out
801
+ reason str | None — why not, when ``known`` is False
802
+ """
803
+ unknown = {
804
+ "available": None,
805
+ "processing": None,
806
+ "daily_captures": None,
807
+ "daily_captures_limit": None,
808
+ "daily_exhausted": False,
809
+ "known": False,
810
+ }
811
+ if not (access_key and secret_key):
812
+ return {
813
+ **unknown,
814
+ "reason": (
815
+ "archive.org reports capture capacity only to an authenticated "
816
+ "account; pass access_key/secret_key"
817
+ ),
818
+ }
819
+ if rate_limited_out():
820
+ return {
821
+ **unknown,
822
+ "reason": (
823
+ "skipped: archive.org rate limit tripped earlier this run "
824
+ "(no capacity lookup attempted)"
825
+ ),
826
+ }
827
+
828
+ headers = dict(DEFAULT_HEADERS)
829
+ headers["Authorization"] = f"LOW {access_key}:{secret_key}"
830
+ headers["Accept"] = "application/json"
831
+ try:
832
+ resp = _paced_get(_SAVE_USER_STATUS_API, timeout, headers=headers)
833
+ data = resp.json()
834
+ except Exception as exc:
835
+ log.debug("Wayback capture-capacity lookup failed: %s", exc)
836
+ return {
837
+ **unknown,
838
+ "reason": _transport_failure_summary(exc, "for capture capacity"),
839
+ }
840
+
841
+ def _int(key):
842
+ value = data.get(key)
843
+ return value if isinstance(value, int) else None
844
+
845
+ used, limit = _int("daily_captures"), _int("daily_captures_limit")
846
+ return {
847
+ "available": _int("available"),
848
+ "processing": _int("processing"),
849
+ "daily_captures": used,
850
+ "daily_captures_limit": limit,
851
+ # Only assert exhaustion when both numbers are real. "Unknown" must not
852
+ # collapse into "you are out of quota" and stop a run submitting.
853
+ "daily_exhausted": bool(
854
+ used is not None and limit is not None and used >= limit
855
+ ),
856
+ "known": True,
857
+ "reason": None,
858
+ }
859
+
860
+
861
+ def system_status(timeout=15, access_key=None, secret_key=None):
862
+ """Is archive.org's capture system healthy, or are we the problem? Never raises.
863
+
864
+ ``GET /save/status/system``. Measured live 2026-09-06::
865
+
866
+ {"recent_captures":941,"status":"ok","queues":{"spn2-captures":0, ...13 queues}}
867
+
868
+ Why this is worth asking. When submission degrades, everything a caller
869
+ can see looks the same from the inside: a 520, a read timeout, a refused
870
+ connection. Those are equally consistent with "we asked too often" and with
871
+ "the service is having a bad afternoon" — the same misdiagnosis as treating
872
+ a User-Agent block as rate limiting. The fix for one is to back off, and
873
+ the fix for the other is to wait and stop blaming yourself.
874
+
875
+ Meant to be asked once per run and only when something has already gone
876
+ wrong, so a healthy run pays nothing for it.
877
+
878
+ Returns a dict with:
879
+ ok bool | None — archive.org's own health verdict
880
+ status str — the raw status string it reported
881
+ recent_captures int | None
882
+ busiest_queue (name, depth) | None — the deepest non-empty queue
883
+ known bool — False when we could not find out
884
+ reason str | None
885
+ """
886
+ unknown = {
887
+ "ok": None,
888
+ "status": "",
889
+ "recent_captures": None,
890
+ "busiest_queue": None,
891
+ "known": False,
892
+ }
893
+ if rate_limited_out():
894
+ # Deliberately still asks nothing: the breaker exists because further
895
+ # calls cost the run pacing budget, and that applies to diagnosis too.
896
+ return {
897
+ **unknown,
898
+ "reason": (
899
+ "skipped: archive.org rate limit tripped earlier this run "
900
+ "(no service-status lookup attempted)"
901
+ ),
902
+ }
903
+
904
+ headers = dict(DEFAULT_HEADERS)
905
+ headers["Accept"] = "application/json"
906
+ if access_key and secret_key:
907
+ headers["Authorization"] = f"LOW {access_key}:{secret_key}"
908
+ try:
909
+ resp = _paced_get(_SAVE_SYSTEM_STATUS_API, timeout, headers=headers)
910
+ data = resp.json()
911
+ except Exception as exc:
912
+ log.debug("Wayback system-status lookup failed: %s", exc)
913
+ return {
914
+ **unknown,
915
+ "reason": _transport_failure_summary(exc, "for its service status"),
916
+ }
917
+
918
+ status = str(data.get("status") or "").strip()
919
+ queues = data.get("queues")
920
+ busiest = None
921
+ if isinstance(queues, dict):
922
+ depths = [(n, d) for n, d in queues.items() if isinstance(d, int) and d > 0]
923
+ if depths:
924
+ busiest = max(depths, key=lambda kv: kv[1])
925
+ recent = data.get("recent_captures")
926
+ return {
927
+ "ok": status.lower() == "ok",
928
+ "status": status,
929
+ "recent_captures": recent if isinstance(recent, int) else None,
930
+ "busiest_queue": busiest,
931
+ "known": True,
932
+ "reason": None,
933
+ }
934
+
935
+
936
+ def service_health_note(status):
937
+ """One clause naming who was at fault, or None when it adds nothing.
938
+
939
+ Returns None for a healthy service *and* for an unknown one: appending
940
+ "we could not tell" to every failed submission would be noise on top of a
941
+ failure the caller is already looking at.
942
+ """
943
+ if not status or not status.get("known"):
944
+ return None
945
+ if status.get("ok"):
946
+ busiest = status.get("busiest_queue")
947
+ if busiest and busiest[1] >= _BUSY_QUEUE_DEPTH:
948
+ return (
949
+ f"archive.org reported itself healthy but busy at the time "
950
+ f"({busiest[0]} queue {busiest[1]} deep)"
951
+ )
952
+ return "archive.org reported its capture system healthy at the time"
953
+ return (
954
+ f"archive.org reported its capture system as "
955
+ f"{status.get('status') or 'not ok'} at the time — this was the service, "
956
+ f"not the caller"
957
+ )
@@ -0,0 +1,79 @@
1
+ Metadata-Version: 2.5
2
+ Name: spn-client
3
+ Version: 0.1.0
4
+ Summary: A hardened client for archive.org's availability and Save Page Now (SPN2) APIs — process-wide rate pacing with a circuit breaker, dual-mode anonymous/S3-authenticated submission, and an explicit archive-outcome vocabulary.
5
+ Project-URL: Homepage, https://github.com/MHammett/spn-client
6
+ Project-URL: Issues, https://github.com/MHammett/spn-client/issues
7
+ License-Expression: MIT
8
+ License-File: LICENSE
9
+ Requires-Python: >=3.10
10
+ Requires-Dist: requests<3.0,>=2.31.0
11
+ Description-Content-Type: text/markdown
12
+
13
+ # spn-client
14
+
15
+ A hardened Python client for archive.org's availability API and Save Page Now
16
+ (SPN2) — the pieces that are easy to get wrong when you actually run this
17
+ against archive.org at any volume:
18
+
19
+ - **Process-wide rate pacing with a circuit breaker.** archive.org throttles
20
+ per IP, not per thread or per endpoint. A naive per-thread backoff (or none
21
+ at all) looks fine in testing and then silently stops archiving anything
22
+ the first time it's run concurrently or against a busy queue.
23
+ - **Both submission modes.** Anonymous `GET /save/<url>` (capture runs inline,
24
+ answer comes back as a redirect) and S3-key-authenticated `POST /save`
25
+ (capture is queued, answer comes back as a `job_id` you poll for).
26
+ - **An explicit outcome vocabulary.** `submitted` is not `archived`. This
27
+ library only ever reports `archived: True` alongside a real snapshot URL
28
+ that came back from archive.org — never as a synonym for "we asked."
29
+ - **Staleness checking**, so callers can skip re-archiving a URL that already
30
+ has a recent-enough snapshot.
31
+
32
+ Every non-obvious piece of behavior in `client.py` is a documented response to
33
+ a specific, dated, measured incident against the real archive.org API — not
34
+ a guess.
35
+
36
+ ## Install
37
+
38
+ ```bash
39
+ pip install spn-client
40
+ ```
41
+
42
+ ## Usage
43
+
44
+ ```python
45
+ import spn_client
46
+
47
+ result = spn_client.check("https://example.com/some-page")
48
+ if result["archived"] is False or result.get("snapshot_stale"):
49
+ submission = spn_client.submit(
50
+ "https://example.com/some-page",
51
+ access_key=ACCESS_KEY, # optional — omit for anonymous, lower-rate submission
52
+ secret_key=SECRET_KEY,
53
+ )
54
+ if submission["job_id"]:
55
+ # authenticated path: poll for the real outcome
56
+ status = spn_client.check_job_status(
57
+ submission["job_id"], access_key=ACCESS_KEY, secret_key=SECRET_KEY
58
+ )
59
+ ```
60
+
61
+ Call `spn_client.reset_rate_limit_state()` once at the start of each
62
+ independent run/process if you're running this as a long-lived worker —
63
+ the pacing/breaker state is process-wide and intentionally does not reset
64
+ itself, so a breaker tripped by one run would otherwise silently degrade
65
+ the next.
66
+
67
+ ## What this doesn't do
68
+
69
+ This is the archive.org client only — it has no opinion about:
70
+ - What staleness policy is right for your use case (`stale_days` is always a
71
+ caller-supplied parameter)
72
+ - How you track which URLs you've already archived (that's a caller-side
73
+ ledger/cache concern)
74
+ - Batch-loop concerns like a progress heartbeat or a wall-clock budget across
75
+ many submissions — those depend on your own operational needs
76
+
77
+ ## License
78
+
79
+ MIT
@@ -0,0 +1,8 @@
1
+ spn_client/__init__.py,sha256=o7XRJeXD823itejqfcy9iXig-DeKJx_3eSvB1Px76lI,964
2
+ spn_client/_http.py,sha256=Hzb5Qy56e-g3kvzmTt5ngTEgUs3xlQFARUg5zt_pz5I,706
3
+ spn_client/_redact.py,sha256=ZGoN5V8_Z5m0sW__h2kmj_AVelX5x-zlHc9WSKFmN-U,2125
4
+ spn_client/client.py,sha256=7PZLVsnuK97NHq66PTDzEkhQGEibXRLCCsh6RYJNRo4,41913
5
+ spn_client-0.1.0.dist-info/METADATA,sha256=7QPQ2YcuvlKpIQPiol5isTkY_YoSrQC56diICsnlgsM,3178
6
+ spn_client-0.1.0.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
7
+ spn_client-0.1.0.dist-info/licenses/LICENSE,sha256=e2rCCsFHz2bduyCacklOEqV5DICqfjL9jX6gzpA1PY4,1069
8
+ spn_client-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.32.0
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Mike Hammett
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.