libex-core 0.20.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- libex_core/CHANGELOG.md +379 -0
- libex_core/__init__.py +41 -0
- libex_core/__main__.py +10 -0
- libex_core/asin.py +38 -0
- libex_core/audible/__init__.py +12 -0
- libex_core/audible/_concurrency.py +265 -0
- libex_core/audible/_retry.py +84 -0
- libex_core/audible/authors/__init__.py +14 -0
- libex_core/audible/authors/by_name.py +352 -0
- libex_core/audible/authors/catalog.py +1211 -0
- libex_core/audible/authors/profile.py +106 -0
- libex_core/audible/authors/screens.py +879 -0
- libex_core/audible/books.py +869 -0
- libex_core/audible/chapters.py +261 -0
- libex_core/audible/client.py +899 -0
- libex_core/audible/extras.py +321 -0
- libex_core/audible/releases.py +357 -0
- libex_core/audible/search.py +139 -0
- libex_core/audible/series.py +168 -0
- libex_core/cli/__init__.py +7 -0
- libex_core/cli/_args.py +140 -0
- libex_core/cli/_db_args.py +126 -0
- libex_core/cli/_render.py +618 -0
- libex_core/cli/_run.py +58 -0
- libex_core/cli/_shaping.py +105 -0
- libex_core/cli/commands/__init__.py +4 -0
- libex_core/cli/commands/abs_search.py +83 -0
- libex_core/cli/commands/author.py +110 -0
- libex_core/cli/commands/book.py +146 -0
- libex_core/cli/commands/completion.py +40 -0
- libex_core/cli/commands/config.py +34 -0
- libex_core/cli/commands/db.py +572 -0
- libex_core/cli/commands/narrator.py +44 -0
- libex_core/cli/commands/releases.py +146 -0
- libex_core/cli/commands/search.py +70 -0
- libex_core/cli/commands/series.py +79 -0
- libex_core/cli/environment.py +151 -0
- libex_core/cli/exit_codes.py +96 -0
- libex_core/cli/main.py +69 -0
- libex_core/cli/output.py +86 -0
- libex_core/cli/parser.py +90 -0
- libex_core/cli/session.py +112 -0
- libex_core/cli/store_state.py +43 -0
- libex_core/exceptions.py +99 -0
- libex_core/log_safety.py +111 -0
- libex_core/lookup/__init__.py +58 -0
- libex_core/lookup/_common.py +7 -0
- libex_core/lookup/_shaping.py +76 -0
- libex_core/lookup/_store.py +497 -0
- libex_core/lookup/author_books.py +473 -0
- libex_core/lookup/authors.py +176 -0
- libex_core/lookup/books.py +571 -0
- libex_core/lookup/releases.py +240 -0
- libex_core/lookup/search.py +390 -0
- libex_core/lookup/series.py +290 -0
- libex_core/models.py +630 -0
- libex_core/py.typed +0 -0
- libex_core/shaping.py +173 -0
- libex_core/storage/__init__.py +68 -0
- libex_core/storage/base.py +14 -0
- libex_core/storage/dialect.py +357 -0
- libex_core/storage/filtering.py +204 -0
- libex_core/storage/merge.py +266 -0
- libex_core/storage/migrations/env.py +65 -0
- libex_core/storage/migrations/script.py.mako +28 -0
- libex_core/storage/migrations/versions/438dbe70d041_create_core_tables.py +258 -0
- libex_core/storage/models.py +505 -0
- libex_core/storage/read/__init__.py +11 -0
- libex_core/storage/read/_compat.py +216 -0
- libex_core/storage/read/books.py +541 -0
- libex_core/storage/read/people.py +379 -0
- libex_core/storage/read/series.py +151 -0
- libex_core/storage/read/shapes.py +271 -0
- libex_core/storage/read/stats.py +72 -0
- libex_core/storage/sorting.py +72 -0
- libex_core/storage/store.py +576 -0
- libex_core/storage/types.py +55 -0
- libex_core/storage/upgrade.py +147 -0
- libex_core/storage/write/__init__.py +41 -0
- libex_core/storage/write/books.py +202 -0
- libex_core/storage/write/entities.py +437 -0
- libex_core/storage/write/params.py +103 -0
- libex_core/storage/write/serialize.py +75 -0
- libex_core/storage/write/statements.py +481 -0
- libex_core/storage/write/support.py +164 -0
- libex_core/text.py +49 -0
- libex_core-0.20.0.data/data/share/bash-completion/completions/libex-core +385 -0
- libex_core-0.20.0.data/data/share/fish/vendor_completions.d/libex-core.fish +447 -0
- libex_core-0.20.0.data/data/share/man/man1/libex-core.1 +1556 -0
- libex_core-0.20.0.data/data/share/zsh/site-functions/_libex-core +699 -0
- libex_core-0.20.0.dist-info/METADATA +215 -0
- libex_core-0.20.0.dist-info/RECORD +95 -0
- libex_core-0.20.0.dist-info/WHEEL +4 -0
- libex_core-0.20.0.dist-info/entry_points.txt +3 -0
- libex_core-0.20.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,265 @@
|
|
|
1
|
+
"""
|
|
2
|
+
The process-wide bound on in-flight Audible requests: two semaphores, the
|
|
3
|
+
pool constants that size them, and the ContextVar that selects between them.
|
|
4
|
+
They bound what one process does to one shared exit IP, which is why they
|
|
5
|
+
live here rather than on LibexClient -- every client instance in a process
|
|
6
|
+
draws from the same two.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
# Standard library
|
|
10
|
+
import asyncio
|
|
11
|
+
from collections.abc import Iterator
|
|
12
|
+
from contextlib import contextmanager
|
|
13
|
+
from contextvars import ContextVar
|
|
14
|
+
|
|
15
|
+
# ============================================================
|
|
16
|
+
# CONCURRENCY BOUND
|
|
17
|
+
# ============================================================
|
|
18
|
+
|
|
19
|
+
# Every fan-out in this app sets its own per-walk concurrency constant
|
|
20
|
+
# (SCREENS_FANOUT_CONCURRENCY in authors/screens.py), but those only bound
|
|
21
|
+
# one walk at a time -- two simultaneous requests for a
|
|
22
|
+
# large author already double the in-flight count, and nothing upstream of
|
|
23
|
+
# this module caps the total across every walk running at once.
|
|
24
|
+
# LibexClient.get is the one place every outbound Audible call passes
|
|
25
|
+
# through, so the bound lives here, process-wide, instead of at any
|
|
26
|
+
# individual call site: a per-call-site limit only expresses how eagerly
|
|
27
|
+
# that one walk wants to go, never what one event loop and one exit IP
|
|
28
|
+
# carry across all of them at once.
|
|
29
|
+
#
|
|
30
|
+
# This is the pool every call uses by default. That includes the seeder's
|
|
31
|
+
# continuous, unattended background work -- the workload that once got
|
|
32
|
+
# Libex's exit IP throttled into a VPN rotation, since it runs sustained and
|
|
33
|
+
# unsupervised for as long as the process is up -- and it also includes
|
|
34
|
+
# every bulk route: GET /books takes up to 1000 ASINs (books/router.py),
|
|
35
|
+
# hydrating them as 20 concurrent 50-ASIN chunks, and /author/books?name=,
|
|
36
|
+
# /series/{asin} and /search fan out the same way. Only the author-ASIN
|
|
37
|
+
# path opts out, into the wider pool below.
|
|
38
|
+
#
|
|
39
|
+
# 10 IS A PER-PROCESS FAN-OUT WIDTH, not a share of a deployment-wide
|
|
40
|
+
# budget, and that distinction is why it is a flat literal. The number has
|
|
41
|
+
# two unrelated jobs -- it is a slice of what a shared exit IP tolerates at
|
|
42
|
+
# once, which is divisible, and it is the width one live request's own
|
|
43
|
+
# chunked fan-out passes through, which is not. Sizing it on the first job
|
|
44
|
+
# alone breaks the second: a 1000-ASIN /books call is 20 concurrent calls
|
|
45
|
+
# into this client, two rounds at 10 permits and ten rounds at 2, against
|
|
46
|
+
# a fronting proxy that times out at 30s.
|
|
47
|
+
#
|
|
48
|
+
# So dividing this constant by the uvicorn worker count is not available as
|
|
49
|
+
# a way to hold a budget. It is the same defect as the outage documented
|
|
50
|
+
# under AUDIBLE_AUTHOR_BOOKS_CONCURRENCY_LIMIT below -- a gate narrower
|
|
51
|
+
# than a single request's own fan-out, serialising that fan-out into rounds
|
|
52
|
+
# until it 504s at the proxy -- and it lands harder: those 504s were
|
|
53
|
+
# measured on a 10-wide gate, while a divided figure is 2 at four workers
|
|
54
|
+
# and 1 at six.
|
|
55
|
+
#
|
|
56
|
+
# The deployment-wide total is therefore WEB_CONCURRENCY x (10 + 25), both
|
|
57
|
+
# pools being per-process; at six workers, 210. A concurrency ladder run
|
|
58
|
+
# against Audible at 10, 25, 50, 100, 125, 150, 175, 200 and 250 in flight
|
|
59
|
+
# returned 1122 requests and 1122 HTTP 200 -- zero 429, zero 5xx, zero
|
|
60
|
+
# transport errors. No ceiling was found: the run stopped at its own
|
|
61
|
+
# request budget, not at any signal from Audible, and an uncontended canary
|
|
62
|
+
# fired before and after every burst never trended, so the path was in the
|
|
63
|
+
# same state after 250 in flight as before 10. 210 sits at 84% of the
|
|
64
|
+
# largest figure measured clean.
|
|
65
|
+
#
|
|
66
|
+
# THE CAVEAT THAT LIMITS WHAT THAT BUYS: the ladder ran on the DIRECT path.
|
|
67
|
+
# Production reaches Audible through a VPN proxy, whose exit IP is
|
|
68
|
+
# shared with strangers and whose own headroom is unmeasured. 250 is a real
|
|
69
|
+
# number on a real path -- just not the path that runs.
|
|
70
|
+
#
|
|
71
|
+
# The latency rise the ladder did show inside a burst is Libex's own, not
|
|
72
|
+
# Audible's: at 150 concurrent, process CPU was 81.6% of wall and the event
|
|
73
|
+
# loop sat blocked >50ms for 50.7% of wall, on 0.93 MB/s -- one loop doing
|
|
74
|
+
# TLS decrypt and gunzip. That cost is per event loop and so tracks this
|
|
75
|
+
# constant rather than the worker count; each worker runs its own loop with
|
|
76
|
+
# its own permits and pays its own share.
|
|
77
|
+
AUDIBLE_CONCURRENCY_LIMIT = 10
|
|
78
|
+
|
|
79
|
+
# A second, wider pool reserved for exactly one caller: a live author-books
|
|
80
|
+
# request's own discovery-and-hydration fan-out (screens + catalog walk in
|
|
81
|
+
# authors/, then get_books_by_asins hydrating the result), entered via
|
|
82
|
+
# author_books_concurrency() below. That workload is a fundamentally different
|
|
83
|
+
# shape from the sustained one above -- one user request fires a bounded,
|
|
84
|
+
# self-terminating burst (measured locally: 179 requests for
|
|
85
|
+
# Christie, 651 for Conan Doyle -- several times more than "well under 100"
|
|
86
|
+
# once assumed here, but still capped by the screens plateau and
|
|
87
|
+
# CATALOG_RESULT_CEILING rather than open-ended -- see screens.py and
|
|
88
|
+
# catalog.py) and then stops, driven by real user traffic Libex's own
|
|
89
|
+
# hard-noes already forbid amplifying, not a standing crawl. Reusing
|
|
90
|
+
# AUDIBLE_CONCURRENCY_LIMIT for it was the actual bug behind a live, measured
|
|
91
|
+
# production outage: 5 concurrent author lookups queued behind a shared
|
|
92
|
+
# 10-wide gate all 504'd at the fronting proxy's 30s timeout, and even a
|
|
93
|
+
# single uncontended prolific-author request (Christie) measured at 28.22s
|
|
94
|
+
# wall clock -- inside 2s of that same timeout.
|
|
95
|
+
#
|
|
96
|
+
# Measured directly against Audible (bypassing the production proxy, so
|
|
97
|
+
# these are relative, not absolute, numbers) across Conan Doyle and
|
|
98
|
+
# Christie: 5 concurrent mixed requests ran 12.71s at 10 vs 5.77s at 30,
|
|
99
|
+
# which is exactly the case this pool exists for -- five walks sharing a
|
|
100
|
+
# 10-permit gate queue behind each other. Zero throttled responses were
|
|
101
|
+
# observed at 30, 60, or even 100 in flight, and that reproduces exactly on
|
|
102
|
+
# re-measurement.
|
|
103
|
+
#
|
|
104
|
+
# What does not hold is any claim about where a useful ceiling sits. Two
|
|
105
|
+
# runs of one identical configuration measured 5.537s and 3.992s, so a
|
|
106
|
+
# single walk varies ~40% run to run at a fixed setting -- a 1.545s band
|
|
107
|
+
# that sits entirely inside, and covers ~84% of, the 1.85s spread across 25,
|
|
108
|
+
# 30, 50, 75 and 100 permits (3.69-5.54s). Nothing in that range is
|
|
109
|
+
# distinguishable from noise, and a single-request comparison (5.03s at 10
|
|
110
|
+
# vs 3.56s at 30) differs by 1.47s, narrower than that same 1.545s band, so
|
|
111
|
+
# it does not resolve a ceiling either.
|
|
112
|
+
#
|
|
113
|
+
# 25 therefore stands on cost, not on a measured upstream plateau. New
|
|
114
|
+
# connections opened per walk are at most the permit count, and in practice
|
|
115
|
+
# sit at it (25 permits -> 25, 100 -> 96), so every extra permit is one
|
|
116
|
+
# more handshake on any walk that starts cold -- cheap on a direct path, a
|
|
117
|
+
# full CONNECT+TLS through the production proxy, against a shared IP already
|
|
118
|
+
# throttled once on a different workload. Each also costs event-loop CPU
|
|
119
|
+
# that climbs monotonically for identical work: 0.84s at 25, 1.56s at 50,
|
|
120
|
+
# 3.72s at 100, where the loop sat blocked 3.44s of 5.16s wall. That cost is
|
|
121
|
+
# transport work -- TLS decrypt and gunzip of 50-product pages interleaving
|
|
122
|
+
# across streams -- not parsing, which measured 0.11s decode plus 0.08s
|
|
123
|
+
# normalization throughout. More handshakes and more blocked loop time for a
|
|
124
|
+
# latency gain no measurement here can resolve is a bad trade. None of it
|
|
125
|
+
# was measured through the production proxy, so none of it constrains
|
|
126
|
+
# behaviour on that path.
|
|
127
|
+
#
|
|
128
|
+
# The worker count divides neither pool. What sizes this one is a per-loop
|
|
129
|
+
# constraint -- a single walk's own fan-out has to fit through it -- and
|
|
130
|
+
# dividing it is exactly the change that produced the 504s described above,
|
|
131
|
+
# so it is not available as a way to make room. The same argument, applied
|
|
132
|
+
# to a 1000-ASIN /books call's own fan-out, is what keeps
|
|
133
|
+
# AUDIBLE_CONCURRENCY_LIMIT above a flat literal; see its comment. What
|
|
134
|
+
# bounds this pool instead is the measured throttle-free band: 30, 60 and
|
|
135
|
+
# 100 in flight each drew zero throttled responses here, and the ladder
|
|
136
|
+
# recorded under AUDIBLE_CONCURRENCY_LIMIT reached 250 clean.
|
|
137
|
+
#
|
|
138
|
+
# The worst case counts BOTH pools in every worker, since both are
|
|
139
|
+
# per-process: WEB_CONCURRENCY x (25 + 10), or 210 at six workers, reached
|
|
140
|
+
# only when every worker takes a prolific author in the same moment. 210 is
|
|
141
|
+
# 84% of that ladder's top rung, so the worst case is a point inside the
|
|
142
|
+
# measured band rather than an extrapolation past it. Two things keep that
|
|
143
|
+
# from being reassurance on its own. The ladder ran on the direct path, not
|
|
144
|
+
# through the production proxy whose shared exit IP is what actually
|
|
145
|
+
# carries this traffic, and no run at any width has produced an onset
|
|
146
|
+
# gradient to read a real limit off -- 250 is the largest number tried, not
|
|
147
|
+
# an observed edge.
|
|
148
|
+
#
|
|
149
|
+
# What carries the worker count regardless is that 210 is a burst and the
|
|
150
|
+
# incident was not. The VPN rotation came from the seeder's sustained,
|
|
151
|
+
# unattended crawl, and the seeder runs on the default pool alone at its
|
|
152
|
+
# own steady 10 per process; the peak needs a prolific-author walk in every
|
|
153
|
+
# worker at once to appear at all. Since neither pool divides, the worker
|
|
154
|
+
# count is the only dial that moves that peak, which is what makes it the
|
|
155
|
+
# figure to weigh against a shared exit IP -- not either constant on its
|
|
156
|
+
# own.
|
|
157
|
+
#
|
|
158
|
+
# Halving this constant to 12-13 to buy room back lands in the regime the
|
|
159
|
+
# 504s came from, a gate narrower than a single walk's own fan-out, and
|
|
160
|
+
# pushes more walks past AUTHOR_BOOKS_TIME_BUDGET_SECONDS into
|
|
161
|
+
# truncated_by_deadline, which caches degraded for 900s and so brings those
|
|
162
|
+
# authors back for a re-walk sooner -- more outbound requests from the
|
|
163
|
+
# narrower gate, not fewer.
|
|
164
|
+
AUDIBLE_AUTHOR_BOOKS_CONCURRENCY_LIMIT = 25
|
|
165
|
+
|
|
166
|
+
# asyncio.Semaphore binds its internal waiter state to whichever event loop
|
|
167
|
+
# is running the first time it's touched. Under uvicorn that's one long-lived
|
|
168
|
+
# loop, but the test suite creates and tears down a fresh loop per test, and
|
|
169
|
+
# a semaphore built once at import time and reused across those loops risks
|
|
170
|
+
# waiters left over from a closed loop. Keying the instance to the current
|
|
171
|
+
# running loop and rebuilding it whenever that loop changes sidesteps this:
|
|
172
|
+
# a long-lived process creates it exactly once and keeps reusing it, and each
|
|
173
|
+
# fresh test loop gets its own fresh semaphore instead of inheriting stale
|
|
174
|
+
# state from whatever loop ran before it.
|
|
175
|
+
#
|
|
176
|
+
# These two semaphores, and the pool constants above them, stay process-wide
|
|
177
|
+
# module state rather than becoming LibexClient instance state: they bound
|
|
178
|
+
# what one process does to one shared exit IP, not what one client object
|
|
179
|
+
# does. Every LibexClient instance built in the same process draws from the
|
|
180
|
+
# same two semaphores below regardless of how many instances exist -- making
|
|
181
|
+
# them instance state would let two instances double the sustained fan-out
|
|
182
|
+
# a single exit IP sees, which is the exact failure that cost a VPN rotation
|
|
183
|
+
# once already.
|
|
184
|
+
_audible_semaphore: asyncio.Semaphore | None = None
|
|
185
|
+
_audible_semaphore_loop: asyncio.AbstractEventLoop | None = None
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def _get_audible_semaphore() -> asyncio.Semaphore:
|
|
189
|
+
global _audible_semaphore, _audible_semaphore_loop
|
|
190
|
+
loop = asyncio.get_running_loop()
|
|
191
|
+
if _audible_semaphore is None or _audible_semaphore_loop is not loop:
|
|
192
|
+
_audible_semaphore = asyncio.Semaphore(AUDIBLE_CONCURRENCY_LIMIT)
|
|
193
|
+
_audible_semaphore_loop = loop
|
|
194
|
+
return _audible_semaphore
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
# Same per-loop-rebuild reasoning as _get_audible_semaphore above, kept as a
|
|
198
|
+
# fully separate instance rather than a dict keyed by pool name: two pools
|
|
199
|
+
# only, and a separate pair of module globals means the existing default-pool
|
|
200
|
+
# tests (which poke _audible_semaphore / _audible_semaphore_loop directly)
|
|
201
|
+
# stay exactly as they are, untouched by this pool's own lifecycle.
|
|
202
|
+
_audible_author_books_semaphore: asyncio.Semaphore | None = None
|
|
203
|
+
_audible_author_books_semaphore_loop: asyncio.AbstractEventLoop | None = None
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
def _get_audible_author_books_semaphore() -> asyncio.Semaphore:
|
|
207
|
+
global _audible_author_books_semaphore, _audible_author_books_semaphore_loop
|
|
208
|
+
loop = asyncio.get_running_loop()
|
|
209
|
+
if (
|
|
210
|
+
_audible_author_books_semaphore is None
|
|
211
|
+
or _audible_author_books_semaphore_loop is not loop
|
|
212
|
+
):
|
|
213
|
+
_audible_author_books_semaphore = asyncio.Semaphore(AUDIBLE_AUTHOR_BOOKS_CONCURRENCY_LIMIT)
|
|
214
|
+
_audible_author_books_semaphore_loop = loop
|
|
215
|
+
return _audible_author_books_semaphore
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
# Selects which pool LibexClient.get acquires from, without adding a
|
|
219
|
+
# parameter to get() itself, to the audible_get delegator every call site
|
|
220
|
+
# actually calls, or to any of the call sites behind that: a
|
|
221
|
+
# threaded-through pool parameter would have to be plumbed through every
|
|
222
|
+
# intermediate fetch function in screens.py, catalog.py, and books.py,
|
|
223
|
+
# several of which are shared with the seeder and must never pick up the
|
|
224
|
+
# wider pool, and several existing tests patch audible_get at each of those
|
|
225
|
+
# consuming modules -- app.services.audible.books.audible_get and its
|
|
226
|
+
# siblings -- with narrow, fixed-arity stand-ins that a new always-passed
|
|
227
|
+
# kwarg would break outright. A ContextVar sidesteps both: it's invisible
|
|
228
|
+
# to every call site (none of them change), asyncio.gather's own tasks
|
|
229
|
+
# inherit whichever value was current when gather() created them
|
|
230
|
+
# (contextvars.copy_context() happens at task creation), and it flows
|
|
231
|
+
# unmodified through every further nested await and nested gather inside
|
|
232
|
+
# that task -- which is exactly why wrapping only the single outer gather
|
|
233
|
+
# in _walk_author_books and the single hydration gather in
|
|
234
|
+
# get_books_by_asins (see author_books_concurrency's call sites) is enough
|
|
235
|
+
# to cover every one of those calls' eventual descent into LibexClient.get,
|
|
236
|
+
# with nothing else in either module touched.
|
|
237
|
+
_audible_concurrency_pool: ContextVar[str] = ContextVar(
|
|
238
|
+
"_audible_concurrency_pool", default="default"
|
|
239
|
+
)
|
|
240
|
+
|
|
241
|
+
|
|
242
|
+
@contextmanager
|
|
243
|
+
def author_books_concurrency() -> Iterator[None]:
|
|
244
|
+
"""
|
|
245
|
+
Marks every underlying LibexClient.get call made within this block --
|
|
246
|
+
directly or via any further nested await, task, or gather it spawns --
|
|
247
|
+
as belonging to the wider AUDIBLE_AUTHOR_BOOKS_CONCURRENCY_LIMIT pool
|
|
248
|
+
instead of the default AUDIBLE_CONCURRENCY_LIMIT one. Reserved for the
|
|
249
|
+
live author-books discovery and hydration fan-out (see
|
|
250
|
+
AUDIBLE_AUTHOR_BOOKS_CONCURRENCY_LIMIT's own docstring for why that
|
|
251
|
+
workload, and only that one, gets the wider pool); every other caller
|
|
252
|
+
-- single book/author/series lookups, the seeder, chapter backfills --
|
|
253
|
+
never enters this block and stays on the default pool exactly as before.
|
|
254
|
+
"""
|
|
255
|
+
token = _audible_concurrency_pool.set("author_books")
|
|
256
|
+
try:
|
|
257
|
+
yield
|
|
258
|
+
finally:
|
|
259
|
+
_audible_concurrency_pool.reset(token)
|
|
260
|
+
|
|
261
|
+
|
|
262
|
+
def _current_audible_semaphore() -> asyncio.Semaphore:
|
|
263
|
+
if _audible_concurrency_pool.get() == "author_books":
|
|
264
|
+
return _get_audible_author_books_semaphore()
|
|
265
|
+
return _get_audible_semaphore()
|
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
"""
|
|
2
|
+
The retry and back-off policy for Audible requests: which statuses are worth
|
|
3
|
+
retrying, how many attempts, how long to wait, and how a Retry-After header
|
|
4
|
+
is read.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
# Standard library
|
|
8
|
+
import datetime
|
|
9
|
+
import random
|
|
10
|
+
from email.utils import parsedate_to_datetime
|
|
11
|
+
|
|
12
|
+
# ============================================================
|
|
13
|
+
# RETRY / BACKOFF
|
|
14
|
+
# ============================================================
|
|
15
|
+
|
|
16
|
+
# 429 and 5xx are the only responses worth retrying: they mean Audible (or
|
|
17
|
+
# its edge) is asking for a retry, not answering the request. A 404 is a
|
|
18
|
+
# real, permanent answer -- this database carries ~84k ISBN-keyed records
|
|
19
|
+
# that 404 for chapters in every region, and retrying those would just burn
|
|
20
|
+
# requests against the same already-throttled-once IP for nothing. Every
|
|
21
|
+
# other 4xx is a real answer too and is left alone the same way.
|
|
22
|
+
_RETRYABLE_STATUS_CODES = {429}
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def _is_retryable_status(status_code: int) -> bool:
|
|
26
|
+
return status_code in _RETRYABLE_STATUS_CODES or 500 <= status_code < 600
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
# Kept small on purpose. The hosted service's author-books time budget caps
|
|
30
|
+
# a whole discovery walk's wall-clock time, but that deadline is checked by
|
|
31
|
+
# the callers between requests -- it never reaches LibexClient.get, since
|
|
32
|
+
# this method's signature (region, path, params, extra_headers) doesn't
|
|
33
|
+
# carry one. A wide fan-out (author-books discovery alone can fire ~60 concurrent
|
|
34
|
+
# requests) turning every throttled response into several extra seconds
|
|
35
|
+
# would eat into that budget fast with no way for this module to know it's
|
|
36
|
+
# happening, so attempts and backoff both stay deliberately small rather
|
|
37
|
+
# than aggressive. Making retries budget-aware would need an explicit
|
|
38
|
+
# optional deadline parameter threaded from the author-books walks' existing
|
|
39
|
+
# deadline value down through every intermediate call into LibexClient.get itself --
|
|
40
|
+
# that's a real signature change and out of scope here.
|
|
41
|
+
AUDIBLE_MAX_ATTEMPTS = 3
|
|
42
|
+
AUDIBLE_RETRY_BASE_SECONDS = 0.5
|
|
43
|
+
AUDIBLE_RETRY_MAX_BACKOFF_SECONDS = 8.0
|
|
44
|
+
# Retry-After is Audible telling us exactly how long it wants us to wait.
|
|
45
|
+
# Honoring it is the point, but it's still capped so one large value can't
|
|
46
|
+
# stall a fan-out far past what a few retries should ever cost.
|
|
47
|
+
AUDIBLE_RETRY_AFTER_CAP_SECONDS = 10.0
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _parse_retry_after(value: str | None) -> float | None:
|
|
51
|
+
"""Parses a Retry-After header, which per spec is either a number of
|
|
52
|
+
seconds or an HTTP-date. Returns None on anything unparseable so the
|
|
53
|
+
caller falls back to computed backoff instead of guessing."""
|
|
54
|
+
if not value:
|
|
55
|
+
return None
|
|
56
|
+
value = value.strip()
|
|
57
|
+
try:
|
|
58
|
+
return max(0.0, float(value))
|
|
59
|
+
except ValueError:
|
|
60
|
+
pass
|
|
61
|
+
try:
|
|
62
|
+
retry_at = parsedate_to_datetime(value)
|
|
63
|
+
except (TypeError, ValueError, OverflowError):
|
|
64
|
+
return None
|
|
65
|
+
if retry_at is None:
|
|
66
|
+
return None
|
|
67
|
+
if retry_at.tzinfo is None:
|
|
68
|
+
retry_at = retry_at.replace(tzinfo=datetime.timezone.utc)
|
|
69
|
+
now = datetime.datetime.now(datetime.timezone.utc)
|
|
70
|
+
return max(0.0, (retry_at - now).total_seconds())
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _compute_backoff_seconds(attempt: int, retry_after: float | None) -> float:
|
|
74
|
+
"""attempt is the zero-indexed count of attempts already made. Retry-After,
|
|
75
|
+
when present, wins outright (capped); otherwise full-jitter exponential
|
|
76
|
+
backoff, so a burst of concurrent callers hitting the same throttle
|
|
77
|
+
don't all retry in lockstep."""
|
|
78
|
+
if retry_after is not None:
|
|
79
|
+
return min(retry_after, AUDIBLE_RETRY_AFTER_CAP_SECONDS)
|
|
80
|
+
ceiling = min(
|
|
81
|
+
AUDIBLE_RETRY_MAX_BACKOFF_SECONDS,
|
|
82
|
+
AUDIBLE_RETRY_BASE_SECONDS * (2 ** attempt),
|
|
83
|
+
)
|
|
84
|
+
return random.uniform(0, ceiling)
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Looking authors up on Audible: the profile, a search by name, and the books an
|
|
3
|
+
author is credited with.
|
|
4
|
+
|
|
5
|
+
profile.py fetches and normalizes an author's contributor record and asks
|
|
6
|
+
Audible's search suggestions which authors a name resolves to. The three
|
|
7
|
+
author-books walks each read a different Audible source, and none of them is
|
|
8
|
+
complete alone: screens.py reads the Android author-detail screen (the only
|
|
9
|
+
ASIN-exact source), catalog.py runs the windowed, category-sliced catalog
|
|
10
|
+
search attributed by author ASIN, and by_name.py runs the single-sort,
|
|
11
|
+
name-only catalog search for a caller that has a name and no ASIN. Each
|
|
12
|
+
returns what it found together with whether its walk reached a confirmed end;
|
|
13
|
+
deciding what to do with a walk that did not is the caller's.
|
|
14
|
+
"""
|