sourcelock 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- hc_source/__init__.py +5 -0
- hc_source/adapters/__init__.py +500 -0
- hc_source/adapters/_demo.py +258 -0
- hc_source/adapters/_demo_fixture.json +25 -0
- hc_source/adapters/_leie_sample.csv +15 -0
- hc_source/adapters/codes.py +1232 -0
- hc_source/adapters/coverage.py +1569 -0
- hc_source/adapters/hcc.py +1450 -0
- hc_source/adapters/leie.py +1310 -0
- hc_source/adapters/provider.py +1159 -0
- hc_source/cache.py +664 -0
- hc_source/cli.py +959 -0
- hc_source/cli_manifest.py +207 -0
- hc_source/data/codes/hcpcs_2026q3.csv.gz +0 -0
- hc_source/data/codes/icd10cm_fy2026.csv.gz +0 -0
- hc_source/data/codes/icd10cm_fy2027.csv.gz +0 -0
- hc_source/data/codes/manifest.json +75 -0
- hc_source/data/codes/regenerate.py +291 -0
- hc_source/data/hcc/hcc_data.json.zlib +0 -0
- hc_source/doctor.py +472 -0
- hc_source/guard.py +877 -0
- hc_source/http.py +541 -0
- hc_source/interfaces.py +395 -0
- hc_source/lockfile.py +236 -0
- hc_source/manifest.py +422 -0
- hc_source/mcp_server.py +203 -0
- hc_source/npi.py +50 -0
- hc_source/receipts.py +74 -0
- hc_source/schemas.py +339 -0
- sourcelock-0.1.0.dist-info/METADATA +272 -0
- sourcelock-0.1.0.dist-info/RECORD +34 -0
- sourcelock-0.1.0.dist-info/WHEEL +4 -0
- sourcelock-0.1.0.dist-info/entry_points.txt +2 -0
- sourcelock-0.1.0.dist-info/licenses/LICENSE +21 -0
hc_source/http.py
ADDED
|
@@ -0,0 +1,541 @@
|
|
|
1
|
+
"""The one transport every adapter uses.
|
|
2
|
+
|
|
3
|
+
Adapters do not create their own httpx clients. Going through :func:`fetch`
|
|
4
|
+
buys three things the receipt depends on: the SHA-256 of the exact bytes an
|
|
5
|
+
answer was derived from, a uniform authority/fallback policy, and the guarantee
|
|
6
|
+
that no response body is ever logged.
|
|
7
|
+
|
|
8
|
+
``file://`` URLs are supported so that fixtures and locally-mirrored releases
|
|
9
|
+
travel the same code path as live HTTP.
|
|
10
|
+
|
|
11
|
+
Response bodies are STREAMED under a byte ceiling and hashed incrementally.
|
|
12
|
+
Round-1 finding H6: this module read ``response.content`` whole with no cap
|
|
13
|
+
while explicitly asking for gzip, so a 300 MB body at a canary URL took the
|
|
14
|
+
process to 651 MB RSS and a ~200 KB gzip bomb inflated to 200 MB -- about 1000x
|
|
15
|
+
amplification, from the one side an attacker or a broken CDN actually controls.
|
|
16
|
+
The guard had capped INBOUND parameters at 64 KiB since day one and there was
|
|
17
|
+
no equivalent for the response.
|
|
18
|
+
|
|
19
|
+
The default ceiling comes from live measurement of every URL the shipped
|
|
20
|
+
adapters fetch:
|
|
21
|
+
|
|
22
|
+
=========================================== ============
|
|
23
|
+
``current_lcd.zip`` (coverage bulk crosswalk) 31.8 MB
|
|
24
|
+
``UPDATED.csv`` (OIG LEIE full file) 14.8 MB
|
|
25
|
+
local-coverage-articles report 1.02 MB
|
|
26
|
+
every other full-body fetch < 300 KB
|
|
27
|
+
NPPES bulk files never fetched (7 KB index only)
|
|
28
|
+
=========================================== ============
|
|
29
|
+
|
|
30
|
+
So the default is deliberately tight and the two large fetches raise it for
|
|
31
|
+
themselves. That way the two dozen small routes -- including the NPPES canary
|
|
32
|
+
that H6 was demonstrated against -- get a ceiling proportionate to what they
|
|
33
|
+
actually read, instead of one sized for the biggest file in the product.
|
|
34
|
+
"""
|
|
35
|
+
|
|
36
|
+
from __future__ import annotations
|
|
37
|
+
|
|
38
|
+
import contextvars
|
|
39
|
+
import hashlib
|
|
40
|
+
import json
|
|
41
|
+
import os
|
|
42
|
+
import zlib
|
|
43
|
+
from contextlib import contextmanager
|
|
44
|
+
from dataclasses import dataclass, field, replace
|
|
45
|
+
from datetime import datetime, timezone
|
|
46
|
+
from pathlib import Path
|
|
47
|
+
from typing import Any, Iterator, Mapping
|
|
48
|
+
from urllib.parse import unquote, urlparse
|
|
49
|
+
|
|
50
|
+
import httpx
|
|
51
|
+
|
|
52
|
+
from . import cache
|
|
53
|
+
|
|
54
|
+
__all__ = [
|
|
55
|
+
"DEFAULT_MAX_BYTES",
|
|
56
|
+
"DEFAULT_TIMEOUT",
|
|
57
|
+
"FetchResult",
|
|
58
|
+
"RETRYABLE_STATUSES",
|
|
59
|
+
"ResponseTooLarge",
|
|
60
|
+
"SourceUnreachable",
|
|
61
|
+
"USER_AGENT",
|
|
62
|
+
"as_cache_hit",
|
|
63
|
+
"fetch",
|
|
64
|
+
"fetch_json",
|
|
65
|
+
"is_transient",
|
|
66
|
+
"request_timeout",
|
|
67
|
+
]
|
|
68
|
+
|
|
69
|
+
DEFAULT_TIMEOUT = 20.0
|
|
70
|
+
USER_AGENT = "sourcelock/0.1 (+https://github.com/sourcelock)"
|
|
71
|
+
|
|
72
|
+
#: Statuses worth trying again. Everything else -- 401, 403, 404, a 200 whose
|
|
73
|
+
#: body was malformed -- returns the same answer on the second attempt and just
|
|
74
|
+
#: spends the build's time getting there. Retrying is doctor's decision, but
|
|
75
|
+
#: WHICH failures are transient is knowledge about the transport, so it lives
|
|
76
|
+
#: here beside the exception that carries the status.
|
|
77
|
+
RETRYABLE_STATUSES = frozenset({408, 425, 429, 500, 502, 503, 504})
|
|
78
|
+
|
|
79
|
+
#: Ambient timeout, set by doctor around one canary's observation. A ContextVar
|
|
80
|
+
#: rather than a module global because it is scoped to a call: a global would
|
|
81
|
+
#: leak the last canary's budget into every tool call that followed it.
|
|
82
|
+
_TIMEOUT_VAR: contextvars.ContextVar[float | None] = contextvars.ContextVar(
|
|
83
|
+
"hc_source_http_timeout", default=None
|
|
84
|
+
)
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
@contextmanager
|
|
88
|
+
def request_timeout(seconds: float | None) -> Iterator[None]:
|
|
89
|
+
"""Apply a per-request timeout to every fetch made inside this block.
|
|
90
|
+
|
|
91
|
+
This is how a canary's declared budget reaches a transport it never passes
|
|
92
|
+
arguments to: the canary's ``observe`` takes no parameters by design, so
|
|
93
|
+
doctor sets the policy around the call instead of threading it through every
|
|
94
|
+
adapter. An explicit ``timeout=`` at a call site still wins -- the coverage
|
|
95
|
+
bulk zip needs 300 seconds and should not inherit a canary's twenty.
|
|
96
|
+
"""
|
|
97
|
+
token = _TIMEOUT_VAR.set(seconds if seconds is not None else _TIMEOUT_VAR.get())
|
|
98
|
+
try:
|
|
99
|
+
yield
|
|
100
|
+
finally:
|
|
101
|
+
_TIMEOUT_VAR.reset(token)
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def _env_float(name: str) -> float | None:
|
|
105
|
+
raw = os.environ.get(name)
|
|
106
|
+
if raw is None:
|
|
107
|
+
return None
|
|
108
|
+
try:
|
|
109
|
+
value = float(raw)
|
|
110
|
+
except ValueError:
|
|
111
|
+
return None
|
|
112
|
+
return value if value > 0 else None
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def _resolve_timeout(explicit: float | None) -> float:
|
|
116
|
+
"""Call site, then ambient policy, then environment, then the default."""
|
|
117
|
+
if explicit is not None:
|
|
118
|
+
return explicit
|
|
119
|
+
ambient = _TIMEOUT_VAR.get()
|
|
120
|
+
if ambient is not None:
|
|
121
|
+
return ambient
|
|
122
|
+
return _env_float("HC_SOURCE_HTTP_TIMEOUT") or DEFAULT_TIMEOUT
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def is_transient(exc: SourceUnreachable) -> bool:
|
|
126
|
+
"""Would trying again plausibly get a different answer?
|
|
127
|
+
|
|
128
|
+
A body over the ceiling is excluded on purpose: it will be over the ceiling
|
|
129
|
+
again, and downloading it twice to find that out is the opposite of cheap.
|
|
130
|
+
A ``file://`` failure is excluded for the same reason -- a local file that
|
|
131
|
+
could not be read will not read differently in 500 milliseconds.
|
|
132
|
+
"""
|
|
133
|
+
if isinstance(exc, ResponseTooLarge):
|
|
134
|
+
return False
|
|
135
|
+
if urlparse(exc.url).scheme == "file":
|
|
136
|
+
return False
|
|
137
|
+
return exc.status is None or exc.status in RETRYABLE_STATUSES
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
#: Ceiling on the DECOMPRESSED size of one response, in bytes. Sized against
|
|
141
|
+
#: the measured payloads in the module docstring: 15x the largest routine fetch
|
|
142
|
+
#: (the 1.02 MB articles report) and comfortably below anything that threatens
|
|
143
|
+
#: a 2-7 GB CI container. The two genuinely large fetches -- the LEIE full file
|
|
144
|
+
#: and the MCD bulk zip -- pass their own ``max_bytes``.
|
|
145
|
+
DEFAULT_MAX_BYTES = 16 * 1024 * 1024
|
|
146
|
+
|
|
147
|
+
#: Chunk size for streaming. Small enough that the ceiling is enforced promptly,
|
|
148
|
+
#: large enough not to make a 30 MB download a syscall storm.
|
|
149
|
+
_CHUNK = 64 * 1024
|
|
150
|
+
|
|
151
|
+
#: Content encodings we will inflate ourselves. Anything else is refused rather
|
|
152
|
+
#: than handed to a decoder we cannot bound.
|
|
153
|
+
_SUPPORTED_ENCODINGS = {"", "identity", "gzip", "x-gzip", "deflate"}
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
class SourceUnreachable(RuntimeError):
|
|
157
|
+
"""The upstream source could not be read.
|
|
158
|
+
|
|
159
|
+
Carries no response body: doctor output and logs stay payload-free.
|
|
160
|
+
"""
|
|
161
|
+
|
|
162
|
+
def __init__(self, url: str, reason: str, *, status: int | None = None) -> None:
|
|
163
|
+
self.url = url
|
|
164
|
+
self.reason = reason
|
|
165
|
+
self.status = status
|
|
166
|
+
super().__init__(f"{url} unreachable: {reason}" + (f" (HTTP {status})" if status else ""))
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
class ResponseTooLarge(SourceUnreachable):
|
|
170
|
+
"""The upstream body exceeded this call's byte ceiling.
|
|
171
|
+
|
|
172
|
+
A subclass of :class:`SourceUnreachable` on purpose. Every caller that
|
|
173
|
+
already handles an unreachable source -- doctor, the CLI, the MCP server --
|
|
174
|
+
handles this correctly without changing, and the classification is honest:
|
|
175
|
+
the source was not read, so this run does not know what it says.
|
|
176
|
+
|
|
177
|
+
Never truncate-and-continue. A hash of the first N bytes with a receipt
|
|
178
|
+
attesting to it is a confident wrong answer, which is the one thing this
|
|
179
|
+
product exists not to produce.
|
|
180
|
+
"""
|
|
181
|
+
|
|
182
|
+
def __init__(self, url: str, limit: int, *, status: int | None = None) -> None:
|
|
183
|
+
self.limit = limit
|
|
184
|
+
super().__init__(
|
|
185
|
+
url,
|
|
186
|
+
f"response exceeded the {limit}-byte ceiling for this call. If this source "
|
|
187
|
+
"legitimately grew, raise max_bytes at the call site; otherwise the upstream "
|
|
188
|
+
"is serving something it should not",
|
|
189
|
+
status=status,
|
|
190
|
+
)
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
def _utcnow() -> datetime:
|
|
194
|
+
return datetime.now(timezone.utc)
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
@dataclass(frozen=True)
|
|
198
|
+
class FetchResult:
|
|
199
|
+
"""Raw upstream bytes plus the provenance a receipt needs."""
|
|
200
|
+
|
|
201
|
+
url: str
|
|
202
|
+
status: int | None
|
|
203
|
+
content: bytes
|
|
204
|
+
sha256: str
|
|
205
|
+
headers: dict[str, str]
|
|
206
|
+
fallback_used: bool = False
|
|
207
|
+
fallback_name: str | None = None
|
|
208
|
+
#: When THESE bytes were read from upstream. Stamped here, at the fetch
|
|
209
|
+
#: boundary, and never re-stamped -- round-1 finding: ``build_receipt`` used
|
|
210
|
+
#: to call ``utcnow()`` when the receipt was assembled, so a route served
|
|
211
|
+
#: out of a process cache produced a receipt claiming it had just been to
|
|
212
|
+
#: CMS. The receipt is the product; a retrieval time it invented is the one
|
|
213
|
+
#: field that must never be a guess. A cached ``FetchResult`` carries the
|
|
214
|
+
#: timestamp of the fetch that created it, because the object is built once.
|
|
215
|
+
retrieved_at: datetime = field(default_factory=_utcnow)
|
|
216
|
+
#: True when these bytes came out of the on-disk cache rather than off the
|
|
217
|
+
#: wire. Never inferred: the receipt carries it, so a reader can tell an
|
|
218
|
+
#: answer computed from a fresh download from one computed from bytes stored
|
|
219
|
+
#: last week and re-confirmed a moment ago.
|
|
220
|
+
cache_hit: bool = False
|
|
221
|
+
#: When upstream last confirmed these bytes are current -- the moment of the
|
|
222
|
+
#: 304 on a cache hit, the moment of the download otherwise. Kept apart from
|
|
223
|
+
#: ``retrieved_at`` because "you still have the right bytes" and "these bytes
|
|
224
|
+
#: are new" are different statements, and a receipt that merges them is
|
|
225
|
+
#: back to inventing a retrieval time.
|
|
226
|
+
revalidated_at: datetime | None = None
|
|
227
|
+
|
|
228
|
+
def json(self) -> Any:
|
|
229
|
+
try:
|
|
230
|
+
return json.loads(self.content)
|
|
231
|
+
except ValueError as exc:
|
|
232
|
+
raise SourceUnreachable(self.url, f"response was not valid JSON: {type(exc).__name__}")
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
#: Sentinel for "leave ``revalidated_at`` alone", distinct from ``None``, which
|
|
236
|
+
#: means "nobody upstream has ever confirmed these bytes".
|
|
237
|
+
_KEEP = object()
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
def as_cache_hit(result: FetchResult, *, revalidated_at: Any = _KEEP) -> FetchResult:
|
|
241
|
+
"""The same bytes, served again out of an adapter's own in-process cache.
|
|
242
|
+
|
|
243
|
+
Adapters keep parsed snapshots in memory -- LEIE's 15 MB exclusion file, the
|
|
244
|
+
coverage bulk crosswalk -- and re-answer from them for minutes at a time.
|
|
245
|
+
Those answers used to carry ``cache_hit=false`` because the flag came
|
|
246
|
+
straight off the original :class:`FetchResult`, so the receipt's prose said
|
|
247
|
+
"served from an in-process cache" while its structured field said the
|
|
248
|
+
opposite. A machine reading the receipt believed the field.
|
|
249
|
+
|
|
250
|
+
``retrieved_at``, ``sha256`` and ``status`` are preserved: they describe the
|
|
251
|
+
bytes, and the bytes did not change. ``revalidated_at`` defaults to whatever
|
|
252
|
+
upstream last confirmed -- pass a timestamp ONLY when upstream just
|
|
253
|
+
confirmed these bytes again, and pass ``None`` when it never has. A reuse
|
|
254
|
+
that contacted nobody must not advance it; that is the difference between
|
|
255
|
+
"still current as of a moment ago" and "still what we downloaded on Tuesday".
|
|
256
|
+
"""
|
|
257
|
+
return replace(
|
|
258
|
+
result,
|
|
259
|
+
cache_hit=True,
|
|
260
|
+
revalidated_at=(
|
|
261
|
+
result.revalidated_at if revalidated_at is _KEEP else revalidated_at
|
|
262
|
+
),
|
|
263
|
+
)
|
|
264
|
+
|
|
265
|
+
|
|
266
|
+
def fetch(
|
|
267
|
+
url: str,
|
|
268
|
+
*,
|
|
269
|
+
fallback_url: str | None = None,
|
|
270
|
+
fallback_name: str | None = None,
|
|
271
|
+
params: Mapping[str, Any] | None = None,
|
|
272
|
+
headers: Mapping[str, str] | None = None,
|
|
273
|
+
timeout: float | None = None,
|
|
274
|
+
raise_for_status: bool = True,
|
|
275
|
+
max_bytes: int = DEFAULT_MAX_BYTES,
|
|
276
|
+
cache_scope: str = "",
|
|
277
|
+
) -> FetchResult:
|
|
278
|
+
"""Read ``url``, falling back to ``fallback_url`` only if the authority fails.
|
|
279
|
+
|
|
280
|
+
Raises :class:`SourceUnreachable` when neither location can be read, when
|
|
281
|
+
the status is not 2xx and ``raise_for_status`` is set, or -- as
|
|
282
|
+
:class:`ResponseTooLarge` -- when the body exceeds ``max_bytes`` decompressed.
|
|
283
|
+
Raise ``max_bytes`` at the call site for a route that legitimately reads a
|
|
284
|
+
large file; do not raise the default for one route's sake.
|
|
285
|
+
|
|
286
|
+
``timeout`` defaults to the ambient policy (see :func:`request_timeout`),
|
|
287
|
+
then to ``HC_SOURCE_HTTP_TIMEOUT``, then to :data:`DEFAULT_TIMEOUT`. An
|
|
288
|
+
explicit argument always wins.
|
|
289
|
+
"""
|
|
290
|
+
timeout = _resolve_timeout(timeout)
|
|
291
|
+
try:
|
|
292
|
+
return _read(url, params=params, headers=headers, timeout=timeout,
|
|
293
|
+
raise_for_status=raise_for_status, max_bytes=max_bytes,
|
|
294
|
+
cache_scope=cache_scope)
|
|
295
|
+
except SourceUnreachable as primary:
|
|
296
|
+
if not fallback_url:
|
|
297
|
+
raise
|
|
298
|
+
try:
|
|
299
|
+
result = _read(fallback_url, params=params, headers=headers, timeout=timeout,
|
|
300
|
+
raise_for_status=raise_for_status, max_bytes=max_bytes,
|
|
301
|
+
cache_scope=cache_scope)
|
|
302
|
+
except SourceUnreachable as secondary:
|
|
303
|
+
raise SourceUnreachable(
|
|
304
|
+
url,
|
|
305
|
+
f"authority failed ({primary.reason}) and fallback failed ({secondary.reason})",
|
|
306
|
+
status=primary.status,
|
|
307
|
+
) from secondary
|
|
308
|
+
return FetchResult(
|
|
309
|
+
url=result.url,
|
|
310
|
+
status=result.status,
|
|
311
|
+
content=result.content,
|
|
312
|
+
sha256=result.sha256,
|
|
313
|
+
headers=result.headers,
|
|
314
|
+
fallback_used=True,
|
|
315
|
+
fallback_name=fallback_name or fallback_url,
|
|
316
|
+
# The mirror read is still the read that produced these bytes.
|
|
317
|
+
retrieved_at=result.retrieved_at,
|
|
318
|
+
cache_hit=result.cache_hit,
|
|
319
|
+
revalidated_at=result.revalidated_at,
|
|
320
|
+
)
|
|
321
|
+
|
|
322
|
+
|
|
323
|
+
def fetch_json(url: str, **kwargs: Any) -> tuple[FetchResult, Any]:
|
|
324
|
+
"""``fetch`` plus JSON parsing. Invalid JSON is an unreachable source."""
|
|
325
|
+
result = fetch(url, **kwargs)
|
|
326
|
+
return result, result.json()
|
|
327
|
+
|
|
328
|
+
|
|
329
|
+
def _read(
|
|
330
|
+
url: str,
|
|
331
|
+
*,
|
|
332
|
+
params: Mapping[str, Any] | None,
|
|
333
|
+
headers: Mapping[str, str] | None,
|
|
334
|
+
timeout: float,
|
|
335
|
+
raise_for_status: bool,
|
|
336
|
+
max_bytes: int = DEFAULT_MAX_BYTES,
|
|
337
|
+
cache_scope: str = "",
|
|
338
|
+
) -> FetchResult:
|
|
339
|
+
if urlparse(url).scheme == "file":
|
|
340
|
+
return _read_file(url, max_bytes=max_bytes)
|
|
341
|
+
|
|
342
|
+
# One header object is the source of truth for both the cache signature and
|
|
343
|
+
# the transport. HTTPX applies the case-insensitive replacement semantics
|
|
344
|
+
# that a plain dict.update cannot: ``user-agent`` replaces ``User-Agent``
|
|
345
|
+
# instead of inventing a second occurrence beside the default. It also
|
|
346
|
+
# preserves genuine repeated occurrences from an ``httpx.Headers`` caller.
|
|
347
|
+
request_headers = httpx.Headers(
|
|
348
|
+
{"User-Agent": USER_AGENT, "Accept-Encoding": "gzip"}
|
|
349
|
+
)
|
|
350
|
+
request_headers.update(headers or {})
|
|
351
|
+
|
|
352
|
+
# Conditional revalidation. The cache never answers on its own: it supplies
|
|
353
|
+
# a validator, and only a 304 from upstream turns into a hit. So a cached
|
|
354
|
+
# answer is one the source confirmed a moment ago, and the receipt still
|
|
355
|
+
# reports when the BYTES were retrieved rather than when they were reused.
|
|
356
|
+
#
|
|
357
|
+
# The key covers the ACTUAL merged headers and this call's byte ceiling, not
|
|
358
|
+
# the caller's pre-merge object and not just the URL. The distinction is the
|
|
359
|
+
# contract: a signature more precise than the wire costs one cache miss; a
|
|
360
|
+
# signature less precise than the wire can return a wrong body after a 304.
|
|
361
|
+
#
|
|
362
|
+
# The cache's own validators are the only headers not present yet. They come
|
|
363
|
+
# out of the entry we are about to look up, so including them would make the
|
|
364
|
+
# key depend on its own answer and every revalidation would miss. A caller's
|
|
365
|
+
# conditional header is already in request_headers and remains key material.
|
|
366
|
+
signature = cache.request_signature(
|
|
367
|
+
url,
|
|
368
|
+
params=params,
|
|
369
|
+
scope=cache_scope,
|
|
370
|
+
headers=request_headers,
|
|
371
|
+
max_bytes=max_bytes,
|
|
372
|
+
)
|
|
373
|
+
key = cache.http_key(
|
|
374
|
+
url,
|
|
375
|
+
params=params,
|
|
376
|
+
scope=cache_scope,
|
|
377
|
+
headers=request_headers,
|
|
378
|
+
max_bytes=max_bytes,
|
|
379
|
+
)
|
|
380
|
+
entry = cache.read_http(key, signature=signature)
|
|
381
|
+
if entry is not None:
|
|
382
|
+
request_headers.update(entry.conditional_headers())
|
|
383
|
+
|
|
384
|
+
try:
|
|
385
|
+
with httpx.Client(timeout=timeout, follow_redirects=True) as client:
|
|
386
|
+
# Streamed, not buffered: `response.content` decompresses the whole
|
|
387
|
+
# body before anyone can object to its size, which is exactly how a
|
|
388
|
+
# 200 KB gzip bomb became 200 MB of resident memory.
|
|
389
|
+
with client.stream(
|
|
390
|
+
"GET", url, params=params, headers=request_headers
|
|
391
|
+
) as response:
|
|
392
|
+
if response.status_code == 304 and entry is not None:
|
|
393
|
+
# The ceiling is re-enforced on the way out of the cache, not
|
|
394
|
+
# only on the way in. Otherwise a body stored when the call
|
|
395
|
+
# allowed 30 MB would be replayed whole to a call that
|
|
396
|
+
# declared 16 -- the limit would hold against upstream and
|
|
397
|
+
# quietly not hold against ourselves.
|
|
398
|
+
if len(entry.body) > max_bytes:
|
|
399
|
+
raise ResponseTooLarge(url, max_bytes, status=response.status_code)
|
|
400
|
+
confirmed = _utcnow()
|
|
401
|
+
cache.touch_http(key, entry, confirmed)
|
|
402
|
+
response.close()
|
|
403
|
+
return FetchResult(
|
|
404
|
+
url=entry.url,
|
|
405
|
+
status=entry.status,
|
|
406
|
+
content=entry.body,
|
|
407
|
+
sha256=entry.sha256,
|
|
408
|
+
headers=entry.headers,
|
|
409
|
+
retrieved_at=entry.retrieved_at,
|
|
410
|
+
cache_hit=True,
|
|
411
|
+
revalidated_at=confirmed,
|
|
412
|
+
)
|
|
413
|
+
if raise_for_status and not (200 <= response.status_code < 300):
|
|
414
|
+
raise SourceUnreachable(
|
|
415
|
+
url, "non-success status", status=response.status_code
|
|
416
|
+
)
|
|
417
|
+
|
|
418
|
+
buffer, hasher = _read_bounded(response, url, max_bytes)
|
|
419
|
+
final_url = str(response.request.url)
|
|
420
|
+
status = response.status_code
|
|
421
|
+
response_headers = {k.lower(): v for k, v in response.headers.items()}
|
|
422
|
+
except httpx.HTTPError as exc:
|
|
423
|
+
raise SourceUnreachable(url, type(exc).__name__) from exc
|
|
424
|
+
|
|
425
|
+
retrieved_at = _utcnow()
|
|
426
|
+
result = FetchResult(
|
|
427
|
+
url=final_url,
|
|
428
|
+
status=status,
|
|
429
|
+
content=bytes(buffer),
|
|
430
|
+
sha256=hasher.hexdigest(),
|
|
431
|
+
headers=response_headers,
|
|
432
|
+
retrieved_at=retrieved_at,
|
|
433
|
+
revalidated_at=retrieved_at,
|
|
434
|
+
)
|
|
435
|
+
if 200 <= status < 300:
|
|
436
|
+
# write_http decides what is storable -- a 206 is not, whatever the key
|
|
437
|
+
# says, and neither is a 2xx without a validator.
|
|
438
|
+
cache.write_http(
|
|
439
|
+
key,
|
|
440
|
+
cache.CachedResponse(
|
|
441
|
+
url=result.url,
|
|
442
|
+
status=result.status,
|
|
443
|
+
body=result.content,
|
|
444
|
+
sha256=result.sha256,
|
|
445
|
+
headers=result.headers,
|
|
446
|
+
retrieved_at=retrieved_at,
|
|
447
|
+
revalidated_at=retrieved_at,
|
|
448
|
+
request_signature=signature,
|
|
449
|
+
),
|
|
450
|
+
)
|
|
451
|
+
return result
|
|
452
|
+
|
|
453
|
+
|
|
454
|
+
def _read_bounded(
|
|
455
|
+
response: httpx.Response, url: str, max_bytes: int
|
|
456
|
+
) -> tuple[bytearray, "hashlib._Hash"]:
|
|
457
|
+
"""Stream and inflate a response without ever holding more than the ceiling.
|
|
458
|
+
|
|
459
|
+
We iterate ``iter_raw`` and run the decompressor ourselves rather than using
|
|
460
|
+
``iter_bytes``. httpx's decoder calls ``zlib.decompress`` on a whole raw
|
|
461
|
+
chunk with no length bound, so a single 64 KiB chunk of a gzip bomb inflates
|
|
462
|
+
to tens of megabytes BEFORE any caller gets to object -- measured at 221 MB
|
|
463
|
+
peak RSS on the report's 200 KB / 200 MB bomb even with an outer 16 MiB cap.
|
|
464
|
+
Driving ``decompressobj.decompress(data, max_length)`` ourselves makes the
|
|
465
|
+
ceiling real: the bomb now peaks near the ceiling instead of near the
|
|
466
|
+
payload.
|
|
467
|
+
"""
|
|
468
|
+
encoding = response.headers.get("content-encoding", "").lower().strip()
|
|
469
|
+
if encoding not in _SUPPORTED_ENCODINGS:
|
|
470
|
+
raise SourceUnreachable(
|
|
471
|
+
url,
|
|
472
|
+
f"unsupported content-encoding {encoding!r}; SourceLock will not hand a "
|
|
473
|
+
"response to a decoder whose output size it cannot bound",
|
|
474
|
+
status=response.status_code,
|
|
475
|
+
)
|
|
476
|
+
if encoding in ("gzip", "x-gzip"):
|
|
477
|
+
decoder = zlib.decompressobj(16 + zlib.MAX_WBITS)
|
|
478
|
+
elif encoding == "deflate":
|
|
479
|
+
decoder = zlib.decompressobj()
|
|
480
|
+
else:
|
|
481
|
+
decoder = None
|
|
482
|
+
|
|
483
|
+
buffer = bytearray()
|
|
484
|
+
hasher = hashlib.sha256()
|
|
485
|
+
wire = 0
|
|
486
|
+
|
|
487
|
+
def _absorb(chunk: bytes) -> None:
|
|
488
|
+
if len(buffer) + len(chunk) > max_bytes:
|
|
489
|
+
raise ResponseTooLarge(url, max_bytes, status=response.status_code)
|
|
490
|
+
buffer.extend(chunk)
|
|
491
|
+
hasher.update(chunk)
|
|
492
|
+
|
|
493
|
+
for raw in response.iter_raw(_CHUNK):
|
|
494
|
+
wire += len(raw)
|
|
495
|
+
# A compressed body bigger than the ceiling is over the ceiling too,
|
|
496
|
+
# whatever it inflates to.
|
|
497
|
+
if wire > max_bytes:
|
|
498
|
+
raise ResponseTooLarge(url, max_bytes, status=response.status_code)
|
|
499
|
+
if decoder is None:
|
|
500
|
+
_absorb(raw)
|
|
501
|
+
continue
|
|
502
|
+
pending = raw
|
|
503
|
+
while pending:
|
|
504
|
+
# +1 so that overshooting the ceiling is observable rather than
|
|
505
|
+
# silently clipped to exactly the limit.
|
|
506
|
+
try:
|
|
507
|
+
_absorb(decoder.decompress(pending, max_bytes - len(buffer) + 1))
|
|
508
|
+
except zlib.error as exc:
|
|
509
|
+
raise SourceUnreachable(
|
|
510
|
+
url, f"malformed {encoding} body: {type(exc).__name__}",
|
|
511
|
+
status=response.status_code,
|
|
512
|
+
) from exc
|
|
513
|
+
pending = decoder.unconsumed_tail
|
|
514
|
+
|
|
515
|
+
if decoder is not None:
|
|
516
|
+
try:
|
|
517
|
+
_absorb(decoder.flush())
|
|
518
|
+
except zlib.error as exc:
|
|
519
|
+
raise SourceUnreachable(
|
|
520
|
+
url, f"malformed {encoding} body: {type(exc).__name__}",
|
|
521
|
+
status=response.status_code,
|
|
522
|
+
) from exc
|
|
523
|
+
|
|
524
|
+
return buffer, hasher
|
|
525
|
+
|
|
526
|
+
|
|
527
|
+
def _read_file(url: str, *, max_bytes: int = DEFAULT_MAX_BYTES) -> FetchResult:
|
|
528
|
+
path = Path(unquote(urlparse(url).path))
|
|
529
|
+
try:
|
|
530
|
+
if path.stat().st_size > max_bytes:
|
|
531
|
+
raise ResponseTooLarge(url, max_bytes)
|
|
532
|
+
content = path.read_bytes()
|
|
533
|
+
except OSError as exc:
|
|
534
|
+
raise SourceUnreachable(url, type(exc).__name__) from exc
|
|
535
|
+
return FetchResult(
|
|
536
|
+
url=url,
|
|
537
|
+
status=None,
|
|
538
|
+
content=content,
|
|
539
|
+
sha256=hashlib.sha256(content).hexdigest(),
|
|
540
|
+
headers={},
|
|
541
|
+
)
|