sourcelock 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
hc_source/http.py ADDED
@@ -0,0 +1,541 @@
1
+ """The one transport every adapter uses.
2
+
3
+ Adapters do not create their own httpx clients. Going through :func:`fetch`
4
+ buys three things the receipt depends on: the SHA-256 of the exact bytes an
5
+ answer was derived from, a uniform authority/fallback policy, and the guarantee
6
+ that no response body is ever logged.
7
+
8
+ ``file://`` URLs are supported so that fixtures and locally-mirrored releases
9
+ travel the same code path as live HTTP.
10
+
11
+ Response bodies are STREAMED under a byte ceiling and hashed incrementally.
12
+ Round-1 finding H6: this module read ``response.content`` whole with no cap
13
+ while explicitly asking for gzip, so a 300 MB body at a canary URL took the
14
+ process to 651 MB RSS and a ~200 KB gzip bomb inflated to 200 MB -- about 1000x
15
+ amplification, from the one side an attacker or a broken CDN actually controls.
16
+ The guard had capped INBOUND parameters at 64 KiB since day one and there was
17
+ no equivalent for the response.
18
+
19
+ The default ceiling comes from live measurement of every URL the shipped
20
+ adapters fetch:
21
+
22
+ =========================================== ============
23
+ ``current_lcd.zip`` (coverage bulk crosswalk) 31.8 MB
24
+ ``UPDATED.csv`` (OIG LEIE full file) 14.8 MB
25
+ local-coverage-articles report 1.02 MB
26
+ every other full-body fetch < 300 KB
27
+ NPPES bulk files never fetched (7 KB index only)
28
+ =========================================== ============
29
+
30
+ So the default is deliberately tight and the two large fetches raise it for
31
+ themselves. That way the two dozen small routes -- including the NPPES canary
32
+ that H6 was demonstrated against -- get a ceiling proportionate to what they
33
+ actually read, instead of one sized for the biggest file in the product.
34
+ """
35
+
36
+ from __future__ import annotations
37
+
38
+ import contextvars
39
+ import hashlib
40
+ import json
41
+ import os
42
+ import zlib
43
+ from contextlib import contextmanager
44
+ from dataclasses import dataclass, field, replace
45
+ from datetime import datetime, timezone
46
+ from pathlib import Path
47
+ from typing import Any, Iterator, Mapping
48
+ from urllib.parse import unquote, urlparse
49
+
50
+ import httpx
51
+
52
+ from . import cache
53
+
54
+ __all__ = [
55
+ "DEFAULT_MAX_BYTES",
56
+ "DEFAULT_TIMEOUT",
57
+ "FetchResult",
58
+ "RETRYABLE_STATUSES",
59
+ "ResponseTooLarge",
60
+ "SourceUnreachable",
61
+ "USER_AGENT",
62
+ "as_cache_hit",
63
+ "fetch",
64
+ "fetch_json",
65
+ "is_transient",
66
+ "request_timeout",
67
+ ]
68
+
69
+ DEFAULT_TIMEOUT = 20.0
70
+ USER_AGENT = "sourcelock/0.1 (+https://github.com/sourcelock)"
71
+
72
+ #: Statuses worth trying again. Everything else -- 401, 403, 404, a 200 whose
73
+ #: body was malformed -- returns the same answer on the second attempt and just
74
+ #: spends the build's time getting there. Retrying is doctor's decision, but
75
+ #: WHICH failures are transient is knowledge about the transport, so it lives
76
+ #: here beside the exception that carries the status.
77
+ RETRYABLE_STATUSES = frozenset({408, 425, 429, 500, 502, 503, 504})
78
+
79
+ #: Ambient timeout, set by doctor around one canary's observation. A ContextVar
80
+ #: rather than a module global because it is scoped to a call: a global would
81
+ #: leak the last canary's budget into every tool call that followed it.
82
+ _TIMEOUT_VAR: contextvars.ContextVar[float | None] = contextvars.ContextVar(
83
+ "hc_source_http_timeout", default=None
84
+ )
85
+
86
+
87
+ @contextmanager
88
+ def request_timeout(seconds: float | None) -> Iterator[None]:
89
+ """Apply a per-request timeout to every fetch made inside this block.
90
+
91
+ This is how a canary's declared budget reaches a transport it never passes
92
+ arguments to: the canary's ``observe`` takes no parameters by design, so
93
+ doctor sets the policy around the call instead of threading it through every
94
+ adapter. An explicit ``timeout=`` at a call site still wins -- the coverage
95
+ bulk zip needs 300 seconds and should not inherit a canary's twenty.
96
+ """
97
+ token = _TIMEOUT_VAR.set(seconds if seconds is not None else _TIMEOUT_VAR.get())
98
+ try:
99
+ yield
100
+ finally:
101
+ _TIMEOUT_VAR.reset(token)
102
+
103
+
104
+ def _env_float(name: str) -> float | None:
105
+ raw = os.environ.get(name)
106
+ if raw is None:
107
+ return None
108
+ try:
109
+ value = float(raw)
110
+ except ValueError:
111
+ return None
112
+ return value if value > 0 else None
113
+
114
+
115
+ def _resolve_timeout(explicit: float | None) -> float:
116
+ """Call site, then ambient policy, then environment, then the default."""
117
+ if explicit is not None:
118
+ return explicit
119
+ ambient = _TIMEOUT_VAR.get()
120
+ if ambient is not None:
121
+ return ambient
122
+ return _env_float("HC_SOURCE_HTTP_TIMEOUT") or DEFAULT_TIMEOUT
123
+
124
+
125
+ def is_transient(exc: SourceUnreachable) -> bool:
126
+ """Would trying again plausibly get a different answer?
127
+
128
+ A body over the ceiling is excluded on purpose: it will be over the ceiling
129
+ again, and downloading it twice to find that out is the opposite of cheap.
130
+ A ``file://`` failure is excluded for the same reason -- a local file that
131
+ could not be read will not read differently in 500 milliseconds.
132
+ """
133
+ if isinstance(exc, ResponseTooLarge):
134
+ return False
135
+ if urlparse(exc.url).scheme == "file":
136
+ return False
137
+ return exc.status is None or exc.status in RETRYABLE_STATUSES
138
+
139
+
140
+ #: Ceiling on the DECOMPRESSED size of one response, in bytes. Sized against
141
+ #: the measured payloads in the module docstring: 15x the largest routine fetch
142
+ #: (the 1.02 MB articles report) and comfortably below anything that threatens
143
+ #: a 2-7 GB CI container. The two genuinely large fetches -- the LEIE full file
144
+ #: and the MCD bulk zip -- pass their own ``max_bytes``.
145
+ DEFAULT_MAX_BYTES = 16 * 1024 * 1024
146
+
147
+ #: Chunk size for streaming. Small enough that the ceiling is enforced promptly,
148
+ #: large enough not to make a 30 MB download a syscall storm.
149
+ _CHUNK = 64 * 1024
150
+
151
+ #: Content encodings we will inflate ourselves. Anything else is refused rather
152
+ #: than handed to a decoder we cannot bound.
153
+ _SUPPORTED_ENCODINGS = {"", "identity", "gzip", "x-gzip", "deflate"}
154
+
155
+
156
+ class SourceUnreachable(RuntimeError):
157
+ """The upstream source could not be read.
158
+
159
+ Carries no response body: doctor output and logs stay payload-free.
160
+ """
161
+
162
+ def __init__(self, url: str, reason: str, *, status: int | None = None) -> None:
163
+ self.url = url
164
+ self.reason = reason
165
+ self.status = status
166
+ super().__init__(f"{url} unreachable: {reason}" + (f" (HTTP {status})" if status else ""))
167
+
168
+
169
+ class ResponseTooLarge(SourceUnreachable):
170
+ """The upstream body exceeded this call's byte ceiling.
171
+
172
+ A subclass of :class:`SourceUnreachable` on purpose. Every caller that
173
+ already handles an unreachable source -- doctor, the CLI, the MCP server --
174
+ handles this correctly without changing, and the classification is honest:
175
+ the source was not read, so this run does not know what it says.
176
+
177
+ Never truncate-and-continue. A hash of the first N bytes with a receipt
178
+ attesting to it is a confident wrong answer, which is the one thing this
179
+ product exists not to produce.
180
+ """
181
+
182
+ def __init__(self, url: str, limit: int, *, status: int | None = None) -> None:
183
+ self.limit = limit
184
+ super().__init__(
185
+ url,
186
+ f"response exceeded the {limit}-byte ceiling for this call. If this source "
187
+ "legitimately grew, raise max_bytes at the call site; otherwise the upstream "
188
+ "is serving something it should not",
189
+ status=status,
190
+ )
191
+
192
+
193
+ def _utcnow() -> datetime:
194
+ return datetime.now(timezone.utc)
195
+
196
+
197
+ @dataclass(frozen=True)
198
+ class FetchResult:
199
+ """Raw upstream bytes plus the provenance a receipt needs."""
200
+
201
+ url: str
202
+ status: int | None
203
+ content: bytes
204
+ sha256: str
205
+ headers: dict[str, str]
206
+ fallback_used: bool = False
207
+ fallback_name: str | None = None
208
+ #: When THESE bytes were read from upstream. Stamped here, at the fetch
209
+ #: boundary, and never re-stamped -- round-1 finding: ``build_receipt`` used
210
+ #: to call ``utcnow()`` when the receipt was assembled, so a route served
211
+ #: out of a process cache produced a receipt claiming it had just been to
212
+ #: CMS. The receipt is the product; a retrieval time it invented is the one
213
+ #: field that must never be a guess. A cached ``FetchResult`` carries the
214
+ #: timestamp of the fetch that created it, because the object is built once.
215
+ retrieved_at: datetime = field(default_factory=_utcnow)
216
+ #: True when these bytes came out of the on-disk cache rather than off the
217
+ #: wire. Never inferred: the receipt carries it, so a reader can tell an
218
+ #: answer computed from a fresh download from one computed from bytes stored
219
+ #: last week and re-confirmed a moment ago.
220
+ cache_hit: bool = False
221
+ #: When upstream last confirmed these bytes are current -- the moment of the
222
+ #: 304 on a cache hit, the moment of the download otherwise. Kept apart from
223
+ #: ``retrieved_at`` because "you still have the right bytes" and "these bytes
224
+ #: are new" are different statements, and a receipt that merges them is
225
+ #: back to inventing a retrieval time.
226
+ revalidated_at: datetime | None = None
227
+
228
+ def json(self) -> Any:
229
+ try:
230
+ return json.loads(self.content)
231
+ except ValueError as exc:
232
+ raise SourceUnreachable(self.url, f"response was not valid JSON: {type(exc).__name__}")
233
+
234
+
235
+ #: Sentinel for "leave ``revalidated_at`` alone", distinct from ``None``, which
236
+ #: means "nobody upstream has ever confirmed these bytes".
237
+ _KEEP = object()
238
+
239
+
240
+ def as_cache_hit(result: FetchResult, *, revalidated_at: Any = _KEEP) -> FetchResult:
241
+ """The same bytes, served again out of an adapter's own in-process cache.
242
+
243
+ Adapters keep parsed snapshots in memory -- LEIE's 15 MB exclusion file, the
244
+ coverage bulk crosswalk -- and re-answer from them for minutes at a time.
245
+ Those answers used to carry ``cache_hit=false`` because the flag came
246
+ straight off the original :class:`FetchResult`, so the receipt's prose said
247
+ "served from an in-process cache" while its structured field said the
248
+ opposite. A machine reading the receipt believed the field.
249
+
250
+ ``retrieved_at``, ``sha256`` and ``status`` are preserved: they describe the
251
+ bytes, and the bytes did not change. ``revalidated_at`` defaults to whatever
252
+ upstream last confirmed -- pass a timestamp ONLY when upstream just
253
+ confirmed these bytes again, and pass ``None`` when it never has. A reuse
254
+ that contacted nobody must not advance it; that is the difference between
255
+ "still current as of a moment ago" and "still what we downloaded on Tuesday".
256
+ """
257
+ return replace(
258
+ result,
259
+ cache_hit=True,
260
+ revalidated_at=(
261
+ result.revalidated_at if revalidated_at is _KEEP else revalidated_at
262
+ ),
263
+ )
264
+
265
+
266
+ def fetch(
267
+ url: str,
268
+ *,
269
+ fallback_url: str | None = None,
270
+ fallback_name: str | None = None,
271
+ params: Mapping[str, Any] | None = None,
272
+ headers: Mapping[str, str] | None = None,
273
+ timeout: float | None = None,
274
+ raise_for_status: bool = True,
275
+ max_bytes: int = DEFAULT_MAX_BYTES,
276
+ cache_scope: str = "",
277
+ ) -> FetchResult:
278
+ """Read ``url``, falling back to ``fallback_url`` only if the authority fails.
279
+
280
+ Raises :class:`SourceUnreachable` when neither location can be read, when
281
+ the status is not 2xx and ``raise_for_status`` is set, or -- as
282
+ :class:`ResponseTooLarge` -- when the body exceeds ``max_bytes`` decompressed.
283
+ Raise ``max_bytes`` at the call site for a route that legitimately reads a
284
+ large file; do not raise the default for one route's sake.
285
+
286
+ ``timeout`` defaults to the ambient policy (see :func:`request_timeout`),
287
+ then to ``HC_SOURCE_HTTP_TIMEOUT``, then to :data:`DEFAULT_TIMEOUT`. An
288
+ explicit argument always wins.
289
+ """
290
+ timeout = _resolve_timeout(timeout)
291
+ try:
292
+ return _read(url, params=params, headers=headers, timeout=timeout,
293
+ raise_for_status=raise_for_status, max_bytes=max_bytes,
294
+ cache_scope=cache_scope)
295
+ except SourceUnreachable as primary:
296
+ if not fallback_url:
297
+ raise
298
+ try:
299
+ result = _read(fallback_url, params=params, headers=headers, timeout=timeout,
300
+ raise_for_status=raise_for_status, max_bytes=max_bytes,
301
+ cache_scope=cache_scope)
302
+ except SourceUnreachable as secondary:
303
+ raise SourceUnreachable(
304
+ url,
305
+ f"authority failed ({primary.reason}) and fallback failed ({secondary.reason})",
306
+ status=primary.status,
307
+ ) from secondary
308
+ return FetchResult(
309
+ url=result.url,
310
+ status=result.status,
311
+ content=result.content,
312
+ sha256=result.sha256,
313
+ headers=result.headers,
314
+ fallback_used=True,
315
+ fallback_name=fallback_name or fallback_url,
316
+ # The mirror read is still the read that produced these bytes.
317
+ retrieved_at=result.retrieved_at,
318
+ cache_hit=result.cache_hit,
319
+ revalidated_at=result.revalidated_at,
320
+ )
321
+
322
+
323
+ def fetch_json(url: str, **kwargs: Any) -> tuple[FetchResult, Any]:
324
+ """``fetch`` plus JSON parsing. Invalid JSON is an unreachable source."""
325
+ result = fetch(url, **kwargs)
326
+ return result, result.json()
327
+
328
+
329
+ def _read(
330
+ url: str,
331
+ *,
332
+ params: Mapping[str, Any] | None,
333
+ headers: Mapping[str, str] | None,
334
+ timeout: float,
335
+ raise_for_status: bool,
336
+ max_bytes: int = DEFAULT_MAX_BYTES,
337
+ cache_scope: str = "",
338
+ ) -> FetchResult:
339
+ if urlparse(url).scheme == "file":
340
+ return _read_file(url, max_bytes=max_bytes)
341
+
342
+ # One header object is the source of truth for both the cache signature and
343
+ # the transport. HTTPX applies the case-insensitive replacement semantics
344
+ # that a plain dict.update cannot: ``user-agent`` replaces ``User-Agent``
345
+ # instead of inventing a second occurrence beside the default. It also
346
+ # preserves genuine repeated occurrences from an ``httpx.Headers`` caller.
347
+ request_headers = httpx.Headers(
348
+ {"User-Agent": USER_AGENT, "Accept-Encoding": "gzip"}
349
+ )
350
+ request_headers.update(headers or {})
351
+
352
+ # Conditional revalidation. The cache never answers on its own: it supplies
353
+ # a validator, and only a 304 from upstream turns into a hit. So a cached
354
+ # answer is one the source confirmed a moment ago, and the receipt still
355
+ # reports when the BYTES were retrieved rather than when they were reused.
356
+ #
357
+ # The key covers the ACTUAL merged headers and this call's byte ceiling, not
358
+ # the caller's pre-merge object and not just the URL. The distinction is the
359
+ # contract: a signature more precise than the wire costs one cache miss; a
360
+ # signature less precise than the wire can return a wrong body after a 304.
361
+ #
362
+ # The cache's own validators are the only headers not present yet. They come
363
+ # out of the entry we are about to look up, so including them would make the
364
+ # key depend on its own answer and every revalidation would miss. A caller's
365
+ # conditional header is already in request_headers and remains key material.
366
+ signature = cache.request_signature(
367
+ url,
368
+ params=params,
369
+ scope=cache_scope,
370
+ headers=request_headers,
371
+ max_bytes=max_bytes,
372
+ )
373
+ key = cache.http_key(
374
+ url,
375
+ params=params,
376
+ scope=cache_scope,
377
+ headers=request_headers,
378
+ max_bytes=max_bytes,
379
+ )
380
+ entry = cache.read_http(key, signature=signature)
381
+ if entry is not None:
382
+ request_headers.update(entry.conditional_headers())
383
+
384
+ try:
385
+ with httpx.Client(timeout=timeout, follow_redirects=True) as client:
386
+ # Streamed, not buffered: `response.content` decompresses the whole
387
+ # body before anyone can object to its size, which is exactly how a
388
+ # 200 KB gzip bomb became 200 MB of resident memory.
389
+ with client.stream(
390
+ "GET", url, params=params, headers=request_headers
391
+ ) as response:
392
+ if response.status_code == 304 and entry is not None:
393
+ # The ceiling is re-enforced on the way out of the cache, not
394
+ # only on the way in. Otherwise a body stored when the call
395
+ # allowed 30 MB would be replayed whole to a call that
396
+ # declared 16 -- the limit would hold against upstream and
397
+ # quietly not hold against ourselves.
398
+ if len(entry.body) > max_bytes:
399
+ raise ResponseTooLarge(url, max_bytes, status=response.status_code)
400
+ confirmed = _utcnow()
401
+ cache.touch_http(key, entry, confirmed)
402
+ response.close()
403
+ return FetchResult(
404
+ url=entry.url,
405
+ status=entry.status,
406
+ content=entry.body,
407
+ sha256=entry.sha256,
408
+ headers=entry.headers,
409
+ retrieved_at=entry.retrieved_at,
410
+ cache_hit=True,
411
+ revalidated_at=confirmed,
412
+ )
413
+ if raise_for_status and not (200 <= response.status_code < 300):
414
+ raise SourceUnreachable(
415
+ url, "non-success status", status=response.status_code
416
+ )
417
+
418
+ buffer, hasher = _read_bounded(response, url, max_bytes)
419
+ final_url = str(response.request.url)
420
+ status = response.status_code
421
+ response_headers = {k.lower(): v for k, v in response.headers.items()}
422
+ except httpx.HTTPError as exc:
423
+ raise SourceUnreachable(url, type(exc).__name__) from exc
424
+
425
+ retrieved_at = _utcnow()
426
+ result = FetchResult(
427
+ url=final_url,
428
+ status=status,
429
+ content=bytes(buffer),
430
+ sha256=hasher.hexdigest(),
431
+ headers=response_headers,
432
+ retrieved_at=retrieved_at,
433
+ revalidated_at=retrieved_at,
434
+ )
435
+ if 200 <= status < 300:
436
+ # write_http decides what is storable -- a 206 is not, whatever the key
437
+ # says, and neither is a 2xx without a validator.
438
+ cache.write_http(
439
+ key,
440
+ cache.CachedResponse(
441
+ url=result.url,
442
+ status=result.status,
443
+ body=result.content,
444
+ sha256=result.sha256,
445
+ headers=result.headers,
446
+ retrieved_at=retrieved_at,
447
+ revalidated_at=retrieved_at,
448
+ request_signature=signature,
449
+ ),
450
+ )
451
+ return result
452
+
453
+
454
+ def _read_bounded(
455
+ response: httpx.Response, url: str, max_bytes: int
456
+ ) -> tuple[bytearray, "hashlib._Hash"]:
457
+ """Stream and inflate a response without ever holding more than the ceiling.
458
+
459
+ We iterate ``iter_raw`` and run the decompressor ourselves rather than using
460
+ ``iter_bytes``. httpx's decoder calls ``zlib.decompress`` on a whole raw
461
+ chunk with no length bound, so a single 64 KiB chunk of a gzip bomb inflates
462
+ to tens of megabytes BEFORE any caller gets to object -- measured at 221 MB
463
+ peak RSS on the report's 200 KB / 200 MB bomb even with an outer 16 MiB cap.
464
+ Driving ``decompressobj.decompress(data, max_length)`` ourselves makes the
465
+ ceiling real: the bomb now peaks near the ceiling instead of near the
466
+ payload.
467
+ """
468
+ encoding = response.headers.get("content-encoding", "").lower().strip()
469
+ if encoding not in _SUPPORTED_ENCODINGS:
470
+ raise SourceUnreachable(
471
+ url,
472
+ f"unsupported content-encoding {encoding!r}; SourceLock will not hand a "
473
+ "response to a decoder whose output size it cannot bound",
474
+ status=response.status_code,
475
+ )
476
+ if encoding in ("gzip", "x-gzip"):
477
+ decoder = zlib.decompressobj(16 + zlib.MAX_WBITS)
478
+ elif encoding == "deflate":
479
+ decoder = zlib.decompressobj()
480
+ else:
481
+ decoder = None
482
+
483
+ buffer = bytearray()
484
+ hasher = hashlib.sha256()
485
+ wire = 0
486
+
487
+ def _absorb(chunk: bytes) -> None:
488
+ if len(buffer) + len(chunk) > max_bytes:
489
+ raise ResponseTooLarge(url, max_bytes, status=response.status_code)
490
+ buffer.extend(chunk)
491
+ hasher.update(chunk)
492
+
493
+ for raw in response.iter_raw(_CHUNK):
494
+ wire += len(raw)
495
+ # A compressed body bigger than the ceiling is over the ceiling too,
496
+ # whatever it inflates to.
497
+ if wire > max_bytes:
498
+ raise ResponseTooLarge(url, max_bytes, status=response.status_code)
499
+ if decoder is None:
500
+ _absorb(raw)
501
+ continue
502
+ pending = raw
503
+ while pending:
504
+ # +1 so that overshooting the ceiling is observable rather than
505
+ # silently clipped to exactly the limit.
506
+ try:
507
+ _absorb(decoder.decompress(pending, max_bytes - len(buffer) + 1))
508
+ except zlib.error as exc:
509
+ raise SourceUnreachable(
510
+ url, f"malformed {encoding} body: {type(exc).__name__}",
511
+ status=response.status_code,
512
+ ) from exc
513
+ pending = decoder.unconsumed_tail
514
+
515
+ if decoder is not None:
516
+ try:
517
+ _absorb(decoder.flush())
518
+ except zlib.error as exc:
519
+ raise SourceUnreachable(
520
+ url, f"malformed {encoding} body: {type(exc).__name__}",
521
+ status=response.status_code,
522
+ ) from exc
523
+
524
+ return buffer, hasher
525
+
526
+
527
+ def _read_file(url: str, *, max_bytes: int = DEFAULT_MAX_BYTES) -> FetchResult:
528
+ path = Path(unquote(urlparse(url).path))
529
+ try:
530
+ if path.stat().st_size > max_bytes:
531
+ raise ResponseTooLarge(url, max_bytes)
532
+ content = path.read_bytes()
533
+ except OSError as exc:
534
+ raise SourceUnreachable(url, type(exc).__name__) from exc
535
+ return FetchResult(
536
+ url=url,
537
+ status=None,
538
+ content=content,
539
+ sha256=hashlib.sha256(content).hexdigest(),
540
+ headers={},
541
+ )