standin 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- standin/__init__.py +42 -0
- standin/_codec.py +81 -0
- standin/cassette.py +40 -0
- standin/cli.py +112 -0
- standin/config.py +41 -0
- standin/core.py +68 -0
- standin/engine.py +105 -0
- standin/exceptions.py +22 -0
- standin/interceptors/__init__.py +11 -0
- standin/interceptors/base.py +35 -0
- standin/interceptors/httpx_interceptor.py +89 -0
- standin/matching.py +89 -0
- standin/models.py +63 -0
- standin/py.typed +1 -0
- standin/pytest_plugin.py +37 -0
- standin/redaction.py +73 -0
- standin/storage.py +82 -0
- standin-0.2.0.dist-info/METADATA +182 -0
- standin-0.2.0.dist-info/RECORD +22 -0
- standin-0.2.0.dist-info/WHEEL +4 -0
- standin-0.2.0.dist-info/entry_points.txt +5 -0
- standin-0.2.0.dist-info/licenses/LICENSE +21 -0
standin/__init__.py
ADDED
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
"""standin — a stand-in for the real LLM in your tests.
|
|
2
|
+
|
|
3
|
+
Record real LLM API calls once, then replay them forever: fast, free, offline,
|
|
4
|
+
and deterministic. Provider-agnostic (hooks httpx), streaming and tool-calls
|
|
5
|
+
supported, secrets redacted so cassettes are safe to commit.
|
|
6
|
+
|
|
7
|
+
import standin
|
|
8
|
+
|
|
9
|
+
with standin.use_cassette("tests/cassettes/summary.json"):
|
|
10
|
+
resp = openai_client.chat.completions.create(...) # recorded once, replayed after
|
|
11
|
+
|
|
12
|
+
Architecture (see ARCHITECTURE.md): a transport-neutral policy **engine** sits
|
|
13
|
+
behind pluggable **interceptors** (httpx today), **matchers**, **redactors**,
|
|
14
|
+
and **stores**, so behavior is easy to reason about and extend.
|
|
15
|
+
"""
|
|
16
|
+
from .config import Config
|
|
17
|
+
from .core import use_cassette
|
|
18
|
+
from .exceptions import CannotReplay, CassetteError, ConfigError, StandinError
|
|
19
|
+
from .matching import DefaultMatcher, FuzzyMatcher, Matcher
|
|
20
|
+
from .models import Mode
|
|
21
|
+
from .redaction import DefaultRedactor, NullRedactor, Redactor
|
|
22
|
+
from .storage import CassetteStore, JSONCassetteStore
|
|
23
|
+
|
|
24
|
+
__version__ = "0.2.0"
|
|
25
|
+
__all__ = [
|
|
26
|
+
"use_cassette",
|
|
27
|
+
"Config",
|
|
28
|
+
"Mode",
|
|
29
|
+
"StandinError",
|
|
30
|
+
"CannotReplay",
|
|
31
|
+
"CassetteError",
|
|
32
|
+
"ConfigError",
|
|
33
|
+
"Redactor",
|
|
34
|
+
"DefaultRedactor",
|
|
35
|
+
"NullRedactor",
|
|
36
|
+
"Matcher",
|
|
37
|
+
"DefaultMatcher",
|
|
38
|
+
"FuzzyMatcher",
|
|
39
|
+
"CassetteStore",
|
|
40
|
+
"JSONCassetteStore",
|
|
41
|
+
"__version__",
|
|
42
|
+
]
|
standin/_codec.py
ADDED
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
"""Body (de)serialization and canonicalization.
|
|
2
|
+
|
|
3
|
+
A body is stored as one of ``{"json": ...}``, ``{"text": ...}``, ``{"b64": ...}``
|
|
4
|
+
or ``{"empty": true}`` so cassettes stay human-readable when possible. Canonical
|
|
5
|
+
forms are used only for request matching, never written to disk.
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import base64
|
|
10
|
+
import json
|
|
11
|
+
from typing import Any
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def _try_json(raw: bytes, content_type: str) -> Any | None:
|
|
15
|
+
if "application/json" not in (content_type or ""):
|
|
16
|
+
# Some providers omit the header; still try if it smells like JSON.
|
|
17
|
+
stripped = raw[:1]
|
|
18
|
+
if stripped not in (b"{", b"["):
|
|
19
|
+
return None
|
|
20
|
+
try:
|
|
21
|
+
return json.loads(raw.decode("utf-8"))
|
|
22
|
+
except (ValueError, UnicodeDecodeError):
|
|
23
|
+
return None
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def encode_body(raw: bytes, content_type: str, redactor=None) -> dict[str, Any]:
|
|
27
|
+
if not raw:
|
|
28
|
+
return {"empty": True}
|
|
29
|
+
obj = _try_json(raw, content_type)
|
|
30
|
+
if obj is not None:
|
|
31
|
+
if redactor is not None:
|
|
32
|
+
obj = redactor.redact_obj(obj)
|
|
33
|
+
return {"json": obj}
|
|
34
|
+
try:
|
|
35
|
+
text = raw.decode("utf-8")
|
|
36
|
+
if redactor is not None:
|
|
37
|
+
text = redactor.redact_text(text)
|
|
38
|
+
return {"text": text}
|
|
39
|
+
except UnicodeDecodeError:
|
|
40
|
+
return {"b64": base64.b64encode(raw).decode("ascii")}
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def decode_body(body: dict[str, Any]) -> bytes:
|
|
44
|
+
if not body or body.get("empty"):
|
|
45
|
+
return b""
|
|
46
|
+
if "json" in body:
|
|
47
|
+
return json.dumps(body["json"], ensure_ascii=False).encode("utf-8")
|
|
48
|
+
if "text" in body:
|
|
49
|
+
return body["text"].encode("utf-8")
|
|
50
|
+
if "b64" in body:
|
|
51
|
+
return base64.b64decode(body["b64"])
|
|
52
|
+
return b""
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def canonical_live(raw: bytes, content_type: str, redactor=None) -> str:
|
|
56
|
+
"""Canonical string for a live (in-flight) request body, for matching."""
|
|
57
|
+
if not raw:
|
|
58
|
+
return ""
|
|
59
|
+
obj = _try_json(raw, content_type)
|
|
60
|
+
if obj is not None:
|
|
61
|
+
if redactor is not None:
|
|
62
|
+
obj = redactor.redact_obj(obj)
|
|
63
|
+
return json.dumps(obj, sort_keys=True, ensure_ascii=False)
|
|
64
|
+
try:
|
|
65
|
+
text = raw.decode("utf-8")
|
|
66
|
+
return redactor.redact_text(text) if redactor is not None else text
|
|
67
|
+
except UnicodeDecodeError:
|
|
68
|
+
return base64.b64encode(raw).decode("ascii")
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def canonical_stored(body: dict[str, Any]) -> str:
|
|
72
|
+
"""Canonical string for a stored request body, for matching."""
|
|
73
|
+
if not body or body.get("empty"):
|
|
74
|
+
return ""
|
|
75
|
+
if "json" in body:
|
|
76
|
+
return json.dumps(body["json"], sort_keys=True, ensure_ascii=False)
|
|
77
|
+
if "text" in body:
|
|
78
|
+
return body["text"]
|
|
79
|
+
if "b64" in body:
|
|
80
|
+
return body["b64"]
|
|
81
|
+
return ""
|
standin/cassette.py
ADDED
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
"""In-memory cassette: the interactions plus the replay cursor."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import threading
|
|
5
|
+
from dataclasses import dataclass, field
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
from .models import Interaction
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
@dataclass
|
|
12
|
+
class Cassette:
|
|
13
|
+
path: Path
|
|
14
|
+
interactions: list[Interaction] = field(default_factory=list)
|
|
15
|
+
dirty: bool = False
|
|
16
|
+
preexisting: int = 0
|
|
17
|
+
_lock: threading.Lock = field(default_factory=threading.Lock, repr=False, compare=False)
|
|
18
|
+
|
|
19
|
+
def __post_init__(self):
|
|
20
|
+
self.preexisting = len(self.interactions)
|
|
21
|
+
|
|
22
|
+
def append(self, interaction: Interaction) -> None:
|
|
23
|
+
with self._lock:
|
|
24
|
+
self.interactions.append(interaction)
|
|
25
|
+
self.dirty = True
|
|
26
|
+
|
|
27
|
+
def find_unplayed(self, matches, live_request) -> Interaction | None:
|
|
28
|
+
"""Return the first not-yet-played interaction that matches the request.
|
|
29
|
+
|
|
30
|
+
`matches(live_request, stored_request) -> bool` is supplied by the
|
|
31
|
+
matcher, so exact and fuzzy strategies share this loop. Playing in order
|
|
32
|
+
lets repeated calls (e.g. an agent loop) replay their distinct recorded
|
|
33
|
+
responses sequentially. Locked so a shared cassette never double-plays.
|
|
34
|
+
"""
|
|
35
|
+
with self._lock:
|
|
36
|
+
for interaction in self.interactions:
|
|
37
|
+
if not interaction.played and matches(live_request, interaction.request):
|
|
38
|
+
interaction.played = True
|
|
39
|
+
return interaction
|
|
40
|
+
return None
|
standin/cli.py
ADDED
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
"""The `standin` command: inspect and maintain cassettes.
|
|
2
|
+
|
|
3
|
+
standin list cassette.json # one line per interaction
|
|
4
|
+
standin show cassette.json 0 # full request/response of interaction 0
|
|
5
|
+
standin stats cassette.json # summary counts
|
|
6
|
+
standin scrub cassette.json # re-run secret redaction over a cassette
|
|
7
|
+
"""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import argparse
|
|
11
|
+
import json
|
|
12
|
+
import sys
|
|
13
|
+
from collections import Counter
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
|
|
16
|
+
from . import __version__
|
|
17
|
+
from .redaction import DefaultRedactor
|
|
18
|
+
from .storage import JSONCassetteStore
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def _load(path: str):
|
|
22
|
+
return JSONCassetteStore().load(Path(path))
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def _redact_body(body: dict, r: DefaultRedactor) -> dict:
|
|
26
|
+
if "json" in body:
|
|
27
|
+
return {"json": r.redact_obj(body["json"])}
|
|
28
|
+
if "text" in body:
|
|
29
|
+
return {"text": r.redact_text(body["text"])}
|
|
30
|
+
return body
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def cmd_list(args) -> int:
|
|
34
|
+
items = _load(args.cassette)
|
|
35
|
+
print(f"{len(items)} interaction(s) in {args.cassette}")
|
|
36
|
+
for i, it in enumerate(items):
|
|
37
|
+
print(f" [{i}] {it.request.method:<6} {it.response.status_code} {it.request.url}")
|
|
38
|
+
return 0
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def cmd_show(args) -> int:
|
|
42
|
+
items = _load(args.cassette)
|
|
43
|
+
it = items[args.index]
|
|
44
|
+
print(json.dumps({"request": vars(it.request), "response": vars(it.response)}, indent=2, ensure_ascii=False))
|
|
45
|
+
return 0
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def cmd_stats(args) -> int:
|
|
49
|
+
items = _load(args.cassette)
|
|
50
|
+
print(f"interactions : {len(items)}")
|
|
51
|
+
print(f"methods : {dict(Counter(it.request.method for it in items))}")
|
|
52
|
+
print(f"statuses : {dict(Counter(it.response.status_code for it in items))}")
|
|
53
|
+
return 0
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def cmd_scrub(args) -> int:
|
|
57
|
+
store = JSONCassetteStore()
|
|
58
|
+
path = Path(args.cassette)
|
|
59
|
+
items = store.load(path)
|
|
60
|
+
r = DefaultRedactor()
|
|
61
|
+
for it in items:
|
|
62
|
+
it.request.headers = r.redact_headers(it.request.headers)
|
|
63
|
+
it.response.headers = r.redact_headers(it.response.headers)
|
|
64
|
+
it.request.body = _redact_body(it.request.body, r)
|
|
65
|
+
it.response.body = _redact_body(it.response.body, r)
|
|
66
|
+
store.save(path, items)
|
|
67
|
+
print(f"scrubbed {len(items)} interaction(s) in {path}")
|
|
68
|
+
return 0
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def main(argv=None) -> int:
|
|
72
|
+
reconfigure = getattr(sys.stdout, "reconfigure", None)
|
|
73
|
+
if reconfigure is not None:
|
|
74
|
+
try:
|
|
75
|
+
reconfigure(encoding="utf-8")
|
|
76
|
+
except Exception:
|
|
77
|
+
pass
|
|
78
|
+
|
|
79
|
+
ap = argparse.ArgumentParser(prog="standin", description="Inspect and maintain standin cassettes.")
|
|
80
|
+
ap.add_argument("--version", action="version", version=f"standin {__version__}")
|
|
81
|
+
sub = ap.add_subparsers(dest="cmd", required=True)
|
|
82
|
+
|
|
83
|
+
p = sub.add_parser("list", help="list interactions in a cassette")
|
|
84
|
+
p.add_argument("cassette")
|
|
85
|
+
p.set_defaults(func=cmd_list)
|
|
86
|
+
|
|
87
|
+
p = sub.add_parser("show", help="show one interaction as JSON")
|
|
88
|
+
p.add_argument("cassette")
|
|
89
|
+
p.add_argument("index", type=int)
|
|
90
|
+
p.set_defaults(func=cmd_show)
|
|
91
|
+
|
|
92
|
+
p = sub.add_parser("stats", help="summary counts for a cassette")
|
|
93
|
+
p.add_argument("cassette")
|
|
94
|
+
p.set_defaults(func=cmd_stats)
|
|
95
|
+
|
|
96
|
+
p = sub.add_parser("scrub", help="re-run secret redaction over a cassette")
|
|
97
|
+
p.add_argument("cassette")
|
|
98
|
+
p.set_defaults(func=cmd_scrub)
|
|
99
|
+
|
|
100
|
+
args = ap.parse_args(argv)
|
|
101
|
+
try:
|
|
102
|
+
return args.func(args)
|
|
103
|
+
except FileNotFoundError:
|
|
104
|
+
print(f"error: cassette not found: {args.cassette}", file=sys.stderr)
|
|
105
|
+
return 1
|
|
106
|
+
except IndexError:
|
|
107
|
+
print("error: interaction index out of range", file=sys.stderr)
|
|
108
|
+
return 1
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
if __name__ == "__main__":
|
|
112
|
+
raise SystemExit(main())
|
standin/config.py
ADDED
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
"""Configuration object wiring the pluggable pieces together."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
from collections.abc import Sequence
|
|
5
|
+
from dataclasses import dataclass, field
|
|
6
|
+
|
|
7
|
+
from .exceptions import ConfigError
|
|
8
|
+
from .matching import DEFAULT_MATCH_ON, Matcher
|
|
9
|
+
from .models import Mode
|
|
10
|
+
from .redaction import DefaultRedactor, NullRedactor, Redactor
|
|
11
|
+
from .storage import CassetteStore, JSONCassetteStore
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
@dataclass
|
|
15
|
+
class Config:
|
|
16
|
+
mode: Mode = Mode.ONCE
|
|
17
|
+
match_on: Sequence[str] = DEFAULT_MATCH_ON
|
|
18
|
+
redact: bool = True
|
|
19
|
+
redactor: Redactor | None = None
|
|
20
|
+
matcher: Matcher | None = None
|
|
21
|
+
store: CassetteStore = field(default_factory=JSONCassetteStore)
|
|
22
|
+
|
|
23
|
+
def __post_init__(self):
|
|
24
|
+
if not isinstance(self.mode, Mode):
|
|
25
|
+
try:
|
|
26
|
+
self.mode = Mode(self.mode)
|
|
27
|
+
except ValueError as exc:
|
|
28
|
+
raise ConfigError(
|
|
29
|
+
f"invalid mode {self.mode!r}; choose one of {[m.value for m in Mode]}"
|
|
30
|
+
) from exc
|
|
31
|
+
if self.redactor is None:
|
|
32
|
+
self.redactor = DefaultRedactor() if self.redact else NullRedactor()
|
|
33
|
+
# Validate match_on only for the built-in matcher; a custom matcher may
|
|
34
|
+
# define its own vocabulary.
|
|
35
|
+
if self.matcher is None:
|
|
36
|
+
valid = {"method", "url", "body"}
|
|
37
|
+
unknown = set(self.match_on) - valid
|
|
38
|
+
if unknown:
|
|
39
|
+
raise ConfigError(
|
|
40
|
+
f"unknown match_on {sorted(unknown)}; valid options are {sorted(valid)}"
|
|
41
|
+
)
|
standin/core.py
ADDED
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
"""Public entry point: the ``use_cassette`` context manager."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import os
|
|
5
|
+
from collections.abc import Iterator, Sequence
|
|
6
|
+
from contextlib import contextmanager
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
from .cassette import Cassette
|
|
10
|
+
from .config import Config
|
|
11
|
+
from .engine import Engine
|
|
12
|
+
from .exceptions import ConfigError
|
|
13
|
+
from .interceptors.base import reset_active_engine, set_active_engine
|
|
14
|
+
from .interceptors.httpx_interceptor import HttpxInterceptor
|
|
15
|
+
from .matching import DEFAULT_MATCH_ON, Matcher
|
|
16
|
+
from .models import Mode
|
|
17
|
+
from .redaction import Redactor
|
|
18
|
+
from .storage import CassetteStore
|
|
19
|
+
|
|
20
|
+
_httpx_interceptor = HttpxInterceptor()
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
@contextmanager
|
|
24
|
+
def use_cassette(
|
|
25
|
+
path,
|
|
26
|
+
*,
|
|
27
|
+
mode: str = "once",
|
|
28
|
+
match_on: Sequence[str] = DEFAULT_MATCH_ON,
|
|
29
|
+
redact: bool = True,
|
|
30
|
+
redactor: Redactor | None = None,
|
|
31
|
+
matcher: Matcher | None = None,
|
|
32
|
+
store: CassetteStore | None = None,
|
|
33
|
+
) -> Iterator[Cassette]:
|
|
34
|
+
"""Record LLM/HTTP calls to ``path``, or replay them if already recorded.
|
|
35
|
+
|
|
36
|
+
The environment variable ``STANDIN_MODE`` overrides ``mode`` for a whole run
|
|
37
|
+
(e.g. ``STANDIN_MODE=none pytest`` guarantees no live calls in CI).
|
|
38
|
+
"""
|
|
39
|
+
mode = os.environ.get("STANDIN_MODE", mode)
|
|
40
|
+
try:
|
|
41
|
+
mode_value = mode if isinstance(mode, Mode) else Mode(mode)
|
|
42
|
+
except ValueError as exc:
|
|
43
|
+
raise ConfigError(
|
|
44
|
+
f"invalid mode {mode!r}; choose one of {[m.value for m in Mode]}"
|
|
45
|
+
) from exc
|
|
46
|
+
|
|
47
|
+
config = Config(
|
|
48
|
+
mode=mode_value,
|
|
49
|
+
match_on=match_on,
|
|
50
|
+
redact=redact,
|
|
51
|
+
redactor=redactor,
|
|
52
|
+
matcher=matcher,
|
|
53
|
+
)
|
|
54
|
+
if store is not None:
|
|
55
|
+
config.store = store
|
|
56
|
+
|
|
57
|
+
path = Path(path)
|
|
58
|
+
cassette = Cassette(path=path, interactions=config.store.load(path))
|
|
59
|
+
engine = Engine(cassette, config)
|
|
60
|
+
|
|
61
|
+
_httpx_interceptor.install()
|
|
62
|
+
token = set_active_engine(engine)
|
|
63
|
+
try:
|
|
64
|
+
yield cassette
|
|
65
|
+
finally:
|
|
66
|
+
reset_active_engine(token)
|
|
67
|
+
if cassette.dirty:
|
|
68
|
+
config.store.save(path, cassette.interactions)
|
standin/engine.py
ADDED
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
"""The record/replay policy engine.
|
|
2
|
+
|
|
3
|
+
The engine is deliberately transport-agnostic: it speaks only ``RawRequest`` /
|
|
4
|
+
``RawResponse`` and a ``do_real`` callback. Interceptors adapt a concrete client
|
|
5
|
+
(httpx today, others later) to this interface, so all the decision logic lives
|
|
6
|
+
here in one testable place.
|
|
7
|
+
"""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from collections.abc import Awaitable
|
|
11
|
+
from typing import Callable
|
|
12
|
+
|
|
13
|
+
from . import _codec
|
|
14
|
+
from .cassette import Cassette
|
|
15
|
+
from .config import Config
|
|
16
|
+
from .exceptions import CannotReplay
|
|
17
|
+
from .matching import DefaultMatcher
|
|
18
|
+
from .models import (
|
|
19
|
+
Interaction,
|
|
20
|
+
Mode,
|
|
21
|
+
RawRequest,
|
|
22
|
+
RawResponse,
|
|
23
|
+
RecordedRequest,
|
|
24
|
+
RecordedResponse,
|
|
25
|
+
)
|
|
26
|
+
from .redaction import DefaultRedactor
|
|
27
|
+
|
|
28
|
+
SyncReal = Callable[[RawRequest], RawResponse]
|
|
29
|
+
AsyncReal = Callable[[RawRequest], Awaitable[RawResponse]]
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
class Engine:
|
|
33
|
+
def __init__(self, cassette: Cassette, config: Config):
|
|
34
|
+
self.cassette = cassette
|
|
35
|
+
self.config = config
|
|
36
|
+
# Config.__post_init__ guarantees a redactor; narrow it for the type checker.
|
|
37
|
+
self.redactor = config.redactor if config.redactor is not None else DefaultRedactor()
|
|
38
|
+
self.matcher = config.matcher or DefaultMatcher(self.redactor, config.match_on)
|
|
39
|
+
|
|
40
|
+
# -- policy -------------------------------------------------------------
|
|
41
|
+
def _plan(self) -> tuple[bool, bool]:
|
|
42
|
+
mode = self.config.mode
|
|
43
|
+
try_replay = mode in (Mode.ONCE, Mode.NONE, Mode.NEW_EPISODES)
|
|
44
|
+
record_on_miss = (
|
|
45
|
+
mode in (Mode.ALL, Mode.NEW_EPISODES)
|
|
46
|
+
or (mode is Mode.ONCE and self.cassette.preexisting == 0)
|
|
47
|
+
)
|
|
48
|
+
return try_replay, record_on_miss
|
|
49
|
+
|
|
50
|
+
def _find(self, request: RawRequest):
|
|
51
|
+
return self.cassette.find_unplayed(self.matcher.matches, request)
|
|
52
|
+
|
|
53
|
+
def _replay(self, interaction: Interaction) -> RawResponse:
|
|
54
|
+
r = interaction.response
|
|
55
|
+
return RawResponse(status_code=r.status_code, headers=dict(r.headers), body=_codec.decode_body(r.body))
|
|
56
|
+
|
|
57
|
+
def _record(self, request: RawRequest, response: RawResponse) -> None:
|
|
58
|
+
self.cassette.append(Interaction(
|
|
59
|
+
request=RecordedRequest(
|
|
60
|
+
method=request.method.upper(),
|
|
61
|
+
url=request.url,
|
|
62
|
+
headers=self.redactor.redact_headers(request.headers),
|
|
63
|
+
body=_codec.encode_body(request.body, _ct(request.headers), self.redactor),
|
|
64
|
+
),
|
|
65
|
+
response=RecordedResponse(
|
|
66
|
+
status_code=response.status_code,
|
|
67
|
+
headers=self.redactor.redact_headers(response.headers),
|
|
68
|
+
body=_codec.encode_body(response.body, _ct(response.headers), self.redactor),
|
|
69
|
+
),
|
|
70
|
+
))
|
|
71
|
+
|
|
72
|
+
def _miss(self, request: RawRequest) -> CannotReplay:
|
|
73
|
+
return CannotReplay(f"standin: no recorded interaction for {request.method} {request.url}")
|
|
74
|
+
|
|
75
|
+
# -- entry points -------------------------------------------------------
|
|
76
|
+
def handle(self, request: RawRequest, do_real: SyncReal) -> RawResponse:
|
|
77
|
+
try_replay, record_on_miss = self._plan()
|
|
78
|
+
if try_replay:
|
|
79
|
+
inter = self._find(request)
|
|
80
|
+
if inter is not None:
|
|
81
|
+
return self._replay(inter)
|
|
82
|
+
if not record_on_miss:
|
|
83
|
+
raise self._miss(request)
|
|
84
|
+
response = do_real(request)
|
|
85
|
+
self._record(request, response)
|
|
86
|
+
return response
|
|
87
|
+
|
|
88
|
+
async def handle_async(self, request: RawRequest, do_real: AsyncReal) -> RawResponse:
|
|
89
|
+
try_replay, record_on_miss = self._plan()
|
|
90
|
+
if try_replay:
|
|
91
|
+
inter = self._find(request)
|
|
92
|
+
if inter is not None:
|
|
93
|
+
return self._replay(inter)
|
|
94
|
+
if not record_on_miss:
|
|
95
|
+
raise self._miss(request)
|
|
96
|
+
response = await do_real(request)
|
|
97
|
+
self._record(request, response)
|
|
98
|
+
return response
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def _ct(headers) -> str:
|
|
102
|
+
for k, v in headers.items():
|
|
103
|
+
if k.lower() == "content-type":
|
|
104
|
+
return v
|
|
105
|
+
return ""
|
standin/exceptions.py
ADDED
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
"""Exception hierarchy for standin.
|
|
2
|
+
|
|
3
|
+
All errors derive from :class:`StandinError` so callers can catch the whole
|
|
4
|
+
library with a single ``except``.
|
|
5
|
+
"""
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class StandinError(Exception):
|
|
10
|
+
"""Base class for every error raised by standin."""
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class ConfigError(StandinError):
|
|
14
|
+
"""Invalid configuration (bad mode, conflicting options, ...)."""
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class CassetteError(StandinError):
|
|
18
|
+
"""A cassette file is missing, malformed, or on an unsupported version."""
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class CannotReplay(StandinError):
|
|
22
|
+
"""Replay-only mode encountered a request with no matching recording."""
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
"""Interceptors: pluggable adapters between HTTP clients and the engine."""
|
|
2
|
+
from .base import Interceptor, get_active_engine, reset_active_engine, set_active_engine
|
|
3
|
+
from .httpx_interceptor import HttpxInterceptor
|
|
4
|
+
|
|
5
|
+
__all__ = [
|
|
6
|
+
"Interceptor",
|
|
7
|
+
"HttpxInterceptor",
|
|
8
|
+
"get_active_engine",
|
|
9
|
+
"set_active_engine",
|
|
10
|
+
"reset_active_engine",
|
|
11
|
+
]
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
"""Interceptor abstraction + the active-engine context.
|
|
2
|
+
|
|
3
|
+
An interceptor knows how to splice into one HTTP client and translate its calls
|
|
4
|
+
into the engine's transport-neutral interface. The active engine is stored in a
|
|
5
|
+
``ContextVar`` so recording is correct under threads and asyncio, and so the
|
|
6
|
+
patch stays inert whenever no cassette is open.
|
|
7
|
+
"""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from abc import ABC, abstractmethod
|
|
11
|
+
from contextvars import ContextVar, Token
|
|
12
|
+
|
|
13
|
+
from ..engine import Engine
|
|
14
|
+
|
|
15
|
+
_active_engine: ContextVar[Engine | None] = ContextVar("standin_active_engine", default=None)
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def get_active_engine() -> Engine | None:
|
|
19
|
+
return _active_engine.get()
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def set_active_engine(engine: Engine) -> Token:
|
|
23
|
+
return _active_engine.set(engine)
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def reset_active_engine(token: Token) -> None:
|
|
27
|
+
_active_engine.reset(token)
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
class Interceptor(ABC):
|
|
31
|
+
"""Splices into an HTTP client and routes its traffic through the engine."""
|
|
32
|
+
|
|
33
|
+
@abstractmethod
|
|
34
|
+
def install(self) -> None:
|
|
35
|
+
"""Patch the client. Must be idempotent and inert with no active engine."""
|
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
"""httpx interceptor: routes all httpx traffic through the engine.
|
|
2
|
+
|
|
3
|
+
Patches ``httpx.HTTPTransport.handle_request`` and the async transport once.
|
|
4
|
+
Because OpenAI, Anthropic, Gemini, Mistral, Cohere, litellm, LangChain and
|
|
5
|
+
LlamaIndex all send over httpx, this single hook covers them.
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import threading
|
|
10
|
+
|
|
11
|
+
import httpx
|
|
12
|
+
|
|
13
|
+
from ..models import RawRequest, RawResponse
|
|
14
|
+
from .base import Interceptor, get_active_engine
|
|
15
|
+
|
|
16
|
+
_install_lock = threading.Lock()
|
|
17
|
+
|
|
18
|
+
# Headers to drop when rebuilding a response from already-decoded bytes, or httpx
|
|
19
|
+
# would try to decompress / re-length it a second time.
|
|
20
|
+
_DROP = {"content-encoding", "content-length", "transfer-encoding"}
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def _safe_headers(headers):
|
|
24
|
+
return [(k, v) for k, v in headers.items() if k.lower() not in _DROP]
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _to_httpx(raw: RawResponse, request) -> httpx.Response:
|
|
28
|
+
return httpx.Response(
|
|
29
|
+
status_code=raw.status_code,
|
|
30
|
+
headers=_safe_headers(raw.headers),
|
|
31
|
+
content=raw.body,
|
|
32
|
+
request=request,
|
|
33
|
+
)
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def _to_raw_request(request) -> RawRequest:
|
|
37
|
+
return RawRequest(request.method, str(request.url), dict(request.headers), request.content or b"")
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
class HttpxInterceptor(Interceptor):
|
|
41
|
+
_installed = False
|
|
42
|
+
_orig_sync = None
|
|
43
|
+
_orig_async = None
|
|
44
|
+
|
|
45
|
+
def install(self) -> None:
|
|
46
|
+
cls = HttpxInterceptor
|
|
47
|
+
if cls._installed:
|
|
48
|
+
return
|
|
49
|
+
with _install_lock: # double-checked: patch httpx exactly once, thread-safely
|
|
50
|
+
if cls._installed:
|
|
51
|
+
return
|
|
52
|
+
self._patch()
|
|
53
|
+
|
|
54
|
+
def _patch(self) -> None:
|
|
55
|
+
cls = HttpxInterceptor
|
|
56
|
+
cls._orig_sync = httpx.HTTPTransport.handle_request
|
|
57
|
+
cls._orig_async = httpx.AsyncHTTPTransport.handle_async_request
|
|
58
|
+
orig_sync = cls._orig_sync
|
|
59
|
+
orig_async = cls._orig_async
|
|
60
|
+
|
|
61
|
+
def patched_sync(self, request):
|
|
62
|
+
engine = get_active_engine()
|
|
63
|
+
if engine is None:
|
|
64
|
+
return orig_sync(self, request)
|
|
65
|
+
|
|
66
|
+
def do_real(_raw):
|
|
67
|
+
resp = orig_sync(self, request)
|
|
68
|
+
resp.read()
|
|
69
|
+
return RawResponse(resp.status_code, dict(resp.headers), resp.content)
|
|
70
|
+
|
|
71
|
+
raw_resp = engine.handle(_to_raw_request(request), do_real)
|
|
72
|
+
return _to_httpx(raw_resp, request)
|
|
73
|
+
|
|
74
|
+
async def patched_async(self, request):
|
|
75
|
+
engine = get_active_engine()
|
|
76
|
+
if engine is None:
|
|
77
|
+
return await orig_async(self, request)
|
|
78
|
+
|
|
79
|
+
async def do_real(_raw):
|
|
80
|
+
resp = await orig_async(self, request)
|
|
81
|
+
await resp.aread()
|
|
82
|
+
return RawResponse(resp.status_code, dict(resp.headers), resp.content)
|
|
83
|
+
|
|
84
|
+
raw_resp = await engine.handle_async(_to_raw_request(request), do_real)
|
|
85
|
+
return _to_httpx(raw_resp, request)
|
|
86
|
+
|
|
87
|
+
httpx.HTTPTransport.handle_request = patched_sync # type: ignore[method-assign]
|
|
88
|
+
httpx.AsyncHTTPTransport.handle_async_request = patched_async # type: ignore[method-assign]
|
|
89
|
+
cls._installed = True
|
standin/matching.py
ADDED
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
"""Request matching.
|
|
2
|
+
|
|
3
|
+
The `Matcher` protocol is a single predicate, `matches(live, stored)`. Two
|
|
4
|
+
implementations ship:
|
|
5
|
+
|
|
6
|
+
* ``DefaultMatcher`` — exact match on any subset of method/url/body, and
|
|
7
|
+
JSON-body matching is key-order-insensitive.
|
|
8
|
+
* ``FuzzyMatcher`` — method/url exact, but the body may differ up to a string
|
|
9
|
+
similarity threshold, so a reworded or reformatted prompt still replays.
|
|
10
|
+
|
|
11
|
+
A true embedding/semantic matcher plugs in the same way: implement ``matches``.
|
|
12
|
+
"""
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import difflib
|
|
16
|
+
from collections.abc import Sequence
|
|
17
|
+
from typing import Protocol, runtime_checkable
|
|
18
|
+
|
|
19
|
+
from . import _codec
|
|
20
|
+
from .models import RawRequest, RecordedRequest
|
|
21
|
+
from .redaction import Redactor
|
|
22
|
+
|
|
23
|
+
DEFAULT_MATCH_ON = ("method", "url", "body")
|
|
24
|
+
_SEP = ""
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
@runtime_checkable
|
|
28
|
+
class Matcher(Protocol):
|
|
29
|
+
def matches(self, live: RawRequest, stored: RecordedRequest) -> bool: ...
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
class DefaultMatcher:
|
|
33
|
+
"""Exact match on the chosen fields (JSON body compared order-insensitively)."""
|
|
34
|
+
|
|
35
|
+
def __init__(self, redactor: Redactor, match_on: Sequence[str] = DEFAULT_MATCH_ON):
|
|
36
|
+
self._redactor = redactor
|
|
37
|
+
self._match_on = tuple(match_on)
|
|
38
|
+
|
|
39
|
+
def live_key(self, request: RawRequest) -> str:
|
|
40
|
+
parts = []
|
|
41
|
+
if "method" in self._match_on:
|
|
42
|
+
parts.append(request.method.upper())
|
|
43
|
+
if "url" in self._match_on:
|
|
44
|
+
parts.append(request.url)
|
|
45
|
+
if "body" in self._match_on:
|
|
46
|
+
parts.append(_codec.canonical_live(request.body, _content_type(request.headers), self._redactor))
|
|
47
|
+
return _SEP.join(parts)
|
|
48
|
+
|
|
49
|
+
def stored_key(self, request: RecordedRequest) -> str:
|
|
50
|
+
parts = []
|
|
51
|
+
if "method" in self._match_on:
|
|
52
|
+
parts.append(request.method.upper())
|
|
53
|
+
if "url" in self._match_on:
|
|
54
|
+
parts.append(request.url)
|
|
55
|
+
if "body" in self._match_on:
|
|
56
|
+
parts.append(_codec.canonical_stored(request.body))
|
|
57
|
+
return _SEP.join(parts)
|
|
58
|
+
|
|
59
|
+
def matches(self, live: RawRequest, stored: RecordedRequest) -> bool:
|
|
60
|
+
return self.live_key(live) == self.stored_key(stored)
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
class FuzzyMatcher:
|
|
64
|
+
"""Tolerate small body drift. method/url must match (per match_on); the body
|
|
65
|
+
must be at least ``threshold`` similar (difflib ratio, 0..1). Zero deps."""
|
|
66
|
+
|
|
67
|
+
def __init__(self, redactor: Redactor, match_on: Sequence[str] = DEFAULT_MATCH_ON, threshold: float = 0.9):
|
|
68
|
+
self._redactor = redactor
|
|
69
|
+
self._match_on = tuple(match_on)
|
|
70
|
+
self.threshold = threshold
|
|
71
|
+
|
|
72
|
+
def matches(self, live: RawRequest, stored: RecordedRequest) -> bool:
|
|
73
|
+
if "method" in self._match_on and live.method.upper() != stored.method.upper():
|
|
74
|
+
return False
|
|
75
|
+
if "url" in self._match_on and live.url != stored.url:
|
|
76
|
+
return False
|
|
77
|
+
if "body" in self._match_on:
|
|
78
|
+
lb = _codec.canonical_live(live.body, _content_type(live.headers), self._redactor)
|
|
79
|
+
sb = _codec.canonical_stored(stored.body)
|
|
80
|
+
if lb != sb and difflib.SequenceMatcher(None, lb, sb).ratio() < self.threshold:
|
|
81
|
+
return False
|
|
82
|
+
return True
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def _content_type(headers) -> str:
|
|
86
|
+
for k, v in headers.items():
|
|
87
|
+
if k.lower() == "content-type":
|
|
88
|
+
return v
|
|
89
|
+
return ""
|
standin/models.py
ADDED
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
"""Typed data model shared across layers.
|
|
2
|
+
|
|
3
|
+
Two families of types keep the layers decoupled:
|
|
4
|
+
|
|
5
|
+
* ``RawRequest`` / ``RawResponse`` are the *transport-neutral* representation an
|
|
6
|
+
interceptor hands to the engine. They carry raw bytes and never depend on
|
|
7
|
+
httpx (or any other client).
|
|
8
|
+
* ``RecordedRequest`` / ``RecordedResponse`` / ``Interaction`` are the
|
|
9
|
+
*on-disk* representation, with bodies encoded into a JSON-friendly shape.
|
|
10
|
+
"""
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
from dataclasses import dataclass, field
|
|
14
|
+
from enum import Enum
|
|
15
|
+
from typing import Any
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class Mode(str, Enum):
|
|
19
|
+
"""Record/replay policy for a cassette."""
|
|
20
|
+
|
|
21
|
+
ONCE = "once" # replay if the cassette exists, else record
|
|
22
|
+
NONE = "none" # replay only; error on any unmatched call (use in CI)
|
|
23
|
+
ALL = "all" # always re-record, ignoring existing interactions
|
|
24
|
+
NEW_EPISODES = "new_episodes" # replay matches, record anything new
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
# -- transport-neutral (in flight) --------------------------------------------
|
|
28
|
+
@dataclass
|
|
29
|
+
class RawRequest:
|
|
30
|
+
method: str
|
|
31
|
+
url: str
|
|
32
|
+
headers: dict[str, str]
|
|
33
|
+
body: bytes
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
@dataclass
|
|
37
|
+
class RawResponse:
|
|
38
|
+
status_code: int
|
|
39
|
+
headers: dict[str, str]
|
|
40
|
+
body: bytes
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
# -- persisted (on disk) ------------------------------------------------------
|
|
44
|
+
@dataclass
|
|
45
|
+
class RecordedRequest:
|
|
46
|
+
method: str
|
|
47
|
+
url: str
|
|
48
|
+
headers: dict[str, str] = field(default_factory=dict)
|
|
49
|
+
body: dict[str, Any] = field(default_factory=dict)
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
@dataclass
|
|
53
|
+
class RecordedResponse:
|
|
54
|
+
status_code: int
|
|
55
|
+
headers: dict[str, str] = field(default_factory=dict)
|
|
56
|
+
body: dict[str, Any] = field(default_factory=dict)
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
@dataclass
|
|
60
|
+
class Interaction:
|
|
61
|
+
request: RecordedRequest
|
|
62
|
+
response: RecordedResponse
|
|
63
|
+
played: bool = False
|
standin/py.typed
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
standin/pytest_plugin.py
ADDED
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
"""pytest integration: a ``standin`` fixture and a ``@pytest.mark.standin`` marker.
|
|
2
|
+
|
|
3
|
+
Cassettes default to ``<test-dir>/cassettes/<test-name>.json``. First run records;
|
|
4
|
+
later runs replay. In CI, set ``STANDIN_MODE=none`` to fail on any un-recorded call.
|
|
5
|
+
"""
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
|
|
10
|
+
import pytest
|
|
11
|
+
|
|
12
|
+
from .core import use_cassette
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def pytest_configure(config):
|
|
16
|
+
config.addinivalue_line(
|
|
17
|
+
"markers",
|
|
18
|
+
"standin(path=None, mode='once', match_on=None): record/replay LLM HTTP calls for this test",
|
|
19
|
+
)
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
@pytest.fixture
|
|
23
|
+
def standin(request):
|
|
24
|
+
marker = request.node.get_closest_marker("standin")
|
|
25
|
+
opts = dict(marker.kwargs) if marker else {}
|
|
26
|
+
|
|
27
|
+
name = opts.get("path") or f"{request.node.name}.json"
|
|
28
|
+
path = Path(name)
|
|
29
|
+
if not path.is_absolute():
|
|
30
|
+
path = Path(request.node.fspath).parent / "cassettes" / path
|
|
31
|
+
|
|
32
|
+
kwargs = {"mode": opts.get("mode", "once")}
|
|
33
|
+
if opts.get("match_on"):
|
|
34
|
+
kwargs["match_on"] = opts["match_on"]
|
|
35
|
+
|
|
36
|
+
with use_cassette(path, **kwargs) as cassette:
|
|
37
|
+
yield cassette
|
standin/redaction.py
ADDED
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
"""Redaction: keep credentials out of committed cassettes.
|
|
2
|
+
|
|
3
|
+
``Redactor`` is a Protocol so you can drop in your own policy; ``DefaultRedactor``
|
|
4
|
+
covers auth headers and common secret token formats, and ``NullRedactor`` is a
|
|
5
|
+
no-op for when you explicitly want raw cassettes.
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import re
|
|
10
|
+
from re import Pattern
|
|
11
|
+
from typing import Any, Protocol, runtime_checkable
|
|
12
|
+
|
|
13
|
+
PLACEHOLDER = "[REDACTED]"
|
|
14
|
+
|
|
15
|
+
SENSITIVE_HEADERS = frozenset({
|
|
16
|
+
"authorization", "x-api-key", "api-key", "openai-api-key", "anthropic-api-key",
|
|
17
|
+
"x-goog-api-key", "cookie", "set-cookie", "proxy-authorization",
|
|
18
|
+
})
|
|
19
|
+
|
|
20
|
+
SECRET_PATTERNS: list[Pattern] = [
|
|
21
|
+
re.compile(r"sk-ant-[A-Za-z0-9\-_]{20,}"),
|
|
22
|
+
re.compile(r"sk-(?:proj-)?[A-Za-z0-9]{20,}"),
|
|
23
|
+
re.compile(r"AKIA[0-9A-Z]{16}"),
|
|
24
|
+
re.compile(r"AIza[0-9A-Za-z\-_]{35}"),
|
|
25
|
+
re.compile(r"gh[pousr]_[A-Za-z0-9]{36,}"),
|
|
26
|
+
re.compile(r"xox[baprs]-[A-Za-z0-9-]{10,}"),
|
|
27
|
+
re.compile(r"-----BEGIN (?:RSA |EC |OPENSSH )?PRIVATE KEY-----"),
|
|
28
|
+
]
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
@runtime_checkable
|
|
32
|
+
class Redactor(Protocol):
|
|
33
|
+
def redact_headers(self, headers: dict[str, str]) -> dict[str, str]: ...
|
|
34
|
+
def redact_text(self, text: str) -> str: ...
|
|
35
|
+
def redact_obj(self, obj: Any) -> Any: ...
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class DefaultRedactor:
|
|
39
|
+
"""Masks sensitive headers and secret-looking tokens in bodies."""
|
|
40
|
+
|
|
41
|
+
def __init__(self, extra_headers=(), extra_patterns=()):
|
|
42
|
+
self._headers = SENSITIVE_HEADERS | {h.lower() for h in extra_headers}
|
|
43
|
+
self._patterns = list(SECRET_PATTERNS) + [re.compile(p) for p in extra_patterns]
|
|
44
|
+
|
|
45
|
+
def redact_headers(self, headers: dict[str, str]) -> dict[str, str]:
|
|
46
|
+
return {k: (PLACEHOLDER if k.lower() in self._headers else v) for k, v in headers.items()}
|
|
47
|
+
|
|
48
|
+
def redact_text(self, text: str) -> str:
|
|
49
|
+
for rx in self._patterns:
|
|
50
|
+
text = rx.sub(PLACEHOLDER, text)
|
|
51
|
+
return text
|
|
52
|
+
|
|
53
|
+
def redact_obj(self, obj: Any) -> Any:
|
|
54
|
+
if isinstance(obj, str):
|
|
55
|
+
return self.redact_text(obj)
|
|
56
|
+
if isinstance(obj, dict):
|
|
57
|
+
return {k: self.redact_obj(v) for k, v in obj.items()}
|
|
58
|
+
if isinstance(obj, list):
|
|
59
|
+
return [self.redact_obj(v) for v in obj]
|
|
60
|
+
return obj
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
class NullRedactor:
|
|
64
|
+
"""No-op redactor (used when redaction is turned off)."""
|
|
65
|
+
|
|
66
|
+
def redact_headers(self, headers: dict[str, str]) -> dict[str, str]:
|
|
67
|
+
return dict(headers)
|
|
68
|
+
|
|
69
|
+
def redact_text(self, text: str) -> str:
|
|
70
|
+
return text
|
|
71
|
+
|
|
72
|
+
def redact_obj(self, obj: Any) -> Any:
|
|
73
|
+
return obj
|
standin/storage.py
ADDED
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
"""Cassette persistence.
|
|
2
|
+
|
|
3
|
+
``CassetteStore`` is a Protocol so alternative formats (YAML, a single-file
|
|
4
|
+
archive, ...) can be added without touching the engine. ``JSONCassetteStore``
|
|
5
|
+
writes indented, reviewable JSON.
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import contextlib
|
|
10
|
+
import json
|
|
11
|
+
import os
|
|
12
|
+
import tempfile
|
|
13
|
+
from dataclasses import fields
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
from typing import Protocol, runtime_checkable
|
|
16
|
+
|
|
17
|
+
from .exceptions import CassetteError
|
|
18
|
+
from .models import Interaction, RecordedRequest, RecordedResponse
|
|
19
|
+
|
|
20
|
+
FORMAT_VERSION = 1
|
|
21
|
+
|
|
22
|
+
_REQ_FIELDS = {f.name for f in fields(RecordedRequest)}
|
|
23
|
+
_RESP_FIELDS = {f.name for f in fields(RecordedResponse)}
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
@runtime_checkable
|
|
27
|
+
class CassetteStore(Protocol):
|
|
28
|
+
def load(self, path: Path) -> list[Interaction]: ...
|
|
29
|
+
def save(self, path: Path, interactions: list[Interaction]) -> None: ...
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
class JSONCassetteStore:
|
|
33
|
+
def load(self, path: Path) -> list[Interaction]:
|
|
34
|
+
path = Path(path)
|
|
35
|
+
if not path.exists():
|
|
36
|
+
return []
|
|
37
|
+
try:
|
|
38
|
+
data = json.loads(path.read_text(encoding="utf-8"))
|
|
39
|
+
except ValueError as exc:
|
|
40
|
+
raise CassetteError(f"cassette {path} is not valid JSON: {exc}") from exc
|
|
41
|
+
version = data.get("version")
|
|
42
|
+
if version != FORMAT_VERSION:
|
|
43
|
+
raise CassetteError(f"cassette {path} has unsupported version {version!r}")
|
|
44
|
+
out: list[Interaction] = []
|
|
45
|
+
for idx, item in enumerate(data.get("interactions", [])):
|
|
46
|
+
try:
|
|
47
|
+
req = item["request"]
|
|
48
|
+
resp = item["response"]
|
|
49
|
+
# Ignore unknown keys so newer cassettes stay loadable (forward-compat).
|
|
50
|
+
out.append(Interaction(
|
|
51
|
+
request=RecordedRequest(**{k: v for k, v in req.items() if k in _REQ_FIELDS}),
|
|
52
|
+
response=RecordedResponse(**{k: v for k, v in resp.items() if k in _RESP_FIELDS}),
|
|
53
|
+
))
|
|
54
|
+
except (KeyError, TypeError, AttributeError) as exc:
|
|
55
|
+
raise CassetteError(f"cassette {path} interaction {idx} is malformed: {exc}") from exc
|
|
56
|
+
return out
|
|
57
|
+
|
|
58
|
+
def save(self, path: Path, interactions: list[Interaction]) -> None:
|
|
59
|
+
path = Path(path)
|
|
60
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
61
|
+
data = {
|
|
62
|
+
"version": FORMAT_VERSION,
|
|
63
|
+
"recorded_with": "standin",
|
|
64
|
+
"interactions": [
|
|
65
|
+
{
|
|
66
|
+
"request": vars(i.request),
|
|
67
|
+
"response": vars(i.response),
|
|
68
|
+
}
|
|
69
|
+
for i in interactions
|
|
70
|
+
],
|
|
71
|
+
}
|
|
72
|
+
text = json.dumps(data, indent=2, ensure_ascii=False)
|
|
73
|
+
# Atomic write: never leave a half-written cassette if interrupted.
|
|
74
|
+
fd, tmp = tempfile.mkstemp(dir=str(path.parent), prefix=path.name + ".", suffix=".tmp")
|
|
75
|
+
try:
|
|
76
|
+
with os.fdopen(fd, "w", encoding="utf-8") as fh:
|
|
77
|
+
fh.write(text)
|
|
78
|
+
os.replace(tmp, path)
|
|
79
|
+
except BaseException:
|
|
80
|
+
with contextlib.suppress(OSError):
|
|
81
|
+
os.unlink(tmp)
|
|
82
|
+
raise
|
|
@@ -0,0 +1,182 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: standin
|
|
3
|
+
Version: 0.2.0
|
|
4
|
+
Summary: A stand-in for the real LLM in your tests. Record LLM API calls once, replay them forever: fast, free, deterministic, offline.
|
|
5
|
+
Project-URL: Homepage, https://github.com/eeshsaxena/standin
|
|
6
|
+
Project-URL: Source, https://github.com/eeshsaxena/standin
|
|
7
|
+
Project-URL: Issues, https://github.com/eeshsaxena/standin/issues
|
|
8
|
+
Project-URL: Changelog, https://github.com/eeshsaxena/standin/blob/main/CHANGELOG.md
|
|
9
|
+
Author-email: Eesh Saxena <eeshsaxena@gmail.com>
|
|
10
|
+
License-Expression: MIT
|
|
11
|
+
License-File: LICENSE
|
|
12
|
+
Keywords: agents,ai,anthropic,cassette,deterministic,llm,mock,openai,pytest,record,replay,testing,vcr
|
|
13
|
+
Classifier: Development Status :: 4 - Beta
|
|
14
|
+
Classifier: Framework :: Pytest
|
|
15
|
+
Classifier: Intended Audience :: Developers
|
|
16
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
17
|
+
Classifier: Programming Language :: Python :: 3
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
22
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
23
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
24
|
+
Classifier: Topic :: Software Development :: Testing
|
|
25
|
+
Classifier: Typing :: Typed
|
|
26
|
+
Requires-Python: >=3.9
|
|
27
|
+
Requires-Dist: httpx>=0.23
|
|
28
|
+
Provides-Extra: dev
|
|
29
|
+
Requires-Dist: anthropic>=0.30; extra == 'dev'
|
|
30
|
+
Requires-Dist: hypothesis>=6; extra == 'dev'
|
|
31
|
+
Requires-Dist: mypy>=1.8; extra == 'dev'
|
|
32
|
+
Requires-Dist: openai>=1.0; extra == 'dev'
|
|
33
|
+
Requires-Dist: pytest>=7; extra == 'dev'
|
|
34
|
+
Requires-Dist: ruff>=0.5; extra == 'dev'
|
|
35
|
+
Description-Content-Type: text/markdown
|
|
36
|
+
|
|
37
|
+
# standin
|
|
38
|
+
|
|
39
|
+
**A stand-in for the real LLM in your tests.** Record your LLM API calls once, then replay them forever: fast, free, deterministic, and fully offline. One line, any provider.
|
|
40
|
+
|
|
41
|
+
<p align="center"><img src="demo/standin.gif" alt="standin: record LLM calls once, replay them instantly and offline" width="820"></p>
|
|
42
|
+
|
|
43
|
+
```python
|
|
44
|
+
import standin
|
|
45
|
+
|
|
46
|
+
with standin.use_cassette("tests/cassettes/summary.json"):
|
|
47
|
+
reply = client.chat.completions.create(model="gpt-4o", messages=[...])
|
|
48
|
+
# First run: hits the real API and records it.
|
|
49
|
+
# Every run after: replayed from disk. No network, no cost, same answer.
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
Your LLM tests are slow, flaky, and cost money because they hit real APIs. `standin` makes them **deterministic and offline** by recording the real HTTP calls once and replaying them after, with the things LLM devs actually need: **streaming**, **tool-calls**, **secret redaction**, and **body-aware matching**.
|
|
53
|
+
|
|
54
|
+
---
|
|
55
|
+
|
|
56
|
+
## Why not just VCR.py?
|
|
57
|
+
|
|
58
|
+
VCR.py is great, but it's a general HTTP tool. `standin` is built for LLMs:
|
|
59
|
+
|
|
60
|
+
- **Provider-agnostic, zero wiring.** It hooks `httpx` under the hood, so it works with **OpenAI, Anthropic, Gemini, Mistral, Cohere, litellm, LangChain, LlamaIndex** — anything that sends over httpx. No per-SDK adapters.
|
|
61
|
+
- **Streaming just works.** Server-sent event (SSE) responses are recorded and replayed intact.
|
|
62
|
+
- **Safe to commit.** API keys in headers and secret-looking tokens in bodies are **redacted automatically**, so cassettes can live in a public repo.
|
|
63
|
+
- **Body-aware matching.** Requests match on normalized JSON, so key ordering and formatting noise don't break replays. Repeated identical calls (agent loops) replay in order.
|
|
64
|
+
- **One-line pytest fixture**, with sane auto-named cassettes.
|
|
65
|
+
- **Clean, typed, extensible core** (see [ARCHITECTURE.md](ARCHITECTURE.md)) — swap the matcher, redactor, or storage backend.
|
|
66
|
+
|
|
67
|
+
## Install
|
|
68
|
+
|
|
69
|
+
```bash
|
|
70
|
+
pip install standin
|
|
71
|
+
```
|
|
72
|
+
|
|
73
|
+
Python 3.9+ and `httpx` (already a dependency of the major LLM SDKs).
|
|
74
|
+
|
|
75
|
+
## Quickstart
|
|
76
|
+
|
|
77
|
+
### With pytest (recommended)
|
|
78
|
+
|
|
79
|
+
```python
|
|
80
|
+
import pytest
|
|
81
|
+
|
|
82
|
+
@pytest.mark.standin # cassette auto-named tests/cassettes/test_summarize.json
|
|
83
|
+
def test_summarize(standin):
|
|
84
|
+
out = summarize("war and peace") # your code that calls an LLM
|
|
85
|
+
assert "Napoleon" in out
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
First run records against the real API; every run after replays from the cassette. Commit the cassette and teammates (and CI) run the test with **no keys and no network**.
|
|
89
|
+
|
|
90
|
+
### Anywhere (context manager)
|
|
91
|
+
|
|
92
|
+
```python
|
|
93
|
+
import standin
|
|
94
|
+
from openai import OpenAI
|
|
95
|
+
|
|
96
|
+
client = OpenAI()
|
|
97
|
+
with standin.use_cassette("tests/cassettes/haiku.json"):
|
|
98
|
+
resp = client.chat.completions.create(
|
|
99
|
+
model="gpt-4o-mini",
|
|
100
|
+
messages=[{"role": "user", "content": "haiku about testing"}],
|
|
101
|
+
)
|
|
102
|
+
```
|
|
103
|
+
|
|
104
|
+
## Modes
|
|
105
|
+
|
|
106
|
+
| Mode | Behavior |
|
|
107
|
+
| --- | --- |
|
|
108
|
+
| `once` (default) | Replay if the cassette exists, otherwise record it. |
|
|
109
|
+
| `none` | Replay only. **Errors on any unrecorded call** — use this in CI. |
|
|
110
|
+
| `all` | Always re-record, ignoring existing interactions. |
|
|
111
|
+
| `new_episodes` | Replay what's recorded, record anything new (great for agent loops). |
|
|
112
|
+
|
|
113
|
+
**In CI**, force replay-only for the whole run so a stray live call fails loudly:
|
|
114
|
+
|
|
115
|
+
```bash
|
|
116
|
+
STANDIN_MODE=none pytest
|
|
117
|
+
```
|
|
118
|
+
|
|
119
|
+
## What a cassette looks like
|
|
120
|
+
|
|
121
|
+
Plain, reviewable JSON, secrets already stripped:
|
|
122
|
+
|
|
123
|
+
```json
|
|
124
|
+
{
|
|
125
|
+
"version": 1,
|
|
126
|
+
"recorded_with": "standin",
|
|
127
|
+
"interactions": [
|
|
128
|
+
{
|
|
129
|
+
"request": {
|
|
130
|
+
"method": "POST",
|
|
131
|
+
"url": "https://api.openai.com/v1/chat/completions",
|
|
132
|
+
"headers": { "authorization": "[REDACTED]" },
|
|
133
|
+
"body": { "json": { "model": "gpt-4o-mini", "messages": [ ] } }
|
|
134
|
+
},
|
|
135
|
+
"response": { "status_code": 200, "body": { "json": { "choices": [ ] } } }
|
|
136
|
+
}
|
|
137
|
+
]
|
|
138
|
+
}
|
|
139
|
+
```
|
|
140
|
+
|
|
141
|
+
## Extending it
|
|
142
|
+
|
|
143
|
+
Everything is a small protocol you can replace (see [ARCHITECTURE.md](ARCHITECTURE.md)):
|
|
144
|
+
|
|
145
|
+
```python
|
|
146
|
+
standin.use_cassette(path, matcher=MyMatcher(), redactor=MyRedactor(), store=MyStore())
|
|
147
|
+
```
|
|
148
|
+
|
|
149
|
+
- **Matcher** — decide when a live request equals a recorded one. Ships with
|
|
150
|
+
`DefaultMatcher` (exact) and `FuzzyMatcher` (body may drift up to a similarity
|
|
151
|
+
threshold, so a reworded prompt still replays):
|
|
152
|
+
|
|
153
|
+
```python
|
|
154
|
+
from standin import use_cassette, FuzzyMatcher, DefaultRedactor
|
|
155
|
+
with use_cassette(path, matcher=FuzzyMatcher(DefaultRedactor(), threshold=0.9)):
|
|
156
|
+
...
|
|
157
|
+
```
|
|
158
|
+
- **Redactor** — control what gets scrubbed before writing.
|
|
159
|
+
- **CassetteStore** — change the on-disk format.
|
|
160
|
+
|
|
161
|
+
## Command line
|
|
162
|
+
|
|
163
|
+
```bash
|
|
164
|
+
standin list tests/cassettes/summary.json # one line per interaction
|
|
165
|
+
standin show tests/cassettes/summary.json 0 # full request/response
|
|
166
|
+
standin stats tests/cassettes/summary.json # counts by method/status
|
|
167
|
+
standin scrub tests/cassettes/summary.json # re-run secret redaction in place
|
|
168
|
+
```
|
|
169
|
+
|
|
170
|
+
## Roadmap
|
|
171
|
+
|
|
172
|
+
- Embedding-based semantic matching (a `Matcher` you drop in; `FuzzyMatcher`
|
|
173
|
+
already covers string-similarity drift today).
|
|
174
|
+
- `requests` / `aiohttp` interceptors (the engine is already transport-neutral).
|
|
175
|
+
|
|
176
|
+
## Contributing
|
|
177
|
+
|
|
178
|
+
See [CONTRIBUTING.md](CONTRIBUTING.md). Run the suite with `pytest`, lint with `ruff`, type-check with `mypy`.
|
|
179
|
+
|
|
180
|
+
## License
|
|
181
|
+
|
|
182
|
+
MIT. See [LICENSE](LICENSE).
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
standin/__init__.py,sha256=0el2rfz9vyDhB98fAvvXaebp9rxON6NlB53RxkTGirA,1361
|
|
2
|
+
standin/_codec.py,sha256=UD9B1waw36wAartWf_Rl2b7-NVLlXj82NjtgeSp2j10,2658
|
|
3
|
+
standin/cassette.py,sha256=OzY9477oB4Pcr1XXFIV_1gR6JbF4au0dN_y6ZAG2gUM,1468
|
|
4
|
+
standin/cli.py,sha256=Xa7YixVGXU2CWF2f47KlW-uy8XY-RZwVh1IN0OaMHYg,3611
|
|
5
|
+
standin/config.py,sha256=vyUS827KzCMa2of236_Y2MPBF1hw0kawB9-TK4mm-3M,1527
|
|
6
|
+
standin/core.py,sha256=nJUOvqkVE3CsoRu-B65tuT0TdEJwyIKZ9g4DrYpVJH0,2044
|
|
7
|
+
standin/engine.py,sha256=vRyq-n9l5tEC5CvOBKSr1cHtPyFkQOPQG-nXbiOrXQU,3968
|
|
8
|
+
standin/exceptions.py,sha256=favxY3MwMVMkEhlF1JDP_zEksUtI0gr53AJK3QcxfJE,609
|
|
9
|
+
standin/matching.py,sha256=xdEl7XPK9FwjmAsdVqzOgbgIuMgF8EH5s8ef_1CbY1U,3250
|
|
10
|
+
standin/models.py,sha256=OeO6n-kXP7j8yd9vmfVVDdkRbQ6q4wGYKJOKEpL8U_0,1733
|
|
11
|
+
standin/py.typed,sha256=AbpHGcgLb-kRsJGnwFEktk7uzpZOCcBY74-YBdrKVGs,1
|
|
12
|
+
standin/pytest_plugin.py,sha256=ysmYQXY7jcM47R1lV8-B8jf5YZyA9gtRYHH4RSBfNvg,1079
|
|
13
|
+
standin/redaction.py,sha256=ZvSucGoZNITk2LJ9e9TRhDY16KfFWUIFUXfEMnN_lYg,2500
|
|
14
|
+
standin/storage.py,sha256=_rslSlrM0UA7TWpEleY7WS11hZBNOalHhhcycMU2_dE,3070
|
|
15
|
+
standin/interceptors/__init__.py,sha256=GVcw1j7vas4I-mCFV7SRnH_ltLw1NIcO-PEEZmXJfDQ,348
|
|
16
|
+
standin/interceptors/base.py,sha256=OyOBd9lmte4DhaWJzcWaZMdh2ZtEkz3Is_C6tKAEmI0,1072
|
|
17
|
+
standin/interceptors/httpx_interceptor.py,sha256=Uy9d4nEzfAW5KJO6_1hrB_gggB1Xo-v5fnWRUzHTHsc,3041
|
|
18
|
+
standin-0.2.0.dist-info/METADATA,sha256=cc9ysMQ9hpUTz48PX3ljDIsm2-oxh_RRlEFuafQuEW8,6818
|
|
19
|
+
standin-0.2.0.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
|
|
20
|
+
standin-0.2.0.dist-info/entry_points.txt,sha256=Ir7NvIREgtd5MUgDU53ARxmMUSSNDMSg4WKP6Rjz5gQ,89
|
|
21
|
+
standin-0.2.0.dist-info/licenses/LICENSE,sha256=N2YCKaHPyHHbKbWBXzYv8QsWzFPmj_IRoqhL6NSn7QQ,1068
|
|
22
|
+
standin-0.2.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Eesh Saxena
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|