lingua-proxy 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,5 @@
1
+ """lingua-proxy: translate non-English LLM traffic to English to cut token costs."""
2
+
3
+ __version__ = "0.1.0"
4
+
5
+ __all__ = ["__version__"]
lingua_proxy/bench.py ADDED
@@ -0,0 +1,185 @@
1
+ """Measure whether the proxy actually saves money.
2
+
3
+ The premise of this project is a claim about cost, so it ships with the means
4
+ to check that claim against a real upstream, on a corpus that deliberately
5
+ includes workloads where translation is expected to *lose*: three-word prompts
6
+ whose translation fee dwarfs the request, and code-heavy payloads that barely
7
+ compress.
8
+
9
+ Savings are reported in dollars rather than tokens, because the translator
10
+ runs on a cheap model and the request on an expensive one. A verdict of
11
+ ``loses`` is a valid, expected result for some categories.
12
+ """
13
+
14
+ from __future__ import annotations
15
+
16
+ import json
17
+ import pathlib
18
+ from dataclasses import dataclass, field
19
+
20
+ from rich.console import Console
21
+ from rich.table import Table
22
+
23
+ CORPUS_PATH = pathlib.Path(__file__).with_name("bench_corpus.jsonl")
24
+
25
+ #: A category has to clear this to count as worth using.
26
+ PAYS_OFF = 0.25
27
+
28
+
29
+ def load_corpus(path: pathlib.Path | None = None) -> list[dict]:
30
+ """Read the bundled prompt corpus."""
31
+ target = path or CORPUS_PATH
32
+ rows = []
33
+ for line in target.read_text(encoding="utf-8").splitlines():
34
+ line = line.strip()
35
+ if line:
36
+ rows.append(json.loads(line))
37
+ return rows
38
+
39
+
40
+ def verdict_for(ratio: float) -> str:
41
+ if ratio >= PAYS_OFF:
42
+ return "pays off"
43
+ if ratio > 0:
44
+ return "marginal"
45
+ return "loses"
46
+
47
+
48
+ @dataclass
49
+ class BenchResult:
50
+ """One prompt's measurement."""
51
+
52
+ id: str
53
+ lang: str
54
+ category: str
55
+ baseline_cost: float = 0.0
56
+ proxied_cost: float = 0.0
57
+ translator_cost: float = 0.0
58
+ latency_ms: float = 0.0
59
+ error: str | None = None
60
+ detail: dict = field(default_factory=dict)
61
+
62
+ @property
63
+ def ok(self) -> bool:
64
+ return self.error is None
65
+
66
+ @property
67
+ def dollars_saved(self) -> float:
68
+ return self.baseline_cost - self.proxied_cost
69
+
70
+
71
+ def _bucket(results: list[BenchResult]) -> dict:
72
+ baseline = sum(r.baseline_cost for r in results)
73
+ proxied = sum(r.proxied_cost for r in results)
74
+ translator = sum(r.translator_cost for r in results)
75
+ ratio = (baseline - proxied) / baseline if baseline > 0 else 0.0
76
+ return {
77
+ "n": len(results),
78
+ "baseline_cost": round(baseline, 6),
79
+ "proxied_cost": round(proxied, 6),
80
+ "translator_cost": round(translator, 6),
81
+ "dollars_saved": round(baseline - proxied, 6),
82
+ "savings_ratio": round(ratio, 4),
83
+ "verdict": verdict_for(ratio) if results else "no data",
84
+ }
85
+
86
+
87
+ def build_report(results: list[BenchResult]) -> dict:
88
+ """Aggregate per-category and overall figures."""
89
+ good = [r for r in results if r.ok]
90
+
91
+ categories: dict[str, list[BenchResult]] = {}
92
+ for result in good:
93
+ categories.setdefault(result.category, []).append(result)
94
+
95
+ return {
96
+ "overall": _bucket(good),
97
+ "categories": {name: _bucket(rows) for name, rows in sorted(categories.items())},
98
+ "errors": sum(1 for r in results if not r.ok),
99
+ # A truncated reply pins both runs to the same output length and hides
100
+ # real savings, so the count is surfaced rather than buried.
101
+ "truncated": sum(1 for r in good if r.detail.get("truncated")),
102
+ }
103
+
104
+
105
+ def exit_code_for(report: dict, *, min_savings: float) -> int:
106
+ """Non-zero when the headline claim does not hold on real traffic."""
107
+ chat = report["categories"].get("chat")
108
+ if not chat or chat["n"] == 0:
109
+ return 0
110
+ return 0 if chat["savings_ratio"] >= min_savings else 1
111
+
112
+
113
+ def render(report: dict, console: Console | None = None) -> None:
114
+ console = console or Console()
115
+ table = Table(title="lingua-proxy savings benchmark")
116
+ table.add_column("Category")
117
+ table.add_column("n", justify="right")
118
+ table.add_column("Without proxy", justify="right")
119
+ table.add_column("With proxy", justify="right")
120
+ table.add_column("Saved", justify="right")
121
+ table.add_column("Verdict")
122
+
123
+ for name, bucket in report["categories"].items():
124
+ colour = {"pays off": "green", "marginal": "yellow", "loses": "red"}.get(
125
+ bucket["verdict"], "white"
126
+ )
127
+ table.add_row(
128
+ name,
129
+ str(bucket["n"]),
130
+ f"${bucket['baseline_cost']:.5f}",
131
+ f"${bucket['proxied_cost']:.5f}",
132
+ f"{bucket['savings_ratio'] * 100:.1f}%",
133
+ f"[{colour}]{bucket['verdict']}[/{colour}]",
134
+ )
135
+
136
+ overall = report["overall"]
137
+ table.add_section()
138
+ table.add_row(
139
+ "[bold]overall[/bold]",
140
+ str(overall["n"]),
141
+ f"${overall['baseline_cost']:.5f}",
142
+ f"${overall['proxied_cost']:.5f}",
143
+ f"{overall['savings_ratio'] * 100:.1f}%",
144
+ overall["verdict"],
145
+ )
146
+ console.print(table)
147
+
148
+ if report["errors"]:
149
+ console.print(f"[yellow]{report['errors']} prompt(s) failed and were excluded.[/yellow]")
150
+
151
+ if report.get("truncated"):
152
+ console.print(
153
+ f"[yellow]{report['truncated']} reply/replies hit the token ceiling.[/yellow] "
154
+ "Truncated replies understate savings, because both runs are forced to the "
155
+ "same output length. Raise max_tokens for a cleaner measurement."
156
+ )
157
+
158
+
159
+ def run_bench(*, mode: str = "estimate", min_savings: float = 0.30, as_json: bool = False) -> int:
160
+ """Run the benchmark against a configured upstream.
161
+
162
+ Requires an explicit upstream: silently defaulting to the public API would
163
+ spend the user's money without asking.
164
+ """
165
+ import os
166
+
167
+ console = Console()
168
+ upstream = os.environ.get("LINGUA_BENCH_BASE_URL")
169
+ if not upstream:
170
+ console.print(
171
+ "[red]No benchmark upstream configured.[/red]\n"
172
+ "Set LINGUA_BENCH_BASE_URL (and LINGUA_BENCH_AUTH if it needs a credential).\n"
173
+ "This command sends real requests and spends real money, so it will not "
174
+ "guess an endpoint for you."
175
+ )
176
+ return 2
177
+
178
+ from lingua_proxy.bench_runner import run_live_bench
179
+
180
+ report = run_live_bench(upstream=upstream, mode=mode)
181
+ if as_json:
182
+ print(json.dumps(report, indent=2))
183
+ else:
184
+ render(report, console)
185
+ return exit_code_for(report, min_savings=min_savings)
@@ -0,0 +1,21 @@
1
+ {"id": "chat-ko-1", "lang": "ko", "category": "chat", "prompt": "파이썬에서 리스트와 튜플의 차이가 뭔지 설명해 주고, 각각 어떤 상황에서 쓰는 게 좋은지 예시와 함께 알려줘."}
2
+ {"id": "chat-ko-2", "lang": "ko", "category": "chat", "prompt": "우리 서비스 응답 속도가 최근에 느려졌는데, 원인을 찾으려면 어떤 지표부터 확인해야 할까? 순서대로 알려줘."}
3
+ {"id": "chat-ko-3", "lang": "ko", "category": "chat", "prompt": "데이터베이스 인덱스를 추가하면 항상 빨라지는 건 아니라고 들었어. 어떤 경우에 오히려 느려지는지 설명해 줘."}
4
+ {"id": "chat-ja-1", "lang": "ja", "category": "chat", "prompt": "非同期処理と並行処理の違いを、実際のコード例を交えながら初心者にも分かるように説明してください。"}
5
+ {"id": "chat-ja-2", "lang": "ja", "category": "chat", "prompt": "チームでコードレビューを導入したいのですが、最初に決めておくべきルールを優先度順に教えてください。"}
6
+ {"id": "chat-ja-3", "lang": "ja", "category": "chat", "prompt": "本番環境で発生した障害の原因調査を行う際、ログのどこから見ていくのが効率的でしょうか。"}
7
+ {"id": "chat-zh-1", "lang": "zh", "category": "chat", "prompt": "请解释一下什么是数据库事务的隔离级别,以及在实际项目中应该如何选择合适的级别。"}
8
+ {"id": "chat-zh-2", "lang": "zh", "category": "chat", "prompt": "我们的接口偶尔会超时,但重试之后就正常了。请分析可能的原因并给出排查思路。"}
9
+ {"id": "chat-zh-3", "lang": "zh", "category": "chat", "prompt": "微服务架构相比单体架构有哪些代价?请从团队规模和运维成本的角度说明。"}
10
+ {"id": "chat-ar-1", "lang": "ar", "category": "chat", "prompt": "اشرح لي الفرق بين التخزين المؤقت على مستوى التطبيق والتخزين المؤقت على مستوى قاعدة البيانات، ومتى أستخدم كل منهما."}
11
+ {"id": "chat-ar-2", "lang": "ar", "category": "chat", "prompt": "ما هي أفضل الممارسات لتأمين واجهة برمجة التطبيقات العامة؟ اذكرها مرتبة حسب الأهمية."}
12
+ {"id": "chat-ar-3", "lang": "ar", "category": "chat", "prompt": "أواجه بطئاً في تحميل الصفحة الرئيسية لموقعي. ما الخطوات التي يجب أن أتبعها لتشخيص المشكلة؟"}
13
+ {"id": "short-ko-1", "lang": "ko", "category": "short", "prompt": "이거 고쳐줘"}
14
+ {"id": "short-ja-1", "lang": "ja", "category": "short", "prompt": "これを直して"}
15
+ {"id": "short-zh-1", "lang": "zh", "category": "short", "prompt": "帮我修一下"}
16
+ {"id": "long-ko-1", "lang": "ko", "category": "long_form", "prompt": "우리 팀이 새로 도입할 코드 리뷰 프로세스에 대한 문서를 작성해 줘. 배경, 목표, 단계별 절차, 예외 상황 처리, 그리고 도입 후 3개월 시점에 점검할 지표까지 포함해서 자세히 써 줘."}
17
+ {"id": "long-ja-1", "lang": "ja", "category": "long_form", "prompt": "新しいメンバー向けのオンボーディング資料を作成してください。開発環境の構築手順、コードベースの全体像、レビュー文化、よくあるつまずきポイントとその対処法を含めて、詳しく書いてください。"}
18
+ {"id": "long-zh-1", "lang": "zh", "category": "long_form", "prompt": "请为我们的服务撰写一份完整的故障应急预案文档,包括故障分级标准、各级别的响应流程、责任人分工、对外沟通模板,以及事后复盘的具体步骤。"}
19
+ {"id": "code-ko-1", "lang": "ko", "category": "code_heavy", "prompt": "이 함수 성능을 개선해줘:\n\n```python\ndef process_records(records, threshold):\n results = []\n seen = set()\n for record in records:\n key = record.get(\"id\")\n if key in seen:\n continue\n seen.add(key)\n score = 0\n for field in (\"alpha\", \"beta\", \"gamma\"):\n value = record.get(field)\n if value is None:\n continue\n score += value * WEIGHTS[field]\n if score > threshold:\n results.append({\"id\": key, \"score\": score})\n results.sort(key=lambda item: item[\"score\"], reverse=True)\n return results\n```"}
20
+ {"id": "code-ja-1", "lang": "ja", "category": "code_heavy", "prompt": "この関数のバグを見つけてください:\n\n```javascript\nexport function buildIndex(documents) {\n const index = new Map();\n for (const doc of documents) {\n const tokens = doc.text.toLowerCase().split(/\\W+/);\n for (const token of tokens) {\n if (!token) continue;\n if (!index.has(token)) index.set(token, new Set());\n index.get(token).add(doc.id);\n }\n }\n return index;\n}\n```"}
21
+ {"id": "code-zh-1", "lang": "zh", "category": "code_heavy", "prompt": "请帮我优化这个查询:\n\n```sql\nSELECT u.id, u.email, COUNT(o.id) AS order_count, SUM(o.total) AS lifetime_value\nFROM users u\nLEFT JOIN orders o ON o.user_id = u.id AND o.status = 'completed'\nWHERE u.created_at >= NOW() - INTERVAL '90 days'\nGROUP BY u.id, u.email\nHAVING COUNT(o.id) > 3\nORDER BY lifetime_value DESC\nLIMIT 100;\n```"}
@@ -0,0 +1,109 @@
1
+ """Live benchmark execution.
2
+
3
+ Each corpus prompt is sent twice through the proxy: once normally, and once
4
+ with ``x-lingua-bypass`` so the model answers in the original language. That
5
+ gives a measured baseline instead of an estimated one, which matters because
6
+ an estimate is exactly the thing a sceptical reader would not trust.
7
+
8
+ Every request costs money, so this is opt-in and never runs by default.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ import os
14
+ import time
15
+
16
+ import httpx
17
+
18
+ from lingua_proxy.bench import BenchResult, build_report, load_corpus
19
+ from lingua_proxy.codecs import AnthropicMessagesCodec
20
+ from lingua_proxy.config import Settings
21
+ from lingua_proxy.cost_log import price_for
22
+ from lingua_proxy.proxy import create_app
23
+
24
+ DEFAULT_MODEL = "claude-sonnet-4-6"
25
+ # Generous enough that ordinary replies finish naturally. If a reply is cut
26
+ # off at the ceiling, both runs produce identical output token counts and the
27
+ # measurement understates savings, because output is where most cost lives.
28
+ MAX_TOKENS = 4096
29
+
30
+
31
+ def _auth_headers() -> dict[str, str]:
32
+ headers = {"anthropic-version": "2023-06-01", "content-type": "application/json"}
33
+ token = os.environ.get("LINGUA_BENCH_AUTH")
34
+ if token:
35
+ if token.startswith("sk-"):
36
+ headers["x-api-key"] = token
37
+ else:
38
+ headers["authorization"] = f"Bearer {token}"
39
+ return headers
40
+
41
+
42
+ async def _one(
43
+ client: httpx.AsyncClient,
44
+ row: dict,
45
+ model: str,
46
+ codec: AnthropicMessagesCodec,
47
+ ) -> BenchResult:
48
+ body = {
49
+ "model": model,
50
+ "max_tokens": MAX_TOKENS,
51
+ "temperature": 0,
52
+ "messages": [{"role": "user", "content": row["prompt"]}],
53
+ }
54
+ headers = _auth_headers()
55
+ price = price_for(model)
56
+
57
+ started = time.monotonic()
58
+ try:
59
+ # Baseline: the model answers in the original language.
60
+ native = await client.post(
61
+ "/v1/messages", json=body, headers={**headers, "x-lingua-bypass": "true"}, timeout=120
62
+ )
63
+ native.raise_for_status()
64
+ baseline_usage = codec.usage(native.json())
65
+
66
+ # Proxied: translated in and back out.
67
+ proxied = await client.post("/v1/messages", json=body, headers=headers, timeout=180)
68
+ proxied.raise_for_status()
69
+ proxied_usage = codec.usage(proxied.json())
70
+ except Exception as exc: # noqa: BLE001 - one bad prompt must not end the run
71
+ return BenchResult(row["id"], row["lang"], row["category"], error=str(exc)[:200])
72
+
73
+ return BenchResult(
74
+ id=row["id"],
75
+ lang=row["lang"],
76
+ category=row["category"],
77
+ baseline_cost=price.cost(baseline_usage),
78
+ proxied_cost=price.cost(proxied_usage),
79
+ latency_ms=(time.monotonic() - started) * 1000,
80
+ detail={
81
+ "baseline_input": baseline_usage.input_tokens,
82
+ "baseline_output": baseline_usage.output_tokens,
83
+ "proxied_input": proxied_usage.input_tokens,
84
+ "proxied_output": proxied_usage.output_tokens,
85
+ "truncated": native.json().get("stop_reason") == "max_tokens"
86
+ or proxied.json().get("stop_reason") == "max_tokens",
87
+ },
88
+ )
89
+
90
+
91
+ def run_live_bench(*, upstream: str, mode: str = "estimate", model: str | None = None) -> dict:
92
+ """Run every corpus prompt through an in-process proxy against ``upstream``."""
93
+ import asyncio
94
+
95
+ model = model or os.environ.get("LINGUA_BENCH_MODEL", DEFAULT_MODEL)
96
+ settings = Settings(upstream_anthropic_url=upstream, memo_persist=False)
97
+ app = create_app(settings)
98
+ codec = AnthropicMessagesCodec()
99
+
100
+ async def go() -> list[BenchResult]:
101
+ results = []
102
+ async with httpx.AsyncClient(
103
+ transport=httpx.ASGITransport(app=app), base_url="http://bench.test"
104
+ ) as client:
105
+ for row in load_corpus():
106
+ results.append(await _one(client, row, model, codec))
107
+ return results
108
+
109
+ return build_report(asyncio.run(go()))
lingua_proxy/cli.py ADDED
@@ -0,0 +1,301 @@
1
+ """Command line interface."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import json
6
+ import os
7
+ import subprocess
8
+
9
+ import click
10
+ import httpx
11
+ from rich.console import Console
12
+ from rich.table import Table
13
+
14
+ from lingua_proxy import __version__
15
+ from lingua_proxy.config import (
16
+ Settings,
17
+ default_config_path,
18
+ lingua_home,
19
+ resolve_anthropic_upstream,
20
+ resolve_openai_upstream,
21
+ )
22
+ from lingua_proxy.cost_log import CostLog, summarize
23
+ from lingua_proxy.doctor import run_checks
24
+ from lingua_proxy.wrap import (
25
+ WrapError,
26
+ apply_wrap,
27
+ clear_marker,
28
+ find_claude_binary,
29
+ load_marker,
30
+ restore_wrap,
31
+ )
32
+
33
+ console = Console()
34
+
35
+
36
+ @click.group(context_settings={"help_option_names": ["-h", "--help"]})
37
+ @click.version_option(__version__, prog_name="lingua-proxy")
38
+ def main() -> None:
39
+ """Translate non-English LLM traffic to English to cut token costs."""
40
+
41
+
42
+ # -- proxy --------------------------------------------------------------
43
+
44
+
45
+ @main.command()
46
+ @click.option("--host", default=None, help="Bind address (loopback by default).")
47
+ @click.option("--port", type=int, default=None, help="Port to listen on.")
48
+ @click.option("--upstream", default=None, help="Anthropic-format upstream base URL.")
49
+ @click.option("--openai-upstream", default=None, help="OpenAI-format upstream base URL.")
50
+ @click.option("--translator-model", default=None, help="Model used for translation.")
51
+ def proxy(
52
+ host: str | None,
53
+ port: int | None,
54
+ upstream: str | None,
55
+ openai_upstream: str | None,
56
+ translator_model: str | None,
57
+ ) -> None:
58
+ """Run the translating proxy."""
59
+ import uvicorn
60
+
61
+ from lingua_proxy.proxy import create_app
62
+
63
+ settings = Settings(
64
+ upstream_anthropic_url=resolve_anthropic_upstream(upstream),
65
+ upstream_openai_url=resolve_openai_upstream(openai_upstream),
66
+ )
67
+ if host:
68
+ settings.host = host
69
+ if port:
70
+ settings.port = port
71
+ if translator_model:
72
+ settings.translator_model = translator_model
73
+
74
+ console.print(f"[bold]lingua-proxy[/bold] {__version__}")
75
+ console.print(f" listening on http://{settings.host}:{settings.port}")
76
+ console.print(f" upstream {settings.upstream_anthropic_url}")
77
+ console.print(f" translator {settings.translator_model}")
78
+
79
+ uvicorn.run(create_app(settings), host=settings.host, port=settings.port, log_level="warning")
80
+
81
+
82
+ # -- wrap ---------------------------------------------------------------
83
+
84
+
85
+ def _proxy_is_up(port: int) -> bool:
86
+ try:
87
+ response = httpx.get(f"http://127.0.0.1:{port}/healthz", timeout=1.0)
88
+ return response.json().get("service") == "lingua-proxy"
89
+ except Exception:
90
+ return False
91
+
92
+
93
+ def _persist_upstream(url: str) -> None:
94
+ """Remember the resolved upstream so proxy/doctor/bench agree."""
95
+ path = default_config_path()
96
+ path.parent.mkdir(parents=True, exist_ok=True)
97
+ existing = path.read_text() if path.exists() else ""
98
+ lines = [ln for ln in existing.splitlines() if not ln.startswith("upstream_anthropic_url")]
99
+ lines.append(f'upstream_anthropic_url = "{url}"')
100
+ path.write_text("\n".join(lines).strip() + "\n")
101
+
102
+
103
+ @main.group()
104
+ def wrap() -> None:
105
+ """Run a client with its traffic routed through lingua-proxy."""
106
+
107
+
108
+ @wrap.command("claude")
109
+ @click.option("--port", type=int, default=8787, show_default=True)
110
+ @click.option("--upstream", default=None, help="Override the detected upstream.")
111
+ @click.option("--no-proxy", is_flag=True, help="Assume the proxy is already running.")
112
+ @click.option("--dry-run", is_flag=True, help="Show what would change, then restore.")
113
+ @click.option("--keep-patched", is_flag=True, hidden=True, help="Testing aid: skip restore.")
114
+ @click.argument("claude_args", nargs=-1, type=click.UNPROCESSED)
115
+ def wrap_claude(
116
+ port: int,
117
+ upstream: str | None,
118
+ no_proxy: bool,
119
+ dry_run: bool,
120
+ keep_patched: bool,
121
+ claude_args: tuple[str, ...],
122
+ ) -> None:
123
+ """Launch Claude Code pointed at the proxy, then restore your settings."""
124
+ if os.environ.get("LINGUA_PROXY_WRAPPED"):
125
+ raise click.ClickException(
126
+ "This session is already wrapped by lingua-proxy. Nesting would loop traffic."
127
+ )
128
+
129
+ resolved = resolve_anthropic_upstream(upstream)
130
+ proxy_url = f"http://127.0.0.1:{port}"
131
+ if resolved.rstrip("/") == proxy_url:
132
+ raise click.ClickException(
133
+ "The detected upstream is lingua-proxy itself, which would loop. "
134
+ "Pass --upstream to name the real API or gateway."
135
+ )
136
+
137
+ binary = None
138
+ if not dry_run and not keep_patched:
139
+ try:
140
+ binary = find_claude_binary()
141
+ except WrapError as exc:
142
+ raise click.ClickException(str(exc)) from exc
143
+
144
+ _persist_upstream(resolved)
145
+
146
+ try:
147
+ state = apply_wrap(proxy_url, port)
148
+ except WrapError as exc:
149
+ raise click.ClickException(str(exc)) from exc
150
+
151
+ console.print("[bold]lingua-proxy[/bold] wrapping Claude Code")
152
+ console.print(f" proxy {proxy_url}")
153
+ console.print(f" upstream {resolved}")
154
+ console.print(f" patched {state.settings_path}")
155
+
156
+ if not no_proxy and not _proxy_is_up(port):
157
+ console.print(
158
+ f" [yellow]note[/yellow] no proxy answering on port {port}; "
159
+ f"start one with 'lingua-proxy proxy --port {port}'"
160
+ )
161
+
162
+ if keep_patched:
163
+ return
164
+
165
+ if dry_run:
166
+ restore_wrap(state)
167
+ clear_marker()
168
+ console.print(" [green]dry run[/green] settings restored")
169
+ return
170
+
171
+ try:
172
+ # Set the base URL on the child too. The settings file is what Claude
173
+ # Code actually reads, but leaving a stale shell variable pointing
174
+ # elsewhere is confusing and would apply to anything else it launches.
175
+ child_env = {
176
+ **os.environ,
177
+ "ANTHROPIC_BASE_URL": proxy_url,
178
+ "LINGUA_PROXY_WRAPPED": "1",
179
+ }
180
+ result = subprocess.run([binary, *claude_args], env=child_env)
181
+ code = result.returncode
182
+ finally:
183
+ restore_wrap(state)
184
+ clear_marker()
185
+ console.print(" settings restored")
186
+
187
+ raise SystemExit(code)
188
+
189
+
190
+ @main.group()
191
+ def unwrap() -> None:
192
+ """Undo a wrap that did not clean up after itself."""
193
+
194
+
195
+ @unwrap.command("claude")
196
+ @click.option("--no-stop-proxy", is_flag=True, help="Leave a running proxy alone.")
197
+ def unwrap_claude(no_stop_proxy: bool) -> None:
198
+ """Restore client settings after a crashed wrap."""
199
+ state = load_marker()
200
+ if state is None:
201
+ console.print("Nothing to undo: no lingua-proxy wrap is recorded for this directory.")
202
+ return
203
+
204
+ restore_wrap(state)
205
+ clear_marker()
206
+ console.print(f"Restored {state.settings_path}")
207
+
208
+
209
+ # -- doctor -------------------------------------------------------------
210
+
211
+
212
+ @main.command()
213
+ @click.option("--port", type=int, default=8787, show_default=True)
214
+ @click.option("--json", "as_json", is_flag=True, help="Machine-readable output.")
215
+ def doctor(port: int, as_json: bool) -> None:
216
+ """Check that everything is wired up correctly."""
217
+ checks = run_checks(port=port)
218
+ worst = max((c.severity for c in checks), default=0)
219
+
220
+ if as_json:
221
+ click.echo(
222
+ json.dumps(
223
+ {
224
+ "version": __version__,
225
+ "exit_code": worst,
226
+ "checks": [c.to_json() for c in checks],
227
+ },
228
+ indent=2,
229
+ )
230
+ )
231
+ raise SystemExit(worst)
232
+
233
+ table = Table(title=f"lingua-proxy {__version__}", show_lines=False)
234
+ table.add_column("")
235
+ table.add_column("Check")
236
+ table.add_column("Result")
237
+ for check in checks:
238
+ table.add_row(check.glyph, check.name, check.summary)
239
+ console.print(table)
240
+
241
+ for check in checks:
242
+ if check.hint and check.severity:
243
+ console.print(f"[yellow]hint[/yellow] {check.name}: {check.hint}")
244
+
245
+ raise SystemExit(worst)
246
+
247
+
248
+ # -- dashboard ----------------------------------------------------------
249
+
250
+
251
+ @main.command()
252
+ @click.option("--json", "as_json", is_flag=True, help="Machine-readable output.")
253
+ def dashboard(as_json: bool) -> None:
254
+ """Show measured token savings."""
255
+ log = CostLog(path=lingua_home() / "cost_log.jsonl")
256
+ summary = summarize(log.rows())
257
+
258
+ if as_json:
259
+ click.echo(json.dumps(summary.to_json(), indent=2))
260
+ return
261
+
262
+ if summary.requests == 0:
263
+ console.print(
264
+ "No requests recorded yet. Start the proxy, send some traffic, then check back."
265
+ )
266
+ return
267
+
268
+ table = Table(title="lingua-proxy savings")
269
+ table.add_column("Metric")
270
+ table.add_column("Value", justify="right")
271
+ table.add_row("Requests", str(summary.requests))
272
+ table.add_row("Translated", str(summary.translated))
273
+ table.add_row("Passed through", str(summary.passthrough))
274
+ table.add_row("Cost with proxy", f"${summary.dollars_proxied:.4f}")
275
+ table.add_row("Cost without", f"${summary.dollars_baseline:.4f}")
276
+ table.add_row("Saved", f"${summary.dollars_saved:.4f}")
277
+ table.add_row("Savings", f"{summary.savings_ratio * 100:.1f}%")
278
+ table.add_row("Translator cost", f"${summary.translator_dollars:.4f}")
279
+ console.print(table)
280
+
281
+ if summary.by_language:
282
+ langs = Table(title="By language")
283
+ langs.add_column("Language")
284
+ langs.add_column("Requests", justify="right")
285
+ for lang, count in sorted(summary.by_language.items(), key=lambda kv: -kv[1]):
286
+ langs.add_row(lang, str(count))
287
+ console.print(langs)
288
+
289
+
290
+ # -- bench --------------------------------------------------------------
291
+
292
+
293
+ @main.command()
294
+ @click.option("--mode", type=click.Choice(["estimate", "ab"]), default="estimate")
295
+ @click.option("--min-savings", type=float, default=0.30, show_default=True)
296
+ @click.option("--json", "as_json", is_flag=True)
297
+ def bench(mode: str, min_savings: float, as_json: bool) -> None:
298
+ """Measure real savings against your own upstream."""
299
+ from lingua_proxy.bench import run_bench
300
+
301
+ raise SystemExit(run_bench(mode=mode, min_savings=min_savings, as_json=as_json))