pcli-agent 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (130) hide show
  1. pcli/__init__.py +1 -0
  2. pcli/__main__.py +4 -0
  3. pcli/agent/__init__.py +0 -0
  4. pcli/agent/activity.py +116 -0
  5. pcli/agent/compaction.py +205 -0
  6. pcli/agent/context_pruning.py +88 -0
  7. pcli/agent/headless.py +209 -0
  8. pcli/agent/loop.py +442 -0
  9. pcli/agent/prompt.py +371 -0
  10. pcli/agent/runtime.py +240 -0
  11. pcli/browser/__init__.py +0 -0
  12. pcli/browser/session.py +135 -0
  13. pcli/cli.py +757 -0
  14. pcli/config/__init__.py +0 -0
  15. pcli/config/paths.py +95 -0
  16. pcli/config/settings.py +435 -0
  17. pcli/cost/__init__.py +0 -0
  18. pcli/cost/context.py +275 -0
  19. pcli/cost/context_detect.py +183 -0
  20. pcli/cost/pricing_table.py +141 -0
  21. pcli/cost/tracker.py +126 -0
  22. pcli/llm/__init__.py +0 -0
  23. pcli/llm/client.py +285 -0
  24. pcli/llm/errors.py +37 -0
  25. pcli/llm/models.py +100 -0
  26. pcli/llm/streaming.py +108 -0
  27. pcli/memory/__init__.py +0 -0
  28. pcli/memory/extraction.py +106 -0
  29. pcli/memory/models.py +103 -0
  30. pcli/memory/store.py +88 -0
  31. pcli/permissions/__init__.py +0 -0
  32. pcli/permissions/guardrails.py +219 -0
  33. pcli/permissions/manager.py +215 -0
  34. pcli/permissions/policy.py +70 -0
  35. pcli/sandbox/__init__.py +0 -0
  36. pcli/sandbox/base.py +50 -0
  37. pcli/sandbox/docker_backend.py +107 -0
  38. pcli/sandbox/limits.py +63 -0
  39. pcli/sandbox/null_backend.py +92 -0
  40. pcli/sandbox/selector.py +75 -0
  41. pcli/sandbox/subprocess_backend.py +376 -0
  42. pcli/scheduler/__init__.py +0 -0
  43. pcli/scheduler/daemon.py +194 -0
  44. pcli/scheduler/models.py +97 -0
  45. pcli/scheduler/runner.py +84 -0
  46. pcli/scheduler/store.py +75 -0
  47. pcli/scheduler/triggers.py +84 -0
  48. pcli/session/__init__.py +0 -0
  49. pcli/session/audit.py +122 -0
  50. pcli/session/directory_check.py +28 -0
  51. pcli/session/export.py +57 -0
  52. pcli/session/importer.py +92 -0
  53. pcli/session/models.py +168 -0
  54. pcli/session/store.py +127 -0
  55. pcli/telegram/__init__.py +0 -0
  56. pcli/telegram/bot.py +266 -0
  57. pcli/telegram/daemon.py +1197 -0
  58. pcli/telegram/permissions.py +131 -0
  59. pcli/telegram/sender.py +58 -0
  60. pcli/tools/__init__.py +0 -0
  61. pcli/tools/_nested_agent.py +204 -0
  62. pcli/tools/agent_tools.py +264 -0
  63. pcli/tools/agent_tools_store.py +69 -0
  64. pcli/tools/artifacts.py +47 -0
  65. pcli/tools/base.py +185 -0
  66. pcli/tools/builtin/__init__.py +0 -0
  67. pcli/tools/builtin/agent_tool_register_tool.py +100 -0
  68. pcli/tools/builtin/artifact_tool.py +212 -0
  69. pcli/tools/builtin/ask_tool.py +77 -0
  70. pcli/tools/builtin/browser_tool.py +253 -0
  71. pcli/tools/builtin/decision_tool.py +73 -0
  72. pcli/tools/builtin/describe_tool.py +389 -0
  73. pcli/tools/builtin/diff_tools.py +225 -0
  74. pcli/tools/builtin/fs_tools.py +371 -0
  75. pcli/tools/builtin/grep_tool.py +88 -0
  76. pcli/tools/builtin/memory_tool.py +108 -0
  77. pcli/tools/builtin/network_tools.py +107 -0
  78. pcli/tools/builtin/pip_tool.py +106 -0
  79. pcli/tools/builtin/shell_tool.py +240 -0
  80. pcli/tools/builtin/subagent_tool.py +146 -0
  81. pcli/tools/builtin/todo_tool.py +122 -0
  82. pcli/tools/builtin/toolbox_register_tool.py +76 -0
  83. pcli/tools/builtin/web_tools.py +322 -0
  84. pcli/tools/pydiscovery/__init__.py +0 -0
  85. pcli/tools/pydiscovery/cache.py +51 -0
  86. pcli/tools/pydiscovery/index.py +48 -0
  87. pcli/tools/pydiscovery/invoke.py +181 -0
  88. pcli/tools/pydiscovery/search.py +117 -0
  89. pcli/tools/registry.py +138 -0
  90. pcli/tools/toolbox/__init__.py +0 -0
  91. pcli/tools/toolbox/introspect.py +48 -0
  92. pcli/tools/toolbox/manager.py +336 -0
  93. pcli/tools/toolbox/plugin_base.py +51 -0
  94. pcli/tools/toolbox/plugins/__init__.py +6 -0
  95. pcli/tools/toolbox/plugins/httpd.py +99 -0
  96. pcli/tools/toolbox/plugins/kafka.py +162 -0
  97. pcli/tools/toolbox/plugins/kubectl.py +211 -0
  98. pcli/tools/toolbox/plugins/sge.py +146 -0
  99. pcli/tools/toolbox/store.py +65 -0
  100. pcli/tools/toolbox/synthesize.py +100 -0
  101. pcli/tui/__init__.py +0 -0
  102. pcli/tui/app.py +37 -0
  103. pcli/tui/screens/__init__.py +0 -0
  104. pcli/tui/screens/ask_question_modal.py +54 -0
  105. pcli/tui/screens/chat.py +2070 -0
  106. pcli/tui/screens/confirm_modal.py +39 -0
  107. pcli/tui/screens/models.py +43 -0
  108. pcli/tui/screens/permission_modal.py +71 -0
  109. pcli/tui/screens/sessions.py +162 -0
  110. pcli/tui/screens/subagent_activity_modal.py +71 -0
  111. pcli/tui/shell_passthrough.py +56 -0
  112. pcli/tui/styles/pcli.tcss +241 -0
  113. pcli/tui/themes.py +84 -0
  114. pcli/tui/widgets/__init__.py +0 -0
  115. pcli/tui/widgets/chat_input.py +240 -0
  116. pcli/tui/widgets/command_suggestions.py +33 -0
  117. pcli/tui/widgets/message_view.py +328 -0
  118. pcli/tui/widgets/paste_input.py +99 -0
  119. pcli/tui/widgets/paste_marker.py +69 -0
  120. pcli/tui/widgets/status_bar.py +133 -0
  121. pcli/tui/widgets/status_pane.py +58 -0
  122. pcli/util/__init__.py +0 -0
  123. pcli/util/ids.py +15 -0
  124. pcli/util/logging.py +18 -0
  125. pcli/util/text.py +10 -0
  126. pcli_agent-0.1.0.dist-info/METADATA +259 -0
  127. pcli_agent-0.1.0.dist-info/RECORD +130 -0
  128. pcli_agent-0.1.0.dist-info/WHEEL +4 -0
  129. pcli_agent-0.1.0.dist-info/entry_points.txt +2 -0
  130. pcli_agent-0.1.0.dist-info/licenses/LICENSE +21 -0
pcli/cost/context.py ADDED
@@ -0,0 +1,275 @@
1
+ """Tracks how much of the model's context window the conversation is using.
2
+
3
+ There's no local tokenizer for a generic gateway, so this doesn't estimate
4
+ from message text — it uses the most recently *reported* usage.total_tokens
5
+ (prompt_tokens + completion_tokens of the last actual LLM call). That number
6
+ is exactly the size of what gets resent as history on the next call, which
7
+ is what "context used" means in practice. Since CostTracker records one
8
+ TurnCost per underlying LLM call (not per user-visible turn — a single turn
9
+ with tool calls makes several), the most recent *main*-conversation entry in
10
+ Session.cost.turns reflects the most recent real call, tool-call round-trips
11
+ included.
12
+
13
+ "Most recent main-conversation entry" is deliberately not just "the last
14
+ entry": TurnCost.source distinguishes a real main-conversation call from a
15
+ subagent's or a compaction summarization's own LLM call, both of which are
16
+ real spend (see CostTracker.record_turn) but reflect a completely different,
17
+ unrelated conversation's size — trusting the literal last entry regardless
18
+ of source previously let a subagent call (in an edge case) or a compaction
19
+ call (every single time compaction ran) leave a stray, unrelated token count
20
+ as what the *next* turn's context-usage/max_tokens-cap math was computed
21
+ from. See _last_main_turn/_main_turns below, used by every function here.
22
+ """
23
+
24
+ from __future__ import annotations
25
+
26
+ import tomllib
27
+ from dataclasses import dataclass
28
+
29
+ from pcli.config.paths import context_limits_file
30
+ from pcli.cost.pricing_table import match_model_pattern
31
+ from pcli.llm.models import Usage
32
+ from pcli.session.models import Session, TurnCost
33
+
34
+
35
+ def _main_turns(turns: list[TurnCost]) -> list[TurnCost]:
36
+ return [turn for turn in turns if turn.source == "main"]
37
+
38
+
39
+ def _last_main_turn_usage(turns: list[TurnCost]) -> Usage | None:
40
+ for turn in reversed(turns):
41
+ if turn.source == "main":
42
+ return turn.usage
43
+ return None
44
+
45
+ DEFAULT_CONTEXT_LIMITS_TOML = """\
46
+ # pcli model context-window limits (tokens). Edit freely — entries here
47
+ # override pcli's built-in defaults. Model names support a trailing '*' as a
48
+ # prefix wildcard, e.g. "gpt-4o*". [default] is used for any model that
49
+ # matches nothing else.
50
+
51
+ [default]
52
+ limit = 128000
53
+ """
54
+
55
+ # Best-effort starting point; intentionally approximate, user-editable.
56
+ _BUILTIN_LIMITS: dict[str, int] = {
57
+ "gpt-4o*": 128_000,
58
+ "gpt-4.1*": 1_000_000,
59
+ "gpt-4-turbo*": 128_000,
60
+ "gpt-3.5*": 16_385,
61
+ "o1*": 200_000,
62
+ "o3*": 200_000,
63
+ "claude-opus*": 200_000,
64
+ "claude-sonnet*": 200_000,
65
+ "claude-haiku*": 200_000,
66
+ "gemini-1.5-pro*": 2_000_000,
67
+ "gemini-1.5-flash*": 1_000_000,
68
+ "llama-3*": 128_000,
69
+ "mistral*": 32_000,
70
+ }
71
+
72
+ _DEFAULT_LIMIT = 128_000
73
+
74
+
75
+ class ContextLimitTable:
76
+ def __init__(
77
+ self, entries: dict[str, int], default: int, auto_detected: set[str] | None = None
78
+ ) -> None:
79
+ self._entries = entries
80
+ self._default = default
81
+ # Exact model names (never wildcard patterns — see
82
+ # set_model_context_limit's exact-match docstring) whose [models]
83
+ # entry came from a previous auto-detect run rather than a manual
84
+ # /context-limit correction. Distinguishing the two matters because
85
+ # a local gateway's loaded context (LM Studio, Ollama, ...) can
86
+ # legitimately change between runs (the user reloads the model with
87
+ # a different context-length setting) - so an auto-detected value
88
+ # can go stale in a way a manual override or a hosted API's fixed
89
+ # limit never does. See should_attempt_detection.
90
+ self._auto_detected = auto_detected or set()
91
+
92
+ @classmethod
93
+ def load(cls) -> ContextLimitTable:
94
+ entries = dict(_BUILTIN_LIMITS)
95
+ default = _DEFAULT_LIMIT
96
+ auto_detected: set[str] = set()
97
+
98
+ path = context_limits_file()
99
+ if path.exists():
100
+ raw = tomllib.loads(path.read_text(encoding="utf-8"))
101
+ models = raw.get("models", {})
102
+ for pattern, limit in models.items():
103
+ entries[pattern] = int(limit)
104
+ if "default" in raw and "limit" in raw["default"]:
105
+ default = int(raw["default"]["limit"])
106
+ auto_detected = {name for name, flag in raw.get("auto_detected", {}).items() if flag}
107
+
108
+ return cls(entries, default, auto_detected)
109
+
110
+ def lookup(self, model_name: str) -> int:
111
+ match = match_model_pattern(model_name, self._entries)
112
+ return match if match is not None else self._default
113
+
114
+ def has_explicit_entry(self, model_name: str) -> bool:
115
+ """True if model_name matches a real (builtin or user/auto-set)
116
+ entry, as opposed to lookup() silently falling back to the generic
117
+ default."""
118
+ return match_model_pattern(model_name, self._entries) is not None
119
+
120
+ def should_attempt_detection(self, model_name: str) -> bool:
121
+ """Used to gate auto-detection (cost/context_detect.py): True if
122
+ there's no entry for this model at all (the has_explicit_entry==False
123
+ case), OR the entry that's there came from a previous auto-detect
124
+ run rather than a manual /context-limit correction or a built-in
125
+ default. A manual override and a built-in are trusted forever and
126
+ never re-probed; an auto-detected value gets re-checked on every
127
+ startup since it can drift out from under pcli without any pcli-side
128
+ signal (see the class docstring note on _auto_detected)."""
129
+ match = match_model_pattern(model_name, self._entries)
130
+ if match is None:
131
+ return True
132
+ return model_name in self._auto_detected
133
+
134
+
135
+ @dataclass
136
+ class ContextUsage:
137
+ used_tokens: int
138
+ limit_tokens: int
139
+
140
+ @property
141
+ def fraction(self) -> float:
142
+ return self.used_tokens / self.limit_tokens if self.limit_tokens else 0.0
143
+
144
+
145
+ def current_context_usage(
146
+ session: Session, *, limit_table: ContextLimitTable | None = None
147
+ ) -> ContextUsage:
148
+ limit_table = limit_table or ContextLimitTable.load()
149
+ last_main_usage = _last_main_turn_usage(session.cost.turns)
150
+ used = last_main_usage.total_tokens if last_main_usage is not None else 0
151
+ return ContextUsage(used_tokens=used, limit_tokens=limit_table.lookup(session.model))
152
+
153
+
154
+ def compute_max_response_tokens(
155
+ session: Session, *, limit_table: ContextLimitTable | None = None, safety_margin: int
156
+ ) -> int | None:
157
+ """A dynamic per-request max_tokens cap: leaves just enough headroom
158
+ that a single response can't consume the *entire* remaining context
159
+ window by itself. Confirmed against a real debugged session:
160
+ auto-compaction only runs *between* turns, checking the fraction used
161
+ as of the last completed one — a turn that ended 69% full (comfortably
162
+ under the 80% auto-compact threshold) gave compaction no reason to run,
163
+ but the very next turn's own response then generated 4663 tokens in a
164
+ single, ~100-minute-long generation, consumed the remaining ~31% of the
165
+ window entirely by itself, and got hard-truncated mid-stream by the
166
+ gateway's own ceiling - there was no checkpoint inside that one
167
+ generation for compaction to catch it at. Recomputed fresh from the
168
+ same usage.total_tokens basis current_context_usage (and therefore
169
+ auto-compaction) already uses - there's no local tokenizer to do better
170
+ (see this module's docstring) - so it's exactly as accurate as pcli's
171
+ other context-usage decisions, no more, no less.
172
+
173
+ Returns None (no cap sent - the gateway's own default applies) if
174
+ there's no prior *main*-conversation usage yet (first turn - nothing to
175
+ compute headroom from yet, regardless of whether a subagent or
176
+ compaction call happened to run before it) or if the computed headroom
177
+ is already <= 0 (essentially full; sending a non-positive max_tokens
178
+ would be nonsensical, and by this point auto-compaction should already
179
+ have intervened)."""
180
+ if _last_main_turn_usage(session.cost.turns) is None:
181
+ return None
182
+ usage = current_context_usage(session, limit_table=limit_table)
183
+ available = usage.limit_tokens - usage.used_tokens - safety_margin
184
+ return available if available > 0 else None
185
+
186
+
187
+ _CEILING_MIN_TOTAL_TOKENS = 2000
188
+ _CEILING_STALL_RATIO = 0.1
189
+
190
+
191
+ def looks_like_context_ceiling(session: Session) -> bool:
192
+ """True if, compared to the immediately preceding turn, prompt_tokens
193
+ grew (more history got sent, as it always does turn over turn) but
194
+ total_tokens barely moved — meaning completion_tokens got squeezed down
195
+ to compensate. That's the fingerprint of a real, gateway-enforced
196
+ context ceiling being hit, independent of whatever limit_tokens pcli
197
+ itself assumed via ContextLimitTable (which is silently wrong — falls
198
+ back to a generic 128000-token guess — for any model without a built-in
199
+ or user-configured entry, so the fraction-based auto-compact trigger
200
+ alone can't catch this: a session that's actually 100% full can read as
201
+ a small fraction of an assumed limit that's far too large). Ordinary
202
+ turn-to-turn variation has *both* prompt_tokens and total_tokens growing
203
+ together; only a real ceiling clamps total_tokens while prompt_tokens
204
+ keeps climbing.
205
+
206
+ Confirmed against a real debugged session: three consecutive turns each
207
+ landed on total_tokens=16384 (or within a few tokens of it) despite
208
+ prompt_tokens climbing every turn — the model's real ~16k window, for a
209
+ model pcli had no entry for (it was assuming 128000, i.e. reading the
210
+ session as ~13% full when it was actually exhausted).
211
+
212
+ Compares the last two *main*-conversation entries specifically (see
213
+ module docstring) — a subagent or compaction call sitting between them
214
+ in Session.cost.turns has its own unrelated prompt/total sizes and would
215
+ otherwise be compared as if it were consecutive turns."""
216
+ turns = _main_turns(session.cost.turns)
217
+ if len(turns) < 2:
218
+ return False
219
+ current = turns[-1].usage
220
+ previous = turns[-2].usage
221
+ if current.total_tokens < _CEILING_MIN_TOTAL_TOKENS:
222
+ return False
223
+ prompt_growth = current.prompt_tokens - previous.prompt_tokens
224
+ if prompt_growth <= 0:
225
+ return False
226
+ total_growth = current.total_tokens - previous.total_tokens
227
+ return total_growth <= prompt_growth * _CEILING_STALL_RATIO
228
+
229
+
230
+ def set_model_context_limit(model: str, limit: int, *, auto_detected: bool = False) -> None:
231
+ """Persists a per-model context-window override to context_limits.toml's
232
+ [models] table (creating the file if it doesn't exist yet), so a wrong
233
+ built-in guess — or no entry at all — can be corrected without hand-
234
+ editing the file. Exact model-name match, not a wildcard pattern:
235
+ simpler and more predictable for a single correction than guessing a
236
+ sensible glob.
237
+
238
+ `auto_detected=True` (the cost/context_detect.py probe path) also
239
+ records the model in the [auto_detected] table, so
240
+ ContextLimitTable.should_attempt_detection knows to re-probe it on a
241
+ later startup rather than trusting it forever. `auto_detected=False`
242
+ (the default — the TUI's /context-limit command) is a human's deliberate
243
+ correction, so it always clears any prior auto-detected marker for this
244
+ model too: once a human has set it, it's sticky forever, even if it was
245
+ auto-detected before."""
246
+ path = context_limits_file()
247
+ raw: dict = {}
248
+ if path.exists():
249
+ raw = dict(tomllib.loads(path.read_text(encoding="utf-8")))
250
+
251
+ default_limit = _DEFAULT_LIMIT
252
+ existing_default = raw.get("default")
253
+ if isinstance(existing_default, dict) and "limit" in existing_default:
254
+ default_limit = int(existing_default["limit"])
255
+
256
+ models = dict(raw.get("models", {}))
257
+ models[model] = limit
258
+
259
+ auto_detected_models = {name for name, flag in raw.get("auto_detected", {}).items() if flag}
260
+ if auto_detected:
261
+ auto_detected_models.add(model)
262
+ else:
263
+ auto_detected_models.discard(model)
264
+
265
+ def _escape(value: str) -> str:
266
+ return value.replace("\\", "\\\\").replace('"', '\\"')
267
+
268
+ lines = ["[default]", f"limit = {default_limit}", "", "[models]"]
269
+ lines.extend(f'"{_escape(pattern)}" = {value}' for pattern, value in models.items())
270
+ if auto_detected_models:
271
+ lines.extend(["", "[auto_detected]"])
272
+ lines.extend(f'"{_escape(name)}" = true' for name in sorted(auto_detected_models))
273
+
274
+ path.parent.mkdir(parents=True, exist_ok=True)
275
+ path.write_text("\n".join(lines) + "\n", encoding="utf-8")
@@ -0,0 +1,183 @@
1
+ """Best-effort auto-detection of a model's real context window, queried
2
+ directly from the gateway.
3
+
4
+ There's no single standard OpenAI-compatible endpoint for this — the vanilla
5
+ `GET /models` response (per the OpenAI spec) carries only `id`/`object`/
6
+ `created`/`owned_by`, nothing about context size. Every backend that DOES
7
+ expose it does so through its own extension:
8
+
9
+ - OpenRouter and vLLM both add an extra field (`context_length` /
10
+ `max_model_len` respectively) directly onto the standard `/models`
11
+ response — covered for free by the first probe below, no separate
12
+ endpoint needed.
13
+ - LM Studio has its own native `GET /api/v0/models` endpoint
14
+ (`max_context_length`, and possibly `loaded_context_length` for a
15
+ currently-loaded model).
16
+ - A raw llama.cpp server (not behind LM Studio) exposes `GET /props`,
17
+ whose `default_generation_settings.n_ctx` reflects whatever single model
18
+ it has loaded.
19
+ - Ollama exposes `POST /api/show` (`{"model": "..."}`), whose `model_info`
20
+ has a `<arch>.context_length` key (the architecture name varies per
21
+ model family, hence the suffix scan below rather than a fixed key).
22
+ - A LiteLLM proxy exposes `GET /model/info`, returning every configured
23
+ model's `model_name` alongside a `model_info` object with
24
+ `max_input_tokens` (the real context window — `max_tokens` on the same
25
+ object is unreliable/inconsistently populated across LiteLLM versions,
26
+ often reflecting max *output* tokens instead, so it's only used as a
27
+ fallback when `max_input_tokens` is absent).
28
+
29
+ Hosted-only gateways (real OpenAI, Anthropic, Azure OpenAI, most plain
30
+ OpenAI-compatible proxies) expose none of this — every probe here just
31
+ fails cleanly and detect_context_limit() returns None, exactly as if the
32
+ model were simply unrecognized.
33
+
34
+ The standard-/models probe deliberately reuses `client`'s own configured
35
+ base_url as-is (same path GatewayClient.list_models() already calls),
36
+ since it's a real OpenAI-style endpoint that legitimately lives under
37
+ whatever prefix (typically '/v1', sometimes something custom) the user
38
+ configured. The other four are each a fixed, well-known path at the
39
+ gateway's *origin* (scheme+host+port) — they must NOT inherit that prefix,
40
+ so `_origin_url` builds each one as a fully-qualified absolute URL rather
41
+ than a bare path. This matters because httpx.AsyncClient deliberately does
42
+ NOT treat a leading '/' as "reset to origin root" the way browsers/urljoin
43
+ do — Client._merge_url always appends onto base_url's own path regardless
44
+ of a leading slash (confirmed directly against httpx's source: it exists
45
+ specifically so a relative-looking path composes predictably under a
46
+ subpath base_url) — only a fully-qualified URL (with its own scheme)
47
+ bypasses that merging entirely.
48
+ """
49
+
50
+ from __future__ import annotations
51
+
52
+ from collections.abc import Awaitable, Callable
53
+
54
+ import httpx
55
+
56
+ _PROBE_TIMEOUT_S = 5.0
57
+
58
+
59
+ def _positive_int(value: object) -> int | None:
60
+ return value if isinstance(value, int) and value > 0 else None
61
+
62
+
63
+ def _origin_url(client: httpx.AsyncClient, path: str) -> str:
64
+ """A fully-qualified URL at client's origin (scheme+host+port),
65
+ ignoring whatever path (e.g. '/v1') is baked into its base_url — see
66
+ the module docstring for why this can't just be a plain '/path'."""
67
+ base = client.base_url
68
+ port_part = f":{base.port}" if base.port else ""
69
+ return f"{base.scheme}://{base.host}{port_part}{path}"
70
+
71
+
72
+ async def _probe_standard_models_endpoint(
73
+ client: httpx.AsyncClient, model: str, timeout: float
74
+ ) -> int | None:
75
+ response = await client.get("/models", timeout=timeout)
76
+ if response.status_code >= 400:
77
+ return None
78
+ entries = response.json().get("data", [])
79
+ for entry in entries:
80
+ if not isinstance(entry, dict) or entry.get("id") != model:
81
+ continue
82
+ for key in ("context_length", "max_model_len"):
83
+ limit = _positive_int(entry.get(key))
84
+ if limit is not None:
85
+ return limit
86
+ return None
87
+
88
+
89
+ async def _probe_lmstudio(client: httpx.AsyncClient, model: str, timeout: float) -> int | None:
90
+ response = await client.get(_origin_url(client, "/api/v0/models"), timeout=timeout)
91
+ if response.status_code >= 400:
92
+ return None
93
+ entries = response.json().get("data", [])
94
+ for entry in entries:
95
+ if not isinstance(entry, dict) or entry.get("id") != model:
96
+ continue
97
+ # loaded_context_length only appears once LM Studio has actually
98
+ # loaded the model into memory. LM Studio JIT-loads on first
99
+ # inference and unloads after an idle TTL, so at pcli's startup-time
100
+ # probe the model is very often still "not-loaded" - the entry then
101
+ # has only max_context_length, the architecture's *maximum
102
+ # supported* window, which can be many times larger than what
103
+ # actually gets allocated once loaded (confirmed against a real
104
+ # session: max_context_length=262144 for a model llama.cpp had
105
+ # loaded with n_ctx_slot=16384). Trusting that fallback previously
106
+ # reintroduced the exact stale-limit bug should_attempt_detection
107
+ # was built to fix, just from a different angle - a "don't know"
108
+ # (None, falls through to the next probe / the generic default)
109
+ # is safer than a confidently wrong number here.
110
+ return _positive_int(entry.get("loaded_context_length"))
111
+ return None
112
+
113
+
114
+ async def _probe_llama_cpp_props(client: httpx.AsyncClient, model: str, timeout: float) -> int | None:
115
+ response = await client.get(_origin_url(client, "/props"), timeout=timeout)
116
+ if response.status_code >= 400:
117
+ return None
118
+ settings = response.json().get("default_generation_settings", {})
119
+ return _positive_int(settings.get("n_ctx")) if isinstance(settings, dict) else None
120
+
121
+
122
+ async def _probe_ollama(client: httpx.AsyncClient, model: str, timeout: float) -> int | None:
123
+ response = await client.post(
124
+ _origin_url(client, "/api/show"), json={"model": model}, timeout=timeout
125
+ )
126
+ if response.status_code >= 400:
127
+ return None
128
+ model_info = response.json().get("model_info", {})
129
+ if not isinstance(model_info, dict):
130
+ return None
131
+ for key, value in model_info.items():
132
+ if key.endswith(".context_length"):
133
+ limit = _positive_int(value)
134
+ if limit is not None:
135
+ return limit
136
+ return None
137
+
138
+
139
+ async def _probe_litellm(client: httpx.AsyncClient, model: str, timeout: float) -> int | None:
140
+ response = await client.get(_origin_url(client, "/model/info"), timeout=timeout)
141
+ if response.status_code >= 400:
142
+ return None
143
+ entries = response.json().get("data", [])
144
+ for entry in entries:
145
+ if not isinstance(entry, dict) or entry.get("model_name") != model:
146
+ continue
147
+ model_info = entry.get("model_info")
148
+ if not isinstance(model_info, dict):
149
+ continue
150
+ for key in ("max_input_tokens", "max_tokens"):
151
+ limit = _positive_int(model_info.get(key))
152
+ if limit is not None:
153
+ return limit
154
+ return None
155
+
156
+
157
+ _PROBES: tuple[Callable[[httpx.AsyncClient, str, float], Awaitable[int | None]], ...] = (
158
+ _probe_standard_models_endpoint,
159
+ _probe_lmstudio,
160
+ _probe_llama_cpp_props,
161
+ _probe_ollama,
162
+ _probe_litellm,
163
+ )
164
+
165
+
166
+ async def detect_context_limit(
167
+ client: httpx.AsyncClient, model: str, *, timeout: float = _PROBE_TIMEOUT_S
168
+ ) -> int | None:
169
+ """Tries each known backend-specific probe in turn against `client`'s
170
+ already-configured base_url/auth, returning the first positive context
171
+ size found. Every probe swallows its own connection/parse failures
172
+ (a backend that doesn't support it looks exactly like one that isn't
173
+ reachable — both just mean "try the next one" or, if all fail, "give
174
+ up and let the caller fall back to the assumed/manual limit"), so this
175
+ never raises."""
176
+ for probe in _PROBES:
177
+ try:
178
+ limit = await probe(client, model, timeout)
179
+ except (httpx.HTTPError, ValueError, TypeError, AttributeError):
180
+ continue
181
+ if limit is not None:
182
+ return limit
183
+ return None
@@ -0,0 +1,141 @@
1
+ """Per-model USD pricing. The gateway is generic/vendor-agnostic, so pricing
2
+ has no single source of truth — it's a small built-in best-effort table,
3
+ overridable by the user's pricing.toml."""
4
+
5
+ from __future__ import annotations
6
+
7
+ import tomllib
8
+ from typing import TypeVar
9
+
10
+ from pydantic import BaseModel
11
+
12
+ from pcli.config.paths import pricing_file
13
+
14
+ T = TypeVar("T")
15
+
16
+
17
+ def match_model_pattern(model_name: str, entries: dict[str, T]) -> T | None:
18
+ """Exact match wins; otherwise the longest '*'-suffixed prefix pattern
19
+ that matches. Shared by PricingTable and context.ContextLimitTable so
20
+ both per-model lookup tables behave identically."""
21
+ if model_name in entries:
22
+ return entries[model_name]
23
+ best_match: T | None = None
24
+ best_len = -1
25
+ for pattern, value in entries.items():
26
+ if pattern.endswith("*"):
27
+ prefix = pattern[:-1]
28
+ if model_name.startswith(prefix) and len(prefix) > best_len:
29
+ best_match = value
30
+ best_len = len(prefix)
31
+ return best_match
32
+
33
+ DEFAULT_PRICING_TOML = """\
34
+ # pcli model pricing (USD per 1M tokens). Edit freely — entries here override
35
+ # pcli's built-in defaults. Model names support a trailing '*' as a prefix
36
+ # wildcard, e.g. "gpt-4o*". The [default] table is used for any model that
37
+ # matches nothing else. cached_input_per_1m is optional - the rate for
38
+ # prompt tokens the gateway reports as served from its prompt cache (see
39
+ # llm/models.py's Usage.cached_tokens); omit it if you don't know the
40
+ # model's real cache-discount rate, and cached tokens are priced the same
41
+ # as ordinary input tokens instead (conservative - never undercounts spend,
42
+ # just won't reflect a real discount the gateway might be applying).
43
+
44
+ [default]
45
+ input_per_1m = 0.0
46
+ output_per_1m = 0.0
47
+ """
48
+
49
+ # Best-effort starting point; intentionally approximate, user-editable.
50
+ # Third element is cached_input_per_1m (None where no well-established
51
+ # cache-discount rate is known for that provider) - OpenAI's documented
52
+ # prompt-caching discount is ~50% of input price, Anthropic's cached-read
53
+ # rate is ~10% of input price; both approximated here consistently rather
54
+ # than guessed per model. Gemini/local models are left at None: real cache
55
+ # pricing exists for some of them too but isn't standardized enough here to
56
+ # approximate with the same confidence.
57
+ _BUILTIN_MODELS: dict[str, tuple[float, float, float | None]] = {
58
+ "gpt-4o*": (2.50, 10.00, 1.25),
59
+ "gpt-4.1*": (2.00, 8.00, 1.00),
60
+ "gpt-4-turbo*": (10.00, 30.00, 5.00),
61
+ "gpt-3.5*": (0.50, 1.50, 0.25),
62
+ "o1*": (15.00, 60.00, 7.50),
63
+ "o3*": (2.00, 8.00, 1.00),
64
+ "claude-opus*": (15.00, 75.00, 1.50),
65
+ "claude-sonnet*": (3.00, 15.00, 0.30),
66
+ "claude-haiku*": (0.80, 4.00, 0.08),
67
+ "gemini-1.5-pro*": (1.25, 5.00, None),
68
+ "gemini-1.5-flash*": (0.075, 0.30, None),
69
+ "llama-3*": (0.20, 0.20, None),
70
+ "mistral*": (0.25, 0.75, None),
71
+ }
72
+
73
+
74
+ class ModelPricing(BaseModel):
75
+ input_per_1m: float = 0.0
76
+ output_per_1m: float = 0.0
77
+ cached_input_per_1m: float | None = None
78
+ """Price for prompt tokens the gateway reports as served from its
79
+ prompt cache (Usage.cached_tokens, a subset of prompt_tokens) - None
80
+ (the common case, and the only option for a user-supplied pricing.toml
81
+ entry that doesn't set it) means no cache-specific rate is modeled;
82
+ cost_usd then falls back to pricing input_per_1m for those tokens too,
83
+ same as before this field existed."""
84
+
85
+
86
+ class PricingTable:
87
+ def __init__(self, entries: dict[str, ModelPricing], default: ModelPricing) -> None:
88
+ self._entries = entries
89
+ self._default = default
90
+
91
+ @classmethod
92
+ def load(cls) -> PricingTable:
93
+ entries = {
94
+ pattern: ModelPricing(input_per_1m=inp, output_per_1m=out, cached_input_per_1m=cached)
95
+ for pattern, (inp, out, cached) in _BUILTIN_MODELS.items()
96
+ }
97
+ default = ModelPricing()
98
+
99
+ path = pricing_file()
100
+ if path.exists():
101
+ raw = tomllib.loads(path.read_text(encoding="utf-8"))
102
+ models = raw.get("models", {})
103
+ for pattern, values in models.items():
104
+ entries[pattern] = ModelPricing(**values)
105
+ if "default" in raw:
106
+ default = ModelPricing(**raw["default"])
107
+
108
+ return cls(entries, default)
109
+
110
+ def lookup(self, model_name: str) -> ModelPricing:
111
+ match = match_model_pattern(model_name, self._entries)
112
+ return match if match is not None else self._default
113
+
114
+ def cost_usd(
115
+ self,
116
+ model_name: str,
117
+ *,
118
+ prompt_tokens: int,
119
+ completion_tokens: int,
120
+ cached_tokens: int = 0,
121
+ ) -> float:
122
+ """cached_tokens is the gateway-reported subset of prompt_tokens
123
+ served from its prompt cache (Usage.cached_tokens) - priced at the
124
+ model's cached_input_per_1m rate if known, otherwise at the
125
+ ordinary input rate (no assumed discount). Clamped to prompt_tokens
126
+ so a gateway reporting cached_tokens >= prompt_tokens (not
127
+ guaranteed never to happen across every backend) can't produce a
128
+ negative uncached-token count."""
129
+ pricing = self.lookup(model_name)
130
+ cached_tokens = min(cached_tokens, prompt_tokens)
131
+ uncached_tokens = prompt_tokens - cached_tokens
132
+ cached_rate = (
133
+ pricing.cached_input_per_1m
134
+ if pricing.cached_input_per_1m is not None
135
+ else pricing.input_per_1m
136
+ )
137
+ return (
138
+ (uncached_tokens / 1_000_000) * pricing.input_per_1m
139
+ + (cached_tokens / 1_000_000) * cached_rate
140
+ + (completion_tokens / 1_000_000) * pricing.output_per_1m
141
+ )