pcli-agent 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pcli/__init__.py +1 -0
- pcli/__main__.py +4 -0
- pcli/agent/__init__.py +0 -0
- pcli/agent/activity.py +116 -0
- pcli/agent/compaction.py +205 -0
- pcli/agent/context_pruning.py +88 -0
- pcli/agent/headless.py +209 -0
- pcli/agent/loop.py +442 -0
- pcli/agent/prompt.py +371 -0
- pcli/agent/runtime.py +240 -0
- pcli/browser/__init__.py +0 -0
- pcli/browser/session.py +135 -0
- pcli/cli.py +757 -0
- pcli/config/__init__.py +0 -0
- pcli/config/paths.py +95 -0
- pcli/config/settings.py +435 -0
- pcli/cost/__init__.py +0 -0
- pcli/cost/context.py +275 -0
- pcli/cost/context_detect.py +183 -0
- pcli/cost/pricing_table.py +141 -0
- pcli/cost/tracker.py +126 -0
- pcli/llm/__init__.py +0 -0
- pcli/llm/client.py +285 -0
- pcli/llm/errors.py +37 -0
- pcli/llm/models.py +100 -0
- pcli/llm/streaming.py +108 -0
- pcli/memory/__init__.py +0 -0
- pcli/memory/extraction.py +106 -0
- pcli/memory/models.py +103 -0
- pcli/memory/store.py +88 -0
- pcli/permissions/__init__.py +0 -0
- pcli/permissions/guardrails.py +219 -0
- pcli/permissions/manager.py +215 -0
- pcli/permissions/policy.py +70 -0
- pcli/sandbox/__init__.py +0 -0
- pcli/sandbox/base.py +50 -0
- pcli/sandbox/docker_backend.py +107 -0
- pcli/sandbox/limits.py +63 -0
- pcli/sandbox/null_backend.py +92 -0
- pcli/sandbox/selector.py +75 -0
- pcli/sandbox/subprocess_backend.py +376 -0
- pcli/scheduler/__init__.py +0 -0
- pcli/scheduler/daemon.py +194 -0
- pcli/scheduler/models.py +97 -0
- pcli/scheduler/runner.py +84 -0
- pcli/scheduler/store.py +75 -0
- pcli/scheduler/triggers.py +84 -0
- pcli/session/__init__.py +0 -0
- pcli/session/audit.py +122 -0
- pcli/session/directory_check.py +28 -0
- pcli/session/export.py +57 -0
- pcli/session/importer.py +92 -0
- pcli/session/models.py +168 -0
- pcli/session/store.py +127 -0
- pcli/telegram/__init__.py +0 -0
- pcli/telegram/bot.py +266 -0
- pcli/telegram/daemon.py +1197 -0
- pcli/telegram/permissions.py +131 -0
- pcli/telegram/sender.py +58 -0
- pcli/tools/__init__.py +0 -0
- pcli/tools/_nested_agent.py +204 -0
- pcli/tools/agent_tools.py +264 -0
- pcli/tools/agent_tools_store.py +69 -0
- pcli/tools/artifacts.py +47 -0
- pcli/tools/base.py +185 -0
- pcli/tools/builtin/__init__.py +0 -0
- pcli/tools/builtin/agent_tool_register_tool.py +100 -0
- pcli/tools/builtin/artifact_tool.py +212 -0
- pcli/tools/builtin/ask_tool.py +77 -0
- pcli/tools/builtin/browser_tool.py +253 -0
- pcli/tools/builtin/decision_tool.py +73 -0
- pcli/tools/builtin/describe_tool.py +389 -0
- pcli/tools/builtin/diff_tools.py +225 -0
- pcli/tools/builtin/fs_tools.py +371 -0
- pcli/tools/builtin/grep_tool.py +88 -0
- pcli/tools/builtin/memory_tool.py +108 -0
- pcli/tools/builtin/network_tools.py +107 -0
- pcli/tools/builtin/pip_tool.py +106 -0
- pcli/tools/builtin/shell_tool.py +240 -0
- pcli/tools/builtin/subagent_tool.py +146 -0
- pcli/tools/builtin/todo_tool.py +122 -0
- pcli/tools/builtin/toolbox_register_tool.py +76 -0
- pcli/tools/builtin/web_tools.py +322 -0
- pcli/tools/pydiscovery/__init__.py +0 -0
- pcli/tools/pydiscovery/cache.py +51 -0
- pcli/tools/pydiscovery/index.py +48 -0
- pcli/tools/pydiscovery/invoke.py +181 -0
- pcli/tools/pydiscovery/search.py +117 -0
- pcli/tools/registry.py +138 -0
- pcli/tools/toolbox/__init__.py +0 -0
- pcli/tools/toolbox/introspect.py +48 -0
- pcli/tools/toolbox/manager.py +336 -0
- pcli/tools/toolbox/plugin_base.py +51 -0
- pcli/tools/toolbox/plugins/__init__.py +6 -0
- pcli/tools/toolbox/plugins/httpd.py +99 -0
- pcli/tools/toolbox/plugins/kafka.py +162 -0
- pcli/tools/toolbox/plugins/kubectl.py +211 -0
- pcli/tools/toolbox/plugins/sge.py +146 -0
- pcli/tools/toolbox/store.py +65 -0
- pcli/tools/toolbox/synthesize.py +100 -0
- pcli/tui/__init__.py +0 -0
- pcli/tui/app.py +37 -0
- pcli/tui/screens/__init__.py +0 -0
- pcli/tui/screens/ask_question_modal.py +54 -0
- pcli/tui/screens/chat.py +2070 -0
- pcli/tui/screens/confirm_modal.py +39 -0
- pcli/tui/screens/models.py +43 -0
- pcli/tui/screens/permission_modal.py +71 -0
- pcli/tui/screens/sessions.py +162 -0
- pcli/tui/screens/subagent_activity_modal.py +71 -0
- pcli/tui/shell_passthrough.py +56 -0
- pcli/tui/styles/pcli.tcss +241 -0
- pcli/tui/themes.py +84 -0
- pcli/tui/widgets/__init__.py +0 -0
- pcli/tui/widgets/chat_input.py +240 -0
- pcli/tui/widgets/command_suggestions.py +33 -0
- pcli/tui/widgets/message_view.py +328 -0
- pcli/tui/widgets/paste_input.py +99 -0
- pcli/tui/widgets/paste_marker.py +69 -0
- pcli/tui/widgets/status_bar.py +133 -0
- pcli/tui/widgets/status_pane.py +58 -0
- pcli/util/__init__.py +0 -0
- pcli/util/ids.py +15 -0
- pcli/util/logging.py +18 -0
- pcli/util/text.py +10 -0
- pcli_agent-0.1.0.dist-info/METADATA +259 -0
- pcli_agent-0.1.0.dist-info/RECORD +130 -0
- pcli_agent-0.1.0.dist-info/WHEEL +4 -0
- pcli_agent-0.1.0.dist-info/entry_points.txt +2 -0
- pcli_agent-0.1.0.dist-info/licenses/LICENSE +21 -0
pcli/cost/context.py
ADDED
|
@@ -0,0 +1,275 @@
|
|
|
1
|
+
"""Tracks how much of the model's context window the conversation is using.
|
|
2
|
+
|
|
3
|
+
There's no local tokenizer for a generic gateway, so this doesn't estimate
|
|
4
|
+
from message text — it uses the most recently *reported* usage.total_tokens
|
|
5
|
+
(prompt_tokens + completion_tokens of the last actual LLM call). That number
|
|
6
|
+
is exactly the size of what gets resent as history on the next call, which
|
|
7
|
+
is what "context used" means in practice. Since CostTracker records one
|
|
8
|
+
TurnCost per underlying LLM call (not per user-visible turn — a single turn
|
|
9
|
+
with tool calls makes several), the most recent *main*-conversation entry in
|
|
10
|
+
Session.cost.turns reflects the most recent real call, tool-call round-trips
|
|
11
|
+
included.
|
|
12
|
+
|
|
13
|
+
"Most recent main-conversation entry" is deliberately not just "the last
|
|
14
|
+
entry": TurnCost.source distinguishes a real main-conversation call from a
|
|
15
|
+
subagent's or a compaction summarization's own LLM call, both of which are
|
|
16
|
+
real spend (see CostTracker.record_turn) but reflect a completely different,
|
|
17
|
+
unrelated conversation's size — trusting the literal last entry regardless
|
|
18
|
+
of source previously let a subagent call (in an edge case) or a compaction
|
|
19
|
+
call (every single time compaction ran) leave a stray, unrelated token count
|
|
20
|
+
as what the *next* turn's context-usage/max_tokens-cap math was computed
|
|
21
|
+
from. See _last_main_turn/_main_turns below, used by every function here.
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
from __future__ import annotations
|
|
25
|
+
|
|
26
|
+
import tomllib
|
|
27
|
+
from dataclasses import dataclass
|
|
28
|
+
|
|
29
|
+
from pcli.config.paths import context_limits_file
|
|
30
|
+
from pcli.cost.pricing_table import match_model_pattern
|
|
31
|
+
from pcli.llm.models import Usage
|
|
32
|
+
from pcli.session.models import Session, TurnCost
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _main_turns(turns: list[TurnCost]) -> list[TurnCost]:
|
|
36
|
+
return [turn for turn in turns if turn.source == "main"]
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _last_main_turn_usage(turns: list[TurnCost]) -> Usage | None:
|
|
40
|
+
for turn in reversed(turns):
|
|
41
|
+
if turn.source == "main":
|
|
42
|
+
return turn.usage
|
|
43
|
+
return None
|
|
44
|
+
|
|
45
|
+
DEFAULT_CONTEXT_LIMITS_TOML = """\
|
|
46
|
+
# pcli model context-window limits (tokens). Edit freely — entries here
|
|
47
|
+
# override pcli's built-in defaults. Model names support a trailing '*' as a
|
|
48
|
+
# prefix wildcard, e.g. "gpt-4o*". [default] is used for any model that
|
|
49
|
+
# matches nothing else.
|
|
50
|
+
|
|
51
|
+
[default]
|
|
52
|
+
limit = 128000
|
|
53
|
+
"""
|
|
54
|
+
|
|
55
|
+
# Best-effort starting point; intentionally approximate, user-editable.
|
|
56
|
+
_BUILTIN_LIMITS: dict[str, int] = {
|
|
57
|
+
"gpt-4o*": 128_000,
|
|
58
|
+
"gpt-4.1*": 1_000_000,
|
|
59
|
+
"gpt-4-turbo*": 128_000,
|
|
60
|
+
"gpt-3.5*": 16_385,
|
|
61
|
+
"o1*": 200_000,
|
|
62
|
+
"o3*": 200_000,
|
|
63
|
+
"claude-opus*": 200_000,
|
|
64
|
+
"claude-sonnet*": 200_000,
|
|
65
|
+
"claude-haiku*": 200_000,
|
|
66
|
+
"gemini-1.5-pro*": 2_000_000,
|
|
67
|
+
"gemini-1.5-flash*": 1_000_000,
|
|
68
|
+
"llama-3*": 128_000,
|
|
69
|
+
"mistral*": 32_000,
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
_DEFAULT_LIMIT = 128_000
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
class ContextLimitTable:
|
|
76
|
+
def __init__(
|
|
77
|
+
self, entries: dict[str, int], default: int, auto_detected: set[str] | None = None
|
|
78
|
+
) -> None:
|
|
79
|
+
self._entries = entries
|
|
80
|
+
self._default = default
|
|
81
|
+
# Exact model names (never wildcard patterns — see
|
|
82
|
+
# set_model_context_limit's exact-match docstring) whose [models]
|
|
83
|
+
# entry came from a previous auto-detect run rather than a manual
|
|
84
|
+
# /context-limit correction. Distinguishing the two matters because
|
|
85
|
+
# a local gateway's loaded context (LM Studio, Ollama, ...) can
|
|
86
|
+
# legitimately change between runs (the user reloads the model with
|
|
87
|
+
# a different context-length setting) - so an auto-detected value
|
|
88
|
+
# can go stale in a way a manual override or a hosted API's fixed
|
|
89
|
+
# limit never does. See should_attempt_detection.
|
|
90
|
+
self._auto_detected = auto_detected or set()
|
|
91
|
+
|
|
92
|
+
@classmethod
|
|
93
|
+
def load(cls) -> ContextLimitTable:
|
|
94
|
+
entries = dict(_BUILTIN_LIMITS)
|
|
95
|
+
default = _DEFAULT_LIMIT
|
|
96
|
+
auto_detected: set[str] = set()
|
|
97
|
+
|
|
98
|
+
path = context_limits_file()
|
|
99
|
+
if path.exists():
|
|
100
|
+
raw = tomllib.loads(path.read_text(encoding="utf-8"))
|
|
101
|
+
models = raw.get("models", {})
|
|
102
|
+
for pattern, limit in models.items():
|
|
103
|
+
entries[pattern] = int(limit)
|
|
104
|
+
if "default" in raw and "limit" in raw["default"]:
|
|
105
|
+
default = int(raw["default"]["limit"])
|
|
106
|
+
auto_detected = {name for name, flag in raw.get("auto_detected", {}).items() if flag}
|
|
107
|
+
|
|
108
|
+
return cls(entries, default, auto_detected)
|
|
109
|
+
|
|
110
|
+
def lookup(self, model_name: str) -> int:
|
|
111
|
+
match = match_model_pattern(model_name, self._entries)
|
|
112
|
+
return match if match is not None else self._default
|
|
113
|
+
|
|
114
|
+
def has_explicit_entry(self, model_name: str) -> bool:
|
|
115
|
+
"""True if model_name matches a real (builtin or user/auto-set)
|
|
116
|
+
entry, as opposed to lookup() silently falling back to the generic
|
|
117
|
+
default."""
|
|
118
|
+
return match_model_pattern(model_name, self._entries) is not None
|
|
119
|
+
|
|
120
|
+
def should_attempt_detection(self, model_name: str) -> bool:
|
|
121
|
+
"""Used to gate auto-detection (cost/context_detect.py): True if
|
|
122
|
+
there's no entry for this model at all (the has_explicit_entry==False
|
|
123
|
+
case), OR the entry that's there came from a previous auto-detect
|
|
124
|
+
run rather than a manual /context-limit correction or a built-in
|
|
125
|
+
default. A manual override and a built-in are trusted forever and
|
|
126
|
+
never re-probed; an auto-detected value gets re-checked on every
|
|
127
|
+
startup since it can drift out from under pcli without any pcli-side
|
|
128
|
+
signal (see the class docstring note on _auto_detected)."""
|
|
129
|
+
match = match_model_pattern(model_name, self._entries)
|
|
130
|
+
if match is None:
|
|
131
|
+
return True
|
|
132
|
+
return model_name in self._auto_detected
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
@dataclass
|
|
136
|
+
class ContextUsage:
|
|
137
|
+
used_tokens: int
|
|
138
|
+
limit_tokens: int
|
|
139
|
+
|
|
140
|
+
@property
|
|
141
|
+
def fraction(self) -> float:
|
|
142
|
+
return self.used_tokens / self.limit_tokens if self.limit_tokens else 0.0
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def current_context_usage(
|
|
146
|
+
session: Session, *, limit_table: ContextLimitTable | None = None
|
|
147
|
+
) -> ContextUsage:
|
|
148
|
+
limit_table = limit_table or ContextLimitTable.load()
|
|
149
|
+
last_main_usage = _last_main_turn_usage(session.cost.turns)
|
|
150
|
+
used = last_main_usage.total_tokens if last_main_usage is not None else 0
|
|
151
|
+
return ContextUsage(used_tokens=used, limit_tokens=limit_table.lookup(session.model))
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def compute_max_response_tokens(
|
|
155
|
+
session: Session, *, limit_table: ContextLimitTable | None = None, safety_margin: int
|
|
156
|
+
) -> int | None:
|
|
157
|
+
"""A dynamic per-request max_tokens cap: leaves just enough headroom
|
|
158
|
+
that a single response can't consume the *entire* remaining context
|
|
159
|
+
window by itself. Confirmed against a real debugged session:
|
|
160
|
+
auto-compaction only runs *between* turns, checking the fraction used
|
|
161
|
+
as of the last completed one — a turn that ended 69% full (comfortably
|
|
162
|
+
under the 80% auto-compact threshold) gave compaction no reason to run,
|
|
163
|
+
but the very next turn's own response then generated 4663 tokens in a
|
|
164
|
+
single, ~100-minute-long generation, consumed the remaining ~31% of the
|
|
165
|
+
window entirely by itself, and got hard-truncated mid-stream by the
|
|
166
|
+
gateway's own ceiling - there was no checkpoint inside that one
|
|
167
|
+
generation for compaction to catch it at. Recomputed fresh from the
|
|
168
|
+
same usage.total_tokens basis current_context_usage (and therefore
|
|
169
|
+
auto-compaction) already uses - there's no local tokenizer to do better
|
|
170
|
+
(see this module's docstring) - so it's exactly as accurate as pcli's
|
|
171
|
+
other context-usage decisions, no more, no less.
|
|
172
|
+
|
|
173
|
+
Returns None (no cap sent - the gateway's own default applies) if
|
|
174
|
+
there's no prior *main*-conversation usage yet (first turn - nothing to
|
|
175
|
+
compute headroom from yet, regardless of whether a subagent or
|
|
176
|
+
compaction call happened to run before it) or if the computed headroom
|
|
177
|
+
is already <= 0 (essentially full; sending a non-positive max_tokens
|
|
178
|
+
would be nonsensical, and by this point auto-compaction should already
|
|
179
|
+
have intervened)."""
|
|
180
|
+
if _last_main_turn_usage(session.cost.turns) is None:
|
|
181
|
+
return None
|
|
182
|
+
usage = current_context_usage(session, limit_table=limit_table)
|
|
183
|
+
available = usage.limit_tokens - usage.used_tokens - safety_margin
|
|
184
|
+
return available if available > 0 else None
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
_CEILING_MIN_TOTAL_TOKENS = 2000
|
|
188
|
+
_CEILING_STALL_RATIO = 0.1
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
def looks_like_context_ceiling(session: Session) -> bool:
|
|
192
|
+
"""True if, compared to the immediately preceding turn, prompt_tokens
|
|
193
|
+
grew (more history got sent, as it always does turn over turn) but
|
|
194
|
+
total_tokens barely moved — meaning completion_tokens got squeezed down
|
|
195
|
+
to compensate. That's the fingerprint of a real, gateway-enforced
|
|
196
|
+
context ceiling being hit, independent of whatever limit_tokens pcli
|
|
197
|
+
itself assumed via ContextLimitTable (which is silently wrong — falls
|
|
198
|
+
back to a generic 128000-token guess — for any model without a built-in
|
|
199
|
+
or user-configured entry, so the fraction-based auto-compact trigger
|
|
200
|
+
alone can't catch this: a session that's actually 100% full can read as
|
|
201
|
+
a small fraction of an assumed limit that's far too large). Ordinary
|
|
202
|
+
turn-to-turn variation has *both* prompt_tokens and total_tokens growing
|
|
203
|
+
together; only a real ceiling clamps total_tokens while prompt_tokens
|
|
204
|
+
keeps climbing.
|
|
205
|
+
|
|
206
|
+
Confirmed against a real debugged session: three consecutive turns each
|
|
207
|
+
landed on total_tokens=16384 (or within a few tokens of it) despite
|
|
208
|
+
prompt_tokens climbing every turn — the model's real ~16k window, for a
|
|
209
|
+
model pcli had no entry for (it was assuming 128000, i.e. reading the
|
|
210
|
+
session as ~13% full when it was actually exhausted).
|
|
211
|
+
|
|
212
|
+
Compares the last two *main*-conversation entries specifically (see
|
|
213
|
+
module docstring) — a subagent or compaction call sitting between them
|
|
214
|
+
in Session.cost.turns has its own unrelated prompt/total sizes and would
|
|
215
|
+
otherwise be compared as if it were consecutive turns."""
|
|
216
|
+
turns = _main_turns(session.cost.turns)
|
|
217
|
+
if len(turns) < 2:
|
|
218
|
+
return False
|
|
219
|
+
current = turns[-1].usage
|
|
220
|
+
previous = turns[-2].usage
|
|
221
|
+
if current.total_tokens < _CEILING_MIN_TOTAL_TOKENS:
|
|
222
|
+
return False
|
|
223
|
+
prompt_growth = current.prompt_tokens - previous.prompt_tokens
|
|
224
|
+
if prompt_growth <= 0:
|
|
225
|
+
return False
|
|
226
|
+
total_growth = current.total_tokens - previous.total_tokens
|
|
227
|
+
return total_growth <= prompt_growth * _CEILING_STALL_RATIO
|
|
228
|
+
|
|
229
|
+
|
|
230
|
+
def set_model_context_limit(model: str, limit: int, *, auto_detected: bool = False) -> None:
|
|
231
|
+
"""Persists a per-model context-window override to context_limits.toml's
|
|
232
|
+
[models] table (creating the file if it doesn't exist yet), so a wrong
|
|
233
|
+
built-in guess — or no entry at all — can be corrected without hand-
|
|
234
|
+
editing the file. Exact model-name match, not a wildcard pattern:
|
|
235
|
+
simpler and more predictable for a single correction than guessing a
|
|
236
|
+
sensible glob.
|
|
237
|
+
|
|
238
|
+
`auto_detected=True` (the cost/context_detect.py probe path) also
|
|
239
|
+
records the model in the [auto_detected] table, so
|
|
240
|
+
ContextLimitTable.should_attempt_detection knows to re-probe it on a
|
|
241
|
+
later startup rather than trusting it forever. `auto_detected=False`
|
|
242
|
+
(the default — the TUI's /context-limit command) is a human's deliberate
|
|
243
|
+
correction, so it always clears any prior auto-detected marker for this
|
|
244
|
+
model too: once a human has set it, it's sticky forever, even if it was
|
|
245
|
+
auto-detected before."""
|
|
246
|
+
path = context_limits_file()
|
|
247
|
+
raw: dict = {}
|
|
248
|
+
if path.exists():
|
|
249
|
+
raw = dict(tomllib.loads(path.read_text(encoding="utf-8")))
|
|
250
|
+
|
|
251
|
+
default_limit = _DEFAULT_LIMIT
|
|
252
|
+
existing_default = raw.get("default")
|
|
253
|
+
if isinstance(existing_default, dict) and "limit" in existing_default:
|
|
254
|
+
default_limit = int(existing_default["limit"])
|
|
255
|
+
|
|
256
|
+
models = dict(raw.get("models", {}))
|
|
257
|
+
models[model] = limit
|
|
258
|
+
|
|
259
|
+
auto_detected_models = {name for name, flag in raw.get("auto_detected", {}).items() if flag}
|
|
260
|
+
if auto_detected:
|
|
261
|
+
auto_detected_models.add(model)
|
|
262
|
+
else:
|
|
263
|
+
auto_detected_models.discard(model)
|
|
264
|
+
|
|
265
|
+
def _escape(value: str) -> str:
|
|
266
|
+
return value.replace("\\", "\\\\").replace('"', '\\"')
|
|
267
|
+
|
|
268
|
+
lines = ["[default]", f"limit = {default_limit}", "", "[models]"]
|
|
269
|
+
lines.extend(f'"{_escape(pattern)}" = {value}' for pattern, value in models.items())
|
|
270
|
+
if auto_detected_models:
|
|
271
|
+
lines.extend(["", "[auto_detected]"])
|
|
272
|
+
lines.extend(f'"{_escape(name)}" = true' for name in sorted(auto_detected_models))
|
|
273
|
+
|
|
274
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
275
|
+
path.write_text("\n".join(lines) + "\n", encoding="utf-8")
|
|
@@ -0,0 +1,183 @@
|
|
|
1
|
+
"""Best-effort auto-detection of a model's real context window, queried
|
|
2
|
+
directly from the gateway.
|
|
3
|
+
|
|
4
|
+
There's no single standard OpenAI-compatible endpoint for this — the vanilla
|
|
5
|
+
`GET /models` response (per the OpenAI spec) carries only `id`/`object`/
|
|
6
|
+
`created`/`owned_by`, nothing about context size. Every backend that DOES
|
|
7
|
+
expose it does so through its own extension:
|
|
8
|
+
|
|
9
|
+
- OpenRouter and vLLM both add an extra field (`context_length` /
|
|
10
|
+
`max_model_len` respectively) directly onto the standard `/models`
|
|
11
|
+
response — covered for free by the first probe below, no separate
|
|
12
|
+
endpoint needed.
|
|
13
|
+
- LM Studio has its own native `GET /api/v0/models` endpoint
|
|
14
|
+
(`max_context_length`, and possibly `loaded_context_length` for a
|
|
15
|
+
currently-loaded model).
|
|
16
|
+
- A raw llama.cpp server (not behind LM Studio) exposes `GET /props`,
|
|
17
|
+
whose `default_generation_settings.n_ctx` reflects whatever single model
|
|
18
|
+
it has loaded.
|
|
19
|
+
- Ollama exposes `POST /api/show` (`{"model": "..."}`), whose `model_info`
|
|
20
|
+
has a `<arch>.context_length` key (the architecture name varies per
|
|
21
|
+
model family, hence the suffix scan below rather than a fixed key).
|
|
22
|
+
- A LiteLLM proxy exposes `GET /model/info`, returning every configured
|
|
23
|
+
model's `model_name` alongside a `model_info` object with
|
|
24
|
+
`max_input_tokens` (the real context window — `max_tokens` on the same
|
|
25
|
+
object is unreliable/inconsistently populated across LiteLLM versions,
|
|
26
|
+
often reflecting max *output* tokens instead, so it's only used as a
|
|
27
|
+
fallback when `max_input_tokens` is absent).
|
|
28
|
+
|
|
29
|
+
Hosted-only gateways (real OpenAI, Anthropic, Azure OpenAI, most plain
|
|
30
|
+
OpenAI-compatible proxies) expose none of this — every probe here just
|
|
31
|
+
fails cleanly and detect_context_limit() returns None, exactly as if the
|
|
32
|
+
model were simply unrecognized.
|
|
33
|
+
|
|
34
|
+
The standard-/models probe deliberately reuses `client`'s own configured
|
|
35
|
+
base_url as-is (same path GatewayClient.list_models() already calls),
|
|
36
|
+
since it's a real OpenAI-style endpoint that legitimately lives under
|
|
37
|
+
whatever prefix (typically '/v1', sometimes something custom) the user
|
|
38
|
+
configured. The other four are each a fixed, well-known path at the
|
|
39
|
+
gateway's *origin* (scheme+host+port) — they must NOT inherit that prefix,
|
|
40
|
+
so `_origin_url` builds each one as a fully-qualified absolute URL rather
|
|
41
|
+
than a bare path. This matters because httpx.AsyncClient deliberately does
|
|
42
|
+
NOT treat a leading '/' as "reset to origin root" the way browsers/urljoin
|
|
43
|
+
do — Client._merge_url always appends onto base_url's own path regardless
|
|
44
|
+
of a leading slash (confirmed directly against httpx's source: it exists
|
|
45
|
+
specifically so a relative-looking path composes predictably under a
|
|
46
|
+
subpath base_url) — only a fully-qualified URL (with its own scheme)
|
|
47
|
+
bypasses that merging entirely.
|
|
48
|
+
"""
|
|
49
|
+
|
|
50
|
+
from __future__ import annotations
|
|
51
|
+
|
|
52
|
+
from collections.abc import Awaitable, Callable
|
|
53
|
+
|
|
54
|
+
import httpx
|
|
55
|
+
|
|
56
|
+
_PROBE_TIMEOUT_S = 5.0
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _positive_int(value: object) -> int | None:
|
|
60
|
+
return value if isinstance(value, int) and value > 0 else None
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def _origin_url(client: httpx.AsyncClient, path: str) -> str:
|
|
64
|
+
"""A fully-qualified URL at client's origin (scheme+host+port),
|
|
65
|
+
ignoring whatever path (e.g. '/v1') is baked into its base_url — see
|
|
66
|
+
the module docstring for why this can't just be a plain '/path'."""
|
|
67
|
+
base = client.base_url
|
|
68
|
+
port_part = f":{base.port}" if base.port else ""
|
|
69
|
+
return f"{base.scheme}://{base.host}{port_part}{path}"
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
async def _probe_standard_models_endpoint(
|
|
73
|
+
client: httpx.AsyncClient, model: str, timeout: float
|
|
74
|
+
) -> int | None:
|
|
75
|
+
response = await client.get("/models", timeout=timeout)
|
|
76
|
+
if response.status_code >= 400:
|
|
77
|
+
return None
|
|
78
|
+
entries = response.json().get("data", [])
|
|
79
|
+
for entry in entries:
|
|
80
|
+
if not isinstance(entry, dict) or entry.get("id") != model:
|
|
81
|
+
continue
|
|
82
|
+
for key in ("context_length", "max_model_len"):
|
|
83
|
+
limit = _positive_int(entry.get(key))
|
|
84
|
+
if limit is not None:
|
|
85
|
+
return limit
|
|
86
|
+
return None
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
async def _probe_lmstudio(client: httpx.AsyncClient, model: str, timeout: float) -> int | None:
|
|
90
|
+
response = await client.get(_origin_url(client, "/api/v0/models"), timeout=timeout)
|
|
91
|
+
if response.status_code >= 400:
|
|
92
|
+
return None
|
|
93
|
+
entries = response.json().get("data", [])
|
|
94
|
+
for entry in entries:
|
|
95
|
+
if not isinstance(entry, dict) or entry.get("id") != model:
|
|
96
|
+
continue
|
|
97
|
+
# loaded_context_length only appears once LM Studio has actually
|
|
98
|
+
# loaded the model into memory. LM Studio JIT-loads on first
|
|
99
|
+
# inference and unloads after an idle TTL, so at pcli's startup-time
|
|
100
|
+
# probe the model is very often still "not-loaded" - the entry then
|
|
101
|
+
# has only max_context_length, the architecture's *maximum
|
|
102
|
+
# supported* window, which can be many times larger than what
|
|
103
|
+
# actually gets allocated once loaded (confirmed against a real
|
|
104
|
+
# session: max_context_length=262144 for a model llama.cpp had
|
|
105
|
+
# loaded with n_ctx_slot=16384). Trusting that fallback previously
|
|
106
|
+
# reintroduced the exact stale-limit bug should_attempt_detection
|
|
107
|
+
# was built to fix, just from a different angle - a "don't know"
|
|
108
|
+
# (None, falls through to the next probe / the generic default)
|
|
109
|
+
# is safer than a confidently wrong number here.
|
|
110
|
+
return _positive_int(entry.get("loaded_context_length"))
|
|
111
|
+
return None
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
async def _probe_llama_cpp_props(client: httpx.AsyncClient, model: str, timeout: float) -> int | None:
|
|
115
|
+
response = await client.get(_origin_url(client, "/props"), timeout=timeout)
|
|
116
|
+
if response.status_code >= 400:
|
|
117
|
+
return None
|
|
118
|
+
settings = response.json().get("default_generation_settings", {})
|
|
119
|
+
return _positive_int(settings.get("n_ctx")) if isinstance(settings, dict) else None
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
async def _probe_ollama(client: httpx.AsyncClient, model: str, timeout: float) -> int | None:
|
|
123
|
+
response = await client.post(
|
|
124
|
+
_origin_url(client, "/api/show"), json={"model": model}, timeout=timeout
|
|
125
|
+
)
|
|
126
|
+
if response.status_code >= 400:
|
|
127
|
+
return None
|
|
128
|
+
model_info = response.json().get("model_info", {})
|
|
129
|
+
if not isinstance(model_info, dict):
|
|
130
|
+
return None
|
|
131
|
+
for key, value in model_info.items():
|
|
132
|
+
if key.endswith(".context_length"):
|
|
133
|
+
limit = _positive_int(value)
|
|
134
|
+
if limit is not None:
|
|
135
|
+
return limit
|
|
136
|
+
return None
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
async def _probe_litellm(client: httpx.AsyncClient, model: str, timeout: float) -> int | None:
|
|
140
|
+
response = await client.get(_origin_url(client, "/model/info"), timeout=timeout)
|
|
141
|
+
if response.status_code >= 400:
|
|
142
|
+
return None
|
|
143
|
+
entries = response.json().get("data", [])
|
|
144
|
+
for entry in entries:
|
|
145
|
+
if not isinstance(entry, dict) or entry.get("model_name") != model:
|
|
146
|
+
continue
|
|
147
|
+
model_info = entry.get("model_info")
|
|
148
|
+
if not isinstance(model_info, dict):
|
|
149
|
+
continue
|
|
150
|
+
for key in ("max_input_tokens", "max_tokens"):
|
|
151
|
+
limit = _positive_int(model_info.get(key))
|
|
152
|
+
if limit is not None:
|
|
153
|
+
return limit
|
|
154
|
+
return None
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
_PROBES: tuple[Callable[[httpx.AsyncClient, str, float], Awaitable[int | None]], ...] = (
|
|
158
|
+
_probe_standard_models_endpoint,
|
|
159
|
+
_probe_lmstudio,
|
|
160
|
+
_probe_llama_cpp_props,
|
|
161
|
+
_probe_ollama,
|
|
162
|
+
_probe_litellm,
|
|
163
|
+
)
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
async def detect_context_limit(
|
|
167
|
+
client: httpx.AsyncClient, model: str, *, timeout: float = _PROBE_TIMEOUT_S
|
|
168
|
+
) -> int | None:
|
|
169
|
+
"""Tries each known backend-specific probe in turn against `client`'s
|
|
170
|
+
already-configured base_url/auth, returning the first positive context
|
|
171
|
+
size found. Every probe swallows its own connection/parse failures
|
|
172
|
+
(a backend that doesn't support it looks exactly like one that isn't
|
|
173
|
+
reachable — both just mean "try the next one" or, if all fail, "give
|
|
174
|
+
up and let the caller fall back to the assumed/manual limit"), so this
|
|
175
|
+
never raises."""
|
|
176
|
+
for probe in _PROBES:
|
|
177
|
+
try:
|
|
178
|
+
limit = await probe(client, model, timeout)
|
|
179
|
+
except (httpx.HTTPError, ValueError, TypeError, AttributeError):
|
|
180
|
+
continue
|
|
181
|
+
if limit is not None:
|
|
182
|
+
return limit
|
|
183
|
+
return None
|
|
@@ -0,0 +1,141 @@
|
|
|
1
|
+
"""Per-model USD pricing. The gateway is generic/vendor-agnostic, so pricing
|
|
2
|
+
has no single source of truth — it's a small built-in best-effort table,
|
|
3
|
+
overridable by the user's pricing.toml."""
|
|
4
|
+
|
|
5
|
+
from __future__ import annotations
|
|
6
|
+
|
|
7
|
+
import tomllib
|
|
8
|
+
from typing import TypeVar
|
|
9
|
+
|
|
10
|
+
from pydantic import BaseModel
|
|
11
|
+
|
|
12
|
+
from pcli.config.paths import pricing_file
|
|
13
|
+
|
|
14
|
+
T = TypeVar("T")
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def match_model_pattern(model_name: str, entries: dict[str, T]) -> T | None:
|
|
18
|
+
"""Exact match wins; otherwise the longest '*'-suffixed prefix pattern
|
|
19
|
+
that matches. Shared by PricingTable and context.ContextLimitTable so
|
|
20
|
+
both per-model lookup tables behave identically."""
|
|
21
|
+
if model_name in entries:
|
|
22
|
+
return entries[model_name]
|
|
23
|
+
best_match: T | None = None
|
|
24
|
+
best_len = -1
|
|
25
|
+
for pattern, value in entries.items():
|
|
26
|
+
if pattern.endswith("*"):
|
|
27
|
+
prefix = pattern[:-1]
|
|
28
|
+
if model_name.startswith(prefix) and len(prefix) > best_len:
|
|
29
|
+
best_match = value
|
|
30
|
+
best_len = len(prefix)
|
|
31
|
+
return best_match
|
|
32
|
+
|
|
33
|
+
DEFAULT_PRICING_TOML = """\
|
|
34
|
+
# pcli model pricing (USD per 1M tokens). Edit freely — entries here override
|
|
35
|
+
# pcli's built-in defaults. Model names support a trailing '*' as a prefix
|
|
36
|
+
# wildcard, e.g. "gpt-4o*". The [default] table is used for any model that
|
|
37
|
+
# matches nothing else. cached_input_per_1m is optional - the rate for
|
|
38
|
+
# prompt tokens the gateway reports as served from its prompt cache (see
|
|
39
|
+
# llm/models.py's Usage.cached_tokens); omit it if you don't know the
|
|
40
|
+
# model's real cache-discount rate, and cached tokens are priced the same
|
|
41
|
+
# as ordinary input tokens instead (conservative - never undercounts spend,
|
|
42
|
+
# just won't reflect a real discount the gateway might be applying).
|
|
43
|
+
|
|
44
|
+
[default]
|
|
45
|
+
input_per_1m = 0.0
|
|
46
|
+
output_per_1m = 0.0
|
|
47
|
+
"""
|
|
48
|
+
|
|
49
|
+
# Best-effort starting point; intentionally approximate, user-editable.
|
|
50
|
+
# Third element is cached_input_per_1m (None where no well-established
|
|
51
|
+
# cache-discount rate is known for that provider) - OpenAI's documented
|
|
52
|
+
# prompt-caching discount is ~50% of input price, Anthropic's cached-read
|
|
53
|
+
# rate is ~10% of input price; both approximated here consistently rather
|
|
54
|
+
# than guessed per model. Gemini/local models are left at None: real cache
|
|
55
|
+
# pricing exists for some of them too but isn't standardized enough here to
|
|
56
|
+
# approximate with the same confidence.
|
|
57
|
+
_BUILTIN_MODELS: dict[str, tuple[float, float, float | None]] = {
|
|
58
|
+
"gpt-4o*": (2.50, 10.00, 1.25),
|
|
59
|
+
"gpt-4.1*": (2.00, 8.00, 1.00),
|
|
60
|
+
"gpt-4-turbo*": (10.00, 30.00, 5.00),
|
|
61
|
+
"gpt-3.5*": (0.50, 1.50, 0.25),
|
|
62
|
+
"o1*": (15.00, 60.00, 7.50),
|
|
63
|
+
"o3*": (2.00, 8.00, 1.00),
|
|
64
|
+
"claude-opus*": (15.00, 75.00, 1.50),
|
|
65
|
+
"claude-sonnet*": (3.00, 15.00, 0.30),
|
|
66
|
+
"claude-haiku*": (0.80, 4.00, 0.08),
|
|
67
|
+
"gemini-1.5-pro*": (1.25, 5.00, None),
|
|
68
|
+
"gemini-1.5-flash*": (0.075, 0.30, None),
|
|
69
|
+
"llama-3*": (0.20, 0.20, None),
|
|
70
|
+
"mistral*": (0.25, 0.75, None),
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
class ModelPricing(BaseModel):
|
|
75
|
+
input_per_1m: float = 0.0
|
|
76
|
+
output_per_1m: float = 0.0
|
|
77
|
+
cached_input_per_1m: float | None = None
|
|
78
|
+
"""Price for prompt tokens the gateway reports as served from its
|
|
79
|
+
prompt cache (Usage.cached_tokens, a subset of prompt_tokens) - None
|
|
80
|
+
(the common case, and the only option for a user-supplied pricing.toml
|
|
81
|
+
entry that doesn't set it) means no cache-specific rate is modeled;
|
|
82
|
+
cost_usd then falls back to pricing input_per_1m for those tokens too,
|
|
83
|
+
same as before this field existed."""
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
class PricingTable:
|
|
87
|
+
def __init__(self, entries: dict[str, ModelPricing], default: ModelPricing) -> None:
|
|
88
|
+
self._entries = entries
|
|
89
|
+
self._default = default
|
|
90
|
+
|
|
91
|
+
@classmethod
|
|
92
|
+
def load(cls) -> PricingTable:
|
|
93
|
+
entries = {
|
|
94
|
+
pattern: ModelPricing(input_per_1m=inp, output_per_1m=out, cached_input_per_1m=cached)
|
|
95
|
+
for pattern, (inp, out, cached) in _BUILTIN_MODELS.items()
|
|
96
|
+
}
|
|
97
|
+
default = ModelPricing()
|
|
98
|
+
|
|
99
|
+
path = pricing_file()
|
|
100
|
+
if path.exists():
|
|
101
|
+
raw = tomllib.loads(path.read_text(encoding="utf-8"))
|
|
102
|
+
models = raw.get("models", {})
|
|
103
|
+
for pattern, values in models.items():
|
|
104
|
+
entries[pattern] = ModelPricing(**values)
|
|
105
|
+
if "default" in raw:
|
|
106
|
+
default = ModelPricing(**raw["default"])
|
|
107
|
+
|
|
108
|
+
return cls(entries, default)
|
|
109
|
+
|
|
110
|
+
def lookup(self, model_name: str) -> ModelPricing:
|
|
111
|
+
match = match_model_pattern(model_name, self._entries)
|
|
112
|
+
return match if match is not None else self._default
|
|
113
|
+
|
|
114
|
+
def cost_usd(
|
|
115
|
+
self,
|
|
116
|
+
model_name: str,
|
|
117
|
+
*,
|
|
118
|
+
prompt_tokens: int,
|
|
119
|
+
completion_tokens: int,
|
|
120
|
+
cached_tokens: int = 0,
|
|
121
|
+
) -> float:
|
|
122
|
+
"""cached_tokens is the gateway-reported subset of prompt_tokens
|
|
123
|
+
served from its prompt cache (Usage.cached_tokens) - priced at the
|
|
124
|
+
model's cached_input_per_1m rate if known, otherwise at the
|
|
125
|
+
ordinary input rate (no assumed discount). Clamped to prompt_tokens
|
|
126
|
+
so a gateway reporting cached_tokens >= prompt_tokens (not
|
|
127
|
+
guaranteed never to happen across every backend) can't produce a
|
|
128
|
+
negative uncached-token count."""
|
|
129
|
+
pricing = self.lookup(model_name)
|
|
130
|
+
cached_tokens = min(cached_tokens, prompt_tokens)
|
|
131
|
+
uncached_tokens = prompt_tokens - cached_tokens
|
|
132
|
+
cached_rate = (
|
|
133
|
+
pricing.cached_input_per_1m
|
|
134
|
+
if pricing.cached_input_per_1m is not None
|
|
135
|
+
else pricing.input_per_1m
|
|
136
|
+
)
|
|
137
|
+
return (
|
|
138
|
+
(uncached_tokens / 1_000_000) * pricing.input_per_1m
|
|
139
|
+
+ (cached_tokens / 1_000_000) * cached_rate
|
|
140
|
+
+ (completion_tokens / 1_000_000) * pricing.output_per_1m
|
|
141
|
+
)
|