pcli-agent 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pcli/__init__.py +1 -0
- pcli/__main__.py +4 -0
- pcli/agent/__init__.py +0 -0
- pcli/agent/activity.py +116 -0
- pcli/agent/compaction.py +205 -0
- pcli/agent/context_pruning.py +88 -0
- pcli/agent/headless.py +209 -0
- pcli/agent/loop.py +442 -0
- pcli/agent/prompt.py +371 -0
- pcli/agent/runtime.py +240 -0
- pcli/browser/__init__.py +0 -0
- pcli/browser/session.py +135 -0
- pcli/cli.py +757 -0
- pcli/config/__init__.py +0 -0
- pcli/config/paths.py +95 -0
- pcli/config/settings.py +435 -0
- pcli/cost/__init__.py +0 -0
- pcli/cost/context.py +275 -0
- pcli/cost/context_detect.py +183 -0
- pcli/cost/pricing_table.py +141 -0
- pcli/cost/tracker.py +126 -0
- pcli/llm/__init__.py +0 -0
- pcli/llm/client.py +285 -0
- pcli/llm/errors.py +37 -0
- pcli/llm/models.py +100 -0
- pcli/llm/streaming.py +108 -0
- pcli/memory/__init__.py +0 -0
- pcli/memory/extraction.py +106 -0
- pcli/memory/models.py +103 -0
- pcli/memory/store.py +88 -0
- pcli/permissions/__init__.py +0 -0
- pcli/permissions/guardrails.py +219 -0
- pcli/permissions/manager.py +215 -0
- pcli/permissions/policy.py +70 -0
- pcli/sandbox/__init__.py +0 -0
- pcli/sandbox/base.py +50 -0
- pcli/sandbox/docker_backend.py +107 -0
- pcli/sandbox/limits.py +63 -0
- pcli/sandbox/null_backend.py +92 -0
- pcli/sandbox/selector.py +75 -0
- pcli/sandbox/subprocess_backend.py +376 -0
- pcli/scheduler/__init__.py +0 -0
- pcli/scheduler/daemon.py +194 -0
- pcli/scheduler/models.py +97 -0
- pcli/scheduler/runner.py +84 -0
- pcli/scheduler/store.py +75 -0
- pcli/scheduler/triggers.py +84 -0
- pcli/session/__init__.py +0 -0
- pcli/session/audit.py +122 -0
- pcli/session/directory_check.py +28 -0
- pcli/session/export.py +57 -0
- pcli/session/importer.py +92 -0
- pcli/session/models.py +168 -0
- pcli/session/store.py +127 -0
- pcli/telegram/__init__.py +0 -0
- pcli/telegram/bot.py +266 -0
- pcli/telegram/daemon.py +1197 -0
- pcli/telegram/permissions.py +131 -0
- pcli/telegram/sender.py +58 -0
- pcli/tools/__init__.py +0 -0
- pcli/tools/_nested_agent.py +204 -0
- pcli/tools/agent_tools.py +264 -0
- pcli/tools/agent_tools_store.py +69 -0
- pcli/tools/artifacts.py +47 -0
- pcli/tools/base.py +185 -0
- pcli/tools/builtin/__init__.py +0 -0
- pcli/tools/builtin/agent_tool_register_tool.py +100 -0
- pcli/tools/builtin/artifact_tool.py +212 -0
- pcli/tools/builtin/ask_tool.py +77 -0
- pcli/tools/builtin/browser_tool.py +253 -0
- pcli/tools/builtin/decision_tool.py +73 -0
- pcli/tools/builtin/describe_tool.py +389 -0
- pcli/tools/builtin/diff_tools.py +225 -0
- pcli/tools/builtin/fs_tools.py +371 -0
- pcli/tools/builtin/grep_tool.py +88 -0
- pcli/tools/builtin/memory_tool.py +108 -0
- pcli/tools/builtin/network_tools.py +107 -0
- pcli/tools/builtin/pip_tool.py +106 -0
- pcli/tools/builtin/shell_tool.py +240 -0
- pcli/tools/builtin/subagent_tool.py +146 -0
- pcli/tools/builtin/todo_tool.py +122 -0
- pcli/tools/builtin/toolbox_register_tool.py +76 -0
- pcli/tools/builtin/web_tools.py +322 -0
- pcli/tools/pydiscovery/__init__.py +0 -0
- pcli/tools/pydiscovery/cache.py +51 -0
- pcli/tools/pydiscovery/index.py +48 -0
- pcli/tools/pydiscovery/invoke.py +181 -0
- pcli/tools/pydiscovery/search.py +117 -0
- pcli/tools/registry.py +138 -0
- pcli/tools/toolbox/__init__.py +0 -0
- pcli/tools/toolbox/introspect.py +48 -0
- pcli/tools/toolbox/manager.py +336 -0
- pcli/tools/toolbox/plugin_base.py +51 -0
- pcli/tools/toolbox/plugins/__init__.py +6 -0
- pcli/tools/toolbox/plugins/httpd.py +99 -0
- pcli/tools/toolbox/plugins/kafka.py +162 -0
- pcli/tools/toolbox/plugins/kubectl.py +211 -0
- pcli/tools/toolbox/plugins/sge.py +146 -0
- pcli/tools/toolbox/store.py +65 -0
- pcli/tools/toolbox/synthesize.py +100 -0
- pcli/tui/__init__.py +0 -0
- pcli/tui/app.py +37 -0
- pcli/tui/screens/__init__.py +0 -0
- pcli/tui/screens/ask_question_modal.py +54 -0
- pcli/tui/screens/chat.py +2070 -0
- pcli/tui/screens/confirm_modal.py +39 -0
- pcli/tui/screens/models.py +43 -0
- pcli/tui/screens/permission_modal.py +71 -0
- pcli/tui/screens/sessions.py +162 -0
- pcli/tui/screens/subagent_activity_modal.py +71 -0
- pcli/tui/shell_passthrough.py +56 -0
- pcli/tui/styles/pcli.tcss +241 -0
- pcli/tui/themes.py +84 -0
- pcli/tui/widgets/__init__.py +0 -0
- pcli/tui/widgets/chat_input.py +240 -0
- pcli/tui/widgets/command_suggestions.py +33 -0
- pcli/tui/widgets/message_view.py +328 -0
- pcli/tui/widgets/paste_input.py +99 -0
- pcli/tui/widgets/paste_marker.py +69 -0
- pcli/tui/widgets/status_bar.py +133 -0
- pcli/tui/widgets/status_pane.py +58 -0
- pcli/util/__init__.py +0 -0
- pcli/util/ids.py +15 -0
- pcli/util/logging.py +18 -0
- pcli/util/text.py +10 -0
- pcli_agent-0.1.0.dist-info/METADATA +259 -0
- pcli_agent-0.1.0.dist-info/RECORD +130 -0
- pcli_agent-0.1.0.dist-info/WHEEL +4 -0
- pcli_agent-0.1.0.dist-info/entry_points.txt +2 -0
- pcli_agent-0.1.0.dist-info/licenses/LICENSE +21 -0
pcli/cost/tracker.py
ADDED
|
@@ -0,0 +1,126 @@
|
|
|
1
|
+
"""Per-turn/session/global cost aggregation."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
from collections import defaultdict
|
|
7
|
+
from datetime import UTC, datetime
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
from typing import Literal
|
|
10
|
+
|
|
11
|
+
from pcli.config.paths import cost_ledger_file
|
|
12
|
+
from pcli.cost.pricing_table import PricingTable
|
|
13
|
+
from pcli.llm.models import Usage
|
|
14
|
+
from pcli.session.models import Session, TurnCost
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class CostTracker:
|
|
18
|
+
def __init__(
|
|
19
|
+
self,
|
|
20
|
+
session: Session,
|
|
21
|
+
*,
|
|
22
|
+
pricing_table: PricingTable | None = None,
|
|
23
|
+
ledger_path: Path | None = None,
|
|
24
|
+
) -> None:
|
|
25
|
+
self._session = session
|
|
26
|
+
self._pricing = pricing_table or PricingTable.load()
|
|
27
|
+
self._ledger_path = ledger_path or cost_ledger_file()
|
|
28
|
+
|
|
29
|
+
def record_turn(
|
|
30
|
+
self,
|
|
31
|
+
model: str,
|
|
32
|
+
usage: Usage,
|
|
33
|
+
*,
|
|
34
|
+
source: Literal["main", "subagent", "compaction", "memory"] = "main",
|
|
35
|
+
) -> TurnCost:
|
|
36
|
+
cost_usd = self._pricing.cost_usd(
|
|
37
|
+
model,
|
|
38
|
+
prompt_tokens=usage.prompt_tokens,
|
|
39
|
+
completion_tokens=usage.completion_tokens,
|
|
40
|
+
cached_tokens=usage.cached_tokens or 0,
|
|
41
|
+
)
|
|
42
|
+
turn = TurnCost(
|
|
43
|
+
turn_index=len(self._session.cost.turns),
|
|
44
|
+
model=model,
|
|
45
|
+
usage=usage,
|
|
46
|
+
cost_usd=cost_usd,
|
|
47
|
+
estimated=usage.estimated,
|
|
48
|
+
source=source,
|
|
49
|
+
)
|
|
50
|
+
self._session.cost.turns.append(turn)
|
|
51
|
+
self._session.cost.session_total_usd += cost_usd
|
|
52
|
+
self._session.cost.total_tokens += usage.total_tokens
|
|
53
|
+
self._append_to_global_ledger(turn)
|
|
54
|
+
return turn
|
|
55
|
+
|
|
56
|
+
def _append_to_global_ledger(self, turn: TurnCost) -> None:
|
|
57
|
+
record = {
|
|
58
|
+
"session_id": self._session.id,
|
|
59
|
+
"turn_index": turn.turn_index,
|
|
60
|
+
"model": turn.model,
|
|
61
|
+
"prompt_tokens": turn.usage.prompt_tokens,
|
|
62
|
+
"completion_tokens": turn.usage.completion_tokens,
|
|
63
|
+
"cached_tokens": turn.usage.cached_tokens or 0,
|
|
64
|
+
"total_tokens": turn.usage.total_tokens,
|
|
65
|
+
"cost_usd": turn.cost_usd,
|
|
66
|
+
"estimated": turn.estimated,
|
|
67
|
+
"created_at": turn.created_at.isoformat(),
|
|
68
|
+
}
|
|
69
|
+
path = self._ledger_path
|
|
70
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
71
|
+
with path.open("a", encoding="utf-8") as fh:
|
|
72
|
+
fh.write(json.dumps(record) + "\n")
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def global_cost_report(since: datetime | None = None, *, ledger_path: Path | None = None) -> dict:
|
|
76
|
+
path = ledger_path or cost_ledger_file()
|
|
77
|
+
total_cost = 0.0
|
|
78
|
+
total_tokens = 0
|
|
79
|
+
turn_count = 0
|
|
80
|
+
by_model: dict[str, float] = defaultdict(float)
|
|
81
|
+
|
|
82
|
+
if path.exists():
|
|
83
|
+
with path.open("r", encoding="utf-8") as fh:
|
|
84
|
+
for line in fh:
|
|
85
|
+
line = line.strip()
|
|
86
|
+
if not line:
|
|
87
|
+
continue
|
|
88
|
+
record = json.loads(line)
|
|
89
|
+
created_at = datetime.fromisoformat(record["created_at"])
|
|
90
|
+
if created_at.tzinfo is None:
|
|
91
|
+
created_at = created_at.replace(tzinfo=UTC)
|
|
92
|
+
if since is not None and created_at < since:
|
|
93
|
+
continue
|
|
94
|
+
total_cost += record["cost_usd"]
|
|
95
|
+
total_tokens += record["total_tokens"]
|
|
96
|
+
turn_count += 1
|
|
97
|
+
by_model[record["model"]] += record["cost_usd"]
|
|
98
|
+
|
|
99
|
+
return {
|
|
100
|
+
"total_cost_usd": total_cost,
|
|
101
|
+
"total_tokens": total_tokens,
|
|
102
|
+
"turn_count": turn_count,
|
|
103
|
+
"by_model": dict(by_model),
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def cost_budget_reason(session: Session, max_session_cost_usd: float | None) -> str | None:
|
|
108
|
+
"""None if there's no cap (max_session_cost_usd unset) or spend is still
|
|
109
|
+
under it; otherwise a formatted reason string for AgentLoop.run_turn's
|
|
110
|
+
budget_check to surface as a termination note - the one place that owns
|
|
111
|
+
this wording, so every call site (main loop, headless/scheduled runs,
|
|
112
|
+
subagents, memory extraction - see their own run_turn call sites) stays
|
|
113
|
+
consistent. session.cost.session_total_usd already includes subagent/
|
|
114
|
+
compaction/memory-extraction spend folded in (CostTracker.record_turn),
|
|
115
|
+
so this one check covers all of it regardless of which loop is asking."""
|
|
116
|
+
if max_session_cost_usd is None:
|
|
117
|
+
return None
|
|
118
|
+
spent = session.cost.session_total_usd
|
|
119
|
+
if spent < max_session_cost_usd:
|
|
120
|
+
return None
|
|
121
|
+
return (
|
|
122
|
+
f"Reached the session cost budget (${max_session_cost_usd:.2f}) - spent "
|
|
123
|
+
f"${spent:.4f} so far. Raise max_session_cost_usd (in the TUI: /budget <amount>; "
|
|
124
|
+
"otherwise PCLI_MAX_SESSION_COST_USD or config.toml) to continue this session, or "
|
|
125
|
+
"start a new one."
|
|
126
|
+
)
|
pcli/llm/__init__.py
ADDED
|
File without changes
|
pcli/llm/client.py
ADDED
|
@@ -0,0 +1,285 @@
|
|
|
1
|
+
"""Async client for an OpenAI-compatible chat-completions gateway."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import asyncio
|
|
6
|
+
import random
|
|
7
|
+
from collections.abc import AsyncIterator
|
|
8
|
+
from typing import Any, Self
|
|
9
|
+
|
|
10
|
+
import httpx
|
|
11
|
+
from tenacity import retry, retry_if_exception_type, stop_after_attempt, wait_exponential_jitter
|
|
12
|
+
|
|
13
|
+
from pcli.config.settings import Settings
|
|
14
|
+
from pcli.cost.context_detect import detect_context_limit as _detect_context_limit
|
|
15
|
+
from pcli.llm.errors import GatewayError
|
|
16
|
+
from pcli.llm.models import ChatMessage, StreamEvent, ToolDefinition, Usage
|
|
17
|
+
from pcli.llm.streaming import parse_sse_stream
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _backoff_seconds(attempt: int) -> float:
|
|
21
|
+
return min(20.0, float(2**attempt)) + random.uniform(0, 1)
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class GatewayClient:
|
|
25
|
+
def __init__(self, settings: Settings) -> None:
|
|
26
|
+
if not settings.is_configured():
|
|
27
|
+
raise GatewayError("Gateway URL and API key must be configured before use.")
|
|
28
|
+
self._settings = settings
|
|
29
|
+
headers = {"Content-Type": "application/json"}
|
|
30
|
+
if settings.gateway_api_key:
|
|
31
|
+
if settings.gateway_auth_header.lower() == "authorization":
|
|
32
|
+
headers["Authorization"] = f"Bearer {settings.gateway_api_key}"
|
|
33
|
+
else:
|
|
34
|
+
headers[settings.gateway_auth_header] = settings.gateway_api_key
|
|
35
|
+
self._client = httpx.AsyncClient(
|
|
36
|
+
base_url=settings.gateway_base_url.rstrip("/"),
|
|
37
|
+
headers=headers,
|
|
38
|
+
timeout=settings.effective_request_timeout_s,
|
|
39
|
+
)
|
|
40
|
+
# The constructor-level timeout above is only a fallback baseline —
|
|
41
|
+
# every actual request passes timeout=self._effective_timeout()
|
|
42
|
+
# explicitly (httpx supports a per-request override), so a setting
|
|
43
|
+
# change (e.g. via the /timeout command) takes effect on the very
|
|
44
|
+
# next request instead of requiring GatewayClient to be rebuilt.
|
|
45
|
+
|
|
46
|
+
def _effective_timeout(self) -> float:
|
|
47
|
+
return self._settings.effective_request_timeout_s
|
|
48
|
+
|
|
49
|
+
async def aclose(self) -> None:
|
|
50
|
+
await self._client.aclose()
|
|
51
|
+
|
|
52
|
+
async def __aenter__(self) -> Self:
|
|
53
|
+
return self
|
|
54
|
+
|
|
55
|
+
async def __aexit__(self, *exc_info: object) -> None:
|
|
56
|
+
await self.aclose()
|
|
57
|
+
|
|
58
|
+
def _network_error_hint(self, exc: httpx.HTTPError) -> str | None:
|
|
59
|
+
"""Turns a raw httpx transport failure into something the user can
|
|
60
|
+
actually act on: which setting to change, and to what — grounded in
|
|
61
|
+
a real debugged case where a local model streaming at ~5 tokens/sec
|
|
62
|
+
tripped the 120s default mid-response, and pcli's only trace of it
|
|
63
|
+
was an opaque "Gateway error" with the underlying exception text."""
|
|
64
|
+
timeout = self._settings.effective_request_timeout_s
|
|
65
|
+
suggestion = max(600, int(timeout * 3))
|
|
66
|
+
gateway_kind = "local API gateway" if self._settings.is_local_api() else "gateway"
|
|
67
|
+
|
|
68
|
+
if isinstance(exc, httpx.ConnectTimeout):
|
|
69
|
+
return (
|
|
70
|
+
f"Couldn't connect to the {gateway_kind} within {timeout:g}s "
|
|
71
|
+
f"(request_timeout_s={timeout:g}). Check it's actually running at "
|
|
72
|
+
f"{self._settings.gateway_base_url!r} — or, if it's just slow to accept "
|
|
73
|
+
f"connections (e.g. still loading a model), set a larger request_timeout_s "
|
|
74
|
+
f"(e.g. {suggestion}) via PCLI_REQUEST_TIMEOUT_S or config.toml."
|
|
75
|
+
)
|
|
76
|
+
if isinstance(exc, httpx.TimeoutException):
|
|
77
|
+
local_note = (
|
|
78
|
+
"Local models are often much slower than hosted ones — " if self._settings.is_local_api() else ""
|
|
79
|
+
)
|
|
80
|
+
return (
|
|
81
|
+
f"Read timed out on the {gateway_kind} (request_timeout_s={timeout:g}). "
|
|
82
|
+
f"{local_note}Set a larger request_timeout_s (e.g. {suggestion}) via "
|
|
83
|
+
"PCLI_REQUEST_TIMEOUT_S or config.toml."
|
|
84
|
+
)
|
|
85
|
+
if isinstance(exc, httpx.ConnectError):
|
|
86
|
+
return (
|
|
87
|
+
f"Couldn't reach the {gateway_kind} at {self._settings.gateway_base_url!r} — "
|
|
88
|
+
"check that it's running and that gateway_base_url is correct."
|
|
89
|
+
)
|
|
90
|
+
return None
|
|
91
|
+
|
|
92
|
+
def _http_status_hint(self, status_code: int, body: str = "") -> str | None:
|
|
93
|
+
# Checked before the status-code-specific branches below: OpenAI-
|
|
94
|
+
# compatible gateways (including llama.cpp/LM Studio) report a
|
|
95
|
+
# context-length overflow as a 400 with wording that varies by
|
|
96
|
+
# backend, so this is a text match on the body rather than a status
|
|
97
|
+
# code alone. Worth surfacing as its own hint rather than a raw
|
|
98
|
+
# gateway error string, since the cause isn't obvious from a bare
|
|
99
|
+
# "invalid_request_error" — this is also the one context-length
|
|
100
|
+
# failure mode pcli's own turn-based auto-compaction can't catch: a
|
|
101
|
+
# subagent's own nested tool-calling loop has no compaction of its
|
|
102
|
+
# own (see tools/_nested_agent.py), so a long subagent task can hit
|
|
103
|
+
# this mid-turn with no prior warning.
|
|
104
|
+
body_lower = body.lower()
|
|
105
|
+
if status_code == 400 and (
|
|
106
|
+
"context_length_exceeded" in body_lower
|
|
107
|
+
or "context length" in body_lower
|
|
108
|
+
or "context window" in body_lower
|
|
109
|
+
or ("maximum" in body_lower and "token" in body_lower)
|
|
110
|
+
):
|
|
111
|
+
return (
|
|
112
|
+
"This looks like a context-length overflow — the request (conversation history "
|
|
113
|
+
"plus tool results) is larger than the model can accept in one call. If this is "
|
|
114
|
+
"the main conversation, /context-limit sets the context window pcli assumes for "
|
|
115
|
+
"auto-compaction; if it's a subagent's own task, its conversation has no "
|
|
116
|
+
"compaction of its own, so try splitting the task into smaller, narrower steps."
|
|
117
|
+
)
|
|
118
|
+
if status_code in (401, 403):
|
|
119
|
+
if self._settings.gateway_api_key:
|
|
120
|
+
return "The gateway rejected the configured gateway_api_key — check it's correct and hasn't expired."
|
|
121
|
+
return (
|
|
122
|
+
"No gateway_api_key is configured and this gateway appears to require one — "
|
|
123
|
+
"set PCLI_GATEWAY_API_KEY or gateway_api_key in config.toml."
|
|
124
|
+
)
|
|
125
|
+
if status_code == 429:
|
|
126
|
+
return (
|
|
127
|
+
f"Rate-limited by the gateway. pcli already retries with backoff (up to "
|
|
128
|
+
f"max_retries={self._settings.max_retries}) — if this keeps happening, raise "
|
|
129
|
+
"max_retries or reduce request frequency."
|
|
130
|
+
)
|
|
131
|
+
if status_code >= 500:
|
|
132
|
+
return "This is usually transient on the gateway's side — trying again shortly often helps."
|
|
133
|
+
return None
|
|
134
|
+
|
|
135
|
+
def _build_payload(
|
|
136
|
+
self,
|
|
137
|
+
messages: list[ChatMessage],
|
|
138
|
+
*,
|
|
139
|
+
model: str | None,
|
|
140
|
+
tools: list[ToolDefinition] | None,
|
|
141
|
+
stream: bool,
|
|
142
|
+
temperature: float | None,
|
|
143
|
+
max_tokens: int | None = None,
|
|
144
|
+
) -> dict[str, Any]:
|
|
145
|
+
payload: dict[str, Any] = {
|
|
146
|
+
"model": model or self._settings.default_model,
|
|
147
|
+
"messages": [m.to_wire() for m in messages],
|
|
148
|
+
"stream": stream,
|
|
149
|
+
}
|
|
150
|
+
if tools:
|
|
151
|
+
payload["tools"] = [t.model_dump() for t in tools]
|
|
152
|
+
if temperature is not None:
|
|
153
|
+
payload["temperature"] = temperature
|
|
154
|
+
if max_tokens is not None:
|
|
155
|
+
payload["max_tokens"] = max_tokens
|
|
156
|
+
if stream:
|
|
157
|
+
payload["stream_options"] = {"include_usage": True}
|
|
158
|
+
return payload
|
|
159
|
+
|
|
160
|
+
async def chat_stream(
|
|
161
|
+
self,
|
|
162
|
+
messages: list[ChatMessage],
|
|
163
|
+
*,
|
|
164
|
+
model: str | None = None,
|
|
165
|
+
tools: list[ToolDefinition] | None = None,
|
|
166
|
+
temperature: float | None = None,
|
|
167
|
+
max_tokens: int | None = None,
|
|
168
|
+
) -> AsyncIterator[StreamEvent]:
|
|
169
|
+
"""Streams chat events. Retries (with backoff) only apply to failures that
|
|
170
|
+
happen before any event has been yielded — once partial output has reached
|
|
171
|
+
the caller, retrying the whole request would duplicate it, so a mid-stream
|
|
172
|
+
failure is raised immediately instead.
|
|
173
|
+
|
|
174
|
+
`max_tokens`, when given, is the dynamic per-request cap computed by
|
|
175
|
+
cost/context.py's compute_max_response_tokens — see AgentLoop, which
|
|
176
|
+
is what actually sets it.
|
|
177
|
+
"""
|
|
178
|
+
payload = self._build_payload(
|
|
179
|
+
messages, model=model, tools=tools, stream=True, temperature=temperature,
|
|
180
|
+
max_tokens=max_tokens,
|
|
181
|
+
)
|
|
182
|
+
max_attempts = max(1, self._settings.max_retries)
|
|
183
|
+
last_error: GatewayError | None = None
|
|
184
|
+
|
|
185
|
+
for attempt in range(1, max_attempts + 1):
|
|
186
|
+
started = False
|
|
187
|
+
try:
|
|
188
|
+
async with self._client.stream(
|
|
189
|
+
"POST", "/chat/completions", json=payload, timeout=self._effective_timeout()
|
|
190
|
+
) as response:
|
|
191
|
+
if response.status_code >= 400:
|
|
192
|
+
body = await response.aread()
|
|
193
|
+
body_text = body.decode(errors="replace")
|
|
194
|
+
raise GatewayError.from_http_status(
|
|
195
|
+
response.status_code,
|
|
196
|
+
body_text,
|
|
197
|
+
hint=self._http_status_hint(response.status_code, body_text),
|
|
198
|
+
)
|
|
199
|
+
async for event in parse_sse_stream(response.aiter_lines()):
|
|
200
|
+
started = True
|
|
201
|
+
yield event
|
|
202
|
+
return
|
|
203
|
+
except GatewayError as exc:
|
|
204
|
+
last_error = exc
|
|
205
|
+
if not exc.retryable or started or attempt == max_attempts:
|
|
206
|
+
raise
|
|
207
|
+
except httpx.HTTPError as exc:
|
|
208
|
+
last_error = GatewayError.from_network_error(str(exc), hint=self._network_error_hint(exc))
|
|
209
|
+
if started or attempt == max_attempts:
|
|
210
|
+
raise last_error
|
|
211
|
+
|
|
212
|
+
await asyncio.sleep(_backoff_seconds(attempt))
|
|
213
|
+
|
|
214
|
+
if last_error:
|
|
215
|
+
raise last_error
|
|
216
|
+
|
|
217
|
+
async def collect(
|
|
218
|
+
self,
|
|
219
|
+
messages: list[ChatMessage],
|
|
220
|
+
*,
|
|
221
|
+
model: str | None = None,
|
|
222
|
+
tools: list[ToolDefinition] | None = None,
|
|
223
|
+
temperature: float | None = None,
|
|
224
|
+
max_tokens: int | None = None,
|
|
225
|
+
) -> tuple[ChatMessage, Usage]:
|
|
226
|
+
"""Consumes chat_stream and returns the final assembled message + usage."""
|
|
227
|
+
from pcli.llm.models import ToolCall, ToolCallCompleteEvent, UsageEvent
|
|
228
|
+
|
|
229
|
+
text_parts: list[str] = []
|
|
230
|
+
tool_calls: list[ToolCall] = []
|
|
231
|
+
usage = Usage()
|
|
232
|
+
|
|
233
|
+
async for event in self.chat_stream(
|
|
234
|
+
messages, model=model, tools=tools, temperature=temperature, max_tokens=max_tokens
|
|
235
|
+
):
|
|
236
|
+
if event.kind == "text_delta":
|
|
237
|
+
text_parts.append(event.text)
|
|
238
|
+
elif isinstance(event, ToolCallCompleteEvent):
|
|
239
|
+
tool_calls = event.tool_calls
|
|
240
|
+
elif isinstance(event, UsageEvent):
|
|
241
|
+
usage = event.usage
|
|
242
|
+
|
|
243
|
+
message = ChatMessage(
|
|
244
|
+
role="assistant",
|
|
245
|
+
content="".join(text_parts) or None,
|
|
246
|
+
tool_calls=tool_calls or None,
|
|
247
|
+
)
|
|
248
|
+
return message, usage
|
|
249
|
+
|
|
250
|
+
@retry(
|
|
251
|
+
stop=stop_after_attempt(3),
|
|
252
|
+
wait=wait_exponential_jitter(initial=1, max=10),
|
|
253
|
+
retry=retry_if_exception_type(httpx.HTTPError),
|
|
254
|
+
reraise=True,
|
|
255
|
+
)
|
|
256
|
+
async def health_check(self) -> bool:
|
|
257
|
+
"""Best-effort reachability probe against the gateway's models endpoint."""
|
|
258
|
+
response = await self._client.get("/models", timeout=self._effective_timeout())
|
|
259
|
+
return response.status_code < 500
|
|
260
|
+
|
|
261
|
+
async def list_models(self) -> list[str]:
|
|
262
|
+
"""Fetches available model IDs from the gateway's OpenAI-compatible
|
|
263
|
+
`GET /models` endpoint (`{"data": [{"id": "..."}, ...]}`)."""
|
|
264
|
+
try:
|
|
265
|
+
response = await self._client.get("/models", timeout=self._effective_timeout())
|
|
266
|
+
except httpx.HTTPError as exc:
|
|
267
|
+
raise GatewayError.from_network_error(str(exc), hint=self._network_error_hint(exc)) from exc
|
|
268
|
+
if response.status_code >= 400:
|
|
269
|
+
raise GatewayError.from_http_status(
|
|
270
|
+
response.status_code,
|
|
271
|
+
response.text,
|
|
272
|
+
hint=self._http_status_hint(response.status_code, response.text),
|
|
273
|
+
)
|
|
274
|
+
payload = response.json()
|
|
275
|
+
entries = payload.get("data", []) if isinstance(payload, dict) else []
|
|
276
|
+
ids = [entry["id"] for entry in entries if isinstance(entry, dict) and "id" in entry]
|
|
277
|
+
return sorted(ids)
|
|
278
|
+
|
|
279
|
+
async def detect_context_limit(self, model: str) -> int | None:
|
|
280
|
+
"""Best-effort: queries this gateway directly for model's real
|
|
281
|
+
context window, trying several known backend-specific extensions
|
|
282
|
+
(see cost/context_detect.py — no single standard covers this).
|
|
283
|
+
None if nothing answered, including for hosted-only gateways that
|
|
284
|
+
expose no such information at all. Never raises."""
|
|
285
|
+
return await _detect_context_limit(self._client, model)
|
pcli/llm/errors.py
ADDED
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
"""Error hierarchy for the gateway client, with retry classification."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
class GatewayError(Exception):
|
|
7
|
+
"""Raised for any failure talking to the LLM gateway."""
|
|
8
|
+
|
|
9
|
+
def __init__(
|
|
10
|
+
self,
|
|
11
|
+
message: str,
|
|
12
|
+
*,
|
|
13
|
+
status_code: int | None = None,
|
|
14
|
+
retryable: bool = False,
|
|
15
|
+
) -> None:
|
|
16
|
+
super().__init__(message)
|
|
17
|
+
self.message = message
|
|
18
|
+
self.status_code = status_code
|
|
19
|
+
self.retryable = retryable
|
|
20
|
+
|
|
21
|
+
@classmethod
|
|
22
|
+
def from_http_status(cls, status_code: int, message: str, *, hint: str | None = None) -> GatewayError:
|
|
23
|
+
retryable = status_code == 429 or status_code >= 500
|
|
24
|
+
full_message = f"{message} {hint}" if hint else message
|
|
25
|
+
return cls(full_message, status_code=status_code, retryable=retryable)
|
|
26
|
+
|
|
27
|
+
@classmethod
|
|
28
|
+
def from_network_error(cls, message: str, *, hint: str | None = None) -> GatewayError:
|
|
29
|
+
# hint is folded directly into .message (not kept as a separate
|
|
30
|
+
# attribute) so every existing call site that already displays
|
|
31
|
+
# str(exc)/exc.message — chat.py's turn/compaction/models handlers,
|
|
32
|
+
# subagent_tool.py's "Subagent failed: {exc}", ... — picks up
|
|
33
|
+
# actionable config guidance automatically, with no per-site change
|
|
34
|
+
# needed. See GatewayClient._network_error_hint/_http_status_hint
|
|
35
|
+
# for what gets suggested and why.
|
|
36
|
+
full_message = f"{message} {hint}" if hint else message
|
|
37
|
+
return cls(full_message, status_code=None, retryable=True)
|
pcli/llm/models.py
ADDED
|
@@ -0,0 +1,100 @@
|
|
|
1
|
+
"""Pydantic models for the OpenAI-compatible chat-completions wire format."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Any, Literal
|
|
6
|
+
|
|
7
|
+
from pydantic import BaseModel
|
|
8
|
+
|
|
9
|
+
Role = Literal["system", "user", "assistant", "tool"]
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class ToolCallFunction(BaseModel):
|
|
13
|
+
name: str
|
|
14
|
+
arguments: str = ""
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class ToolCall(BaseModel):
|
|
18
|
+
id: str
|
|
19
|
+
type: Literal["function"] = "function"
|
|
20
|
+
function: ToolCallFunction
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class ChatMessage(BaseModel):
|
|
24
|
+
role: Role
|
|
25
|
+
content: str | None = None
|
|
26
|
+
tool_calls: list[ToolCall] | None = None
|
|
27
|
+
tool_call_id: str | None = None
|
|
28
|
+
name: str | None = None
|
|
29
|
+
|
|
30
|
+
def to_wire(self) -> dict[str, Any]:
|
|
31
|
+
data: dict[str, Any] = {"role": self.role}
|
|
32
|
+
if self.content is not None:
|
|
33
|
+
data["content"] = self.content
|
|
34
|
+
if self.tool_calls:
|
|
35
|
+
data["tool_calls"] = [tc.model_dump() for tc in self.tool_calls]
|
|
36
|
+
if self.tool_call_id:
|
|
37
|
+
data["tool_call_id"] = self.tool_call_id
|
|
38
|
+
if self.name:
|
|
39
|
+
data["name"] = self.name
|
|
40
|
+
return data
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
class ToolDefinition(BaseModel):
|
|
44
|
+
"""A single entry in the OpenAI-style `tools=[...]` request payload."""
|
|
45
|
+
|
|
46
|
+
type: Literal["function"] = "function"
|
|
47
|
+
function: dict[str, Any]
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
class Usage(BaseModel):
|
|
51
|
+
prompt_tokens: int = 0
|
|
52
|
+
completion_tokens: int = 0
|
|
53
|
+
total_tokens: int = 0
|
|
54
|
+
cached_tokens: int | None = None
|
|
55
|
+
estimated: bool = False
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
class TextDelta(BaseModel):
|
|
59
|
+
kind: Literal["text_delta"] = "text_delta"
|
|
60
|
+
text: str
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
class ReasoningDelta(BaseModel):
|
|
64
|
+
"""A fragment of a reasoning/"thinking" model's chain-of-thought,
|
|
65
|
+
streamed under `delta.reasoning_content` (or `delta.reasoning`) by
|
|
66
|
+
OpenAI-compatible servers that expose it as a channel separate from the
|
|
67
|
+
final answer (`delta.content`) — e.g. vLLM/SGLang/LM Studio serving
|
|
68
|
+
DeepSeek-R1-style or Nemotron "thinking" models. Never folded into the
|
|
69
|
+
assistant message's actual content; see agent/loop.py."""
|
|
70
|
+
|
|
71
|
+
kind: Literal["reasoning_delta"] = "reasoning_delta"
|
|
72
|
+
text: str
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
class ToolCallDelta(BaseModel):
|
|
76
|
+
kind: Literal["tool_call_delta"] = "tool_call_delta"
|
|
77
|
+
index: int
|
|
78
|
+
id: str | None = None
|
|
79
|
+
name: str | None = None
|
|
80
|
+
arguments_fragment: str = ""
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
class ToolCallCompleteEvent(BaseModel):
|
|
84
|
+
kind: Literal["tool_call_complete"] = "tool_call_complete"
|
|
85
|
+
tool_calls: list[ToolCall]
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
class UsageEvent(BaseModel):
|
|
89
|
+
kind: Literal["usage"] = "usage"
|
|
90
|
+
usage: Usage
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
class FinishEvent(BaseModel):
|
|
94
|
+
kind: Literal["finish"] = "finish"
|
|
95
|
+
reason: str | None = None
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
StreamEvent = (
|
|
99
|
+
TextDelta | ReasoningDelta | ToolCallDelta | ToolCallCompleteEvent | UsageEvent | FinishEvent
|
|
100
|
+
)
|
pcli/llm/streaming.py
ADDED
|
@@ -0,0 +1,108 @@
|
|
|
1
|
+
"""Parses an OpenAI-compatible chat-completions SSE stream into StreamEvents.
|
|
2
|
+
|
|
3
|
+
Tool-call arguments arrive fragmented across multiple chunks, keyed by the
|
|
4
|
+
`index` field within `delta.tool_calls`; fragments must be accumulated until
|
|
5
|
+
`finish_reason == "tool_calls"` before the full arguments JSON is usable.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import json
|
|
11
|
+
from collections.abc import AsyncIterator
|
|
12
|
+
|
|
13
|
+
from pcli.llm.models import (
|
|
14
|
+
FinishEvent,
|
|
15
|
+
ReasoningDelta,
|
|
16
|
+
StreamEvent,
|
|
17
|
+
TextDelta,
|
|
18
|
+
ToolCall,
|
|
19
|
+
ToolCallCompleteEvent,
|
|
20
|
+
ToolCallDelta,
|
|
21
|
+
ToolCallFunction,
|
|
22
|
+
Usage,
|
|
23
|
+
UsageEvent,
|
|
24
|
+
)
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class _ToolCallAccumulator:
|
|
28
|
+
def __init__(self) -> None:
|
|
29
|
+
self.id: str | None = None
|
|
30
|
+
self.name: str | None = None
|
|
31
|
+
self.arguments = ""
|
|
32
|
+
|
|
33
|
+
def to_tool_call(self) -> ToolCall:
|
|
34
|
+
return ToolCall(
|
|
35
|
+
id=self.id or "",
|
|
36
|
+
function=ToolCallFunction(name=self.name or "", arguments=self.arguments),
|
|
37
|
+
)
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
async def parse_sse_stream(lines: AsyncIterator[str]) -> AsyncIterator[StreamEvent]:
|
|
41
|
+
accumulators: dict[int, _ToolCallAccumulator] = {}
|
|
42
|
+
|
|
43
|
+
async for raw_line in lines:
|
|
44
|
+
line = raw_line.strip()
|
|
45
|
+
if not line or not line.startswith("data:"):
|
|
46
|
+
continue
|
|
47
|
+
payload = line[len("data:") :].strip()
|
|
48
|
+
if payload == "[DONE]":
|
|
49
|
+
break
|
|
50
|
+
try:
|
|
51
|
+
chunk = json.loads(payload)
|
|
52
|
+
except json.JSONDecodeError:
|
|
53
|
+
continue
|
|
54
|
+
|
|
55
|
+
usage = chunk.get("usage")
|
|
56
|
+
if usage:
|
|
57
|
+
yield UsageEvent(
|
|
58
|
+
usage=Usage(
|
|
59
|
+
prompt_tokens=usage.get("prompt_tokens", 0),
|
|
60
|
+
completion_tokens=usage.get("completion_tokens", 0),
|
|
61
|
+
total_tokens=usage.get("total_tokens", 0),
|
|
62
|
+
cached_tokens=(usage.get("prompt_tokens_details") or {}).get("cached_tokens"),
|
|
63
|
+
)
|
|
64
|
+
)
|
|
65
|
+
|
|
66
|
+
choices = chunk.get("choices") or []
|
|
67
|
+
if not choices:
|
|
68
|
+
continue
|
|
69
|
+
choice = choices[0]
|
|
70
|
+
delta = choice.get("delta") or {}
|
|
71
|
+
|
|
72
|
+
content = delta.get("content")
|
|
73
|
+
if content:
|
|
74
|
+
yield TextDelta(text=content)
|
|
75
|
+
|
|
76
|
+
# Reasoning/"thinking" models (DeepSeek-R1-style, Nemotron "detailed
|
|
77
|
+
# thinking", ...) served through vLLM/SGLang/LM Studio stream their
|
|
78
|
+
# chain-of-thought under a separate field, never under `content` —
|
|
79
|
+
# a model that spends its whole response reasoning without ever
|
|
80
|
+
# transitioning to `content` previously vanished here entirely: no
|
|
81
|
+
# text, no tool call, no error, just a silently empty turn. Servers
|
|
82
|
+
# aren't consistent about the key name, so both are checked.
|
|
83
|
+
reasoning = delta.get("reasoning_content") or delta.get("reasoning")
|
|
84
|
+
if reasoning:
|
|
85
|
+
yield ReasoningDelta(text=reasoning)
|
|
86
|
+
|
|
87
|
+
for tc in delta.get("tool_calls") or []:
|
|
88
|
+
index = tc.get("index", 0)
|
|
89
|
+
acc = accumulators.setdefault(index, _ToolCallAccumulator())
|
|
90
|
+
if tc.get("id"):
|
|
91
|
+
acc.id = tc["id"]
|
|
92
|
+
fn = tc.get("function") or {}
|
|
93
|
+
if fn.get("name"):
|
|
94
|
+
acc.name = fn["name"]
|
|
95
|
+
frag = fn.get("arguments") or ""
|
|
96
|
+
if frag:
|
|
97
|
+
acc.arguments += frag
|
|
98
|
+
yield ToolCallDelta(
|
|
99
|
+
index=index, id=tc.get("id"), name=fn.get("name"), arguments_fragment=frag
|
|
100
|
+
)
|
|
101
|
+
|
|
102
|
+
finish_reason = choice.get("finish_reason")
|
|
103
|
+
if finish_reason:
|
|
104
|
+
if finish_reason == "tool_calls" and accumulators:
|
|
105
|
+
completed = [accumulators[i].to_tool_call() for i in sorted(accumulators)]
|
|
106
|
+
yield ToolCallCompleteEvent(tool_calls=completed)
|
|
107
|
+
accumulators.clear()
|
|
108
|
+
yield FinishEvent(reason=finish_reason)
|
pcli/memory/__init__.py
ADDED
|
File without changes
|