pcli-agent 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (130) hide show
  1. pcli/__init__.py +1 -0
  2. pcli/__main__.py +4 -0
  3. pcli/agent/__init__.py +0 -0
  4. pcli/agent/activity.py +116 -0
  5. pcli/agent/compaction.py +205 -0
  6. pcli/agent/context_pruning.py +88 -0
  7. pcli/agent/headless.py +209 -0
  8. pcli/agent/loop.py +442 -0
  9. pcli/agent/prompt.py +371 -0
  10. pcli/agent/runtime.py +240 -0
  11. pcli/browser/__init__.py +0 -0
  12. pcli/browser/session.py +135 -0
  13. pcli/cli.py +757 -0
  14. pcli/config/__init__.py +0 -0
  15. pcli/config/paths.py +95 -0
  16. pcli/config/settings.py +435 -0
  17. pcli/cost/__init__.py +0 -0
  18. pcli/cost/context.py +275 -0
  19. pcli/cost/context_detect.py +183 -0
  20. pcli/cost/pricing_table.py +141 -0
  21. pcli/cost/tracker.py +126 -0
  22. pcli/llm/__init__.py +0 -0
  23. pcli/llm/client.py +285 -0
  24. pcli/llm/errors.py +37 -0
  25. pcli/llm/models.py +100 -0
  26. pcli/llm/streaming.py +108 -0
  27. pcli/memory/__init__.py +0 -0
  28. pcli/memory/extraction.py +106 -0
  29. pcli/memory/models.py +103 -0
  30. pcli/memory/store.py +88 -0
  31. pcli/permissions/__init__.py +0 -0
  32. pcli/permissions/guardrails.py +219 -0
  33. pcli/permissions/manager.py +215 -0
  34. pcli/permissions/policy.py +70 -0
  35. pcli/sandbox/__init__.py +0 -0
  36. pcli/sandbox/base.py +50 -0
  37. pcli/sandbox/docker_backend.py +107 -0
  38. pcli/sandbox/limits.py +63 -0
  39. pcli/sandbox/null_backend.py +92 -0
  40. pcli/sandbox/selector.py +75 -0
  41. pcli/sandbox/subprocess_backend.py +376 -0
  42. pcli/scheduler/__init__.py +0 -0
  43. pcli/scheduler/daemon.py +194 -0
  44. pcli/scheduler/models.py +97 -0
  45. pcli/scheduler/runner.py +84 -0
  46. pcli/scheduler/store.py +75 -0
  47. pcli/scheduler/triggers.py +84 -0
  48. pcli/session/__init__.py +0 -0
  49. pcli/session/audit.py +122 -0
  50. pcli/session/directory_check.py +28 -0
  51. pcli/session/export.py +57 -0
  52. pcli/session/importer.py +92 -0
  53. pcli/session/models.py +168 -0
  54. pcli/session/store.py +127 -0
  55. pcli/telegram/__init__.py +0 -0
  56. pcli/telegram/bot.py +266 -0
  57. pcli/telegram/daemon.py +1197 -0
  58. pcli/telegram/permissions.py +131 -0
  59. pcli/telegram/sender.py +58 -0
  60. pcli/tools/__init__.py +0 -0
  61. pcli/tools/_nested_agent.py +204 -0
  62. pcli/tools/agent_tools.py +264 -0
  63. pcli/tools/agent_tools_store.py +69 -0
  64. pcli/tools/artifacts.py +47 -0
  65. pcli/tools/base.py +185 -0
  66. pcli/tools/builtin/__init__.py +0 -0
  67. pcli/tools/builtin/agent_tool_register_tool.py +100 -0
  68. pcli/tools/builtin/artifact_tool.py +212 -0
  69. pcli/tools/builtin/ask_tool.py +77 -0
  70. pcli/tools/builtin/browser_tool.py +253 -0
  71. pcli/tools/builtin/decision_tool.py +73 -0
  72. pcli/tools/builtin/describe_tool.py +389 -0
  73. pcli/tools/builtin/diff_tools.py +225 -0
  74. pcli/tools/builtin/fs_tools.py +371 -0
  75. pcli/tools/builtin/grep_tool.py +88 -0
  76. pcli/tools/builtin/memory_tool.py +108 -0
  77. pcli/tools/builtin/network_tools.py +107 -0
  78. pcli/tools/builtin/pip_tool.py +106 -0
  79. pcli/tools/builtin/shell_tool.py +240 -0
  80. pcli/tools/builtin/subagent_tool.py +146 -0
  81. pcli/tools/builtin/todo_tool.py +122 -0
  82. pcli/tools/builtin/toolbox_register_tool.py +76 -0
  83. pcli/tools/builtin/web_tools.py +322 -0
  84. pcli/tools/pydiscovery/__init__.py +0 -0
  85. pcli/tools/pydiscovery/cache.py +51 -0
  86. pcli/tools/pydiscovery/index.py +48 -0
  87. pcli/tools/pydiscovery/invoke.py +181 -0
  88. pcli/tools/pydiscovery/search.py +117 -0
  89. pcli/tools/registry.py +138 -0
  90. pcli/tools/toolbox/__init__.py +0 -0
  91. pcli/tools/toolbox/introspect.py +48 -0
  92. pcli/tools/toolbox/manager.py +336 -0
  93. pcli/tools/toolbox/plugin_base.py +51 -0
  94. pcli/tools/toolbox/plugins/__init__.py +6 -0
  95. pcli/tools/toolbox/plugins/httpd.py +99 -0
  96. pcli/tools/toolbox/plugins/kafka.py +162 -0
  97. pcli/tools/toolbox/plugins/kubectl.py +211 -0
  98. pcli/tools/toolbox/plugins/sge.py +146 -0
  99. pcli/tools/toolbox/store.py +65 -0
  100. pcli/tools/toolbox/synthesize.py +100 -0
  101. pcli/tui/__init__.py +0 -0
  102. pcli/tui/app.py +37 -0
  103. pcli/tui/screens/__init__.py +0 -0
  104. pcli/tui/screens/ask_question_modal.py +54 -0
  105. pcli/tui/screens/chat.py +2070 -0
  106. pcli/tui/screens/confirm_modal.py +39 -0
  107. pcli/tui/screens/models.py +43 -0
  108. pcli/tui/screens/permission_modal.py +71 -0
  109. pcli/tui/screens/sessions.py +162 -0
  110. pcli/tui/screens/subagent_activity_modal.py +71 -0
  111. pcli/tui/shell_passthrough.py +56 -0
  112. pcli/tui/styles/pcli.tcss +241 -0
  113. pcli/tui/themes.py +84 -0
  114. pcli/tui/widgets/__init__.py +0 -0
  115. pcli/tui/widgets/chat_input.py +240 -0
  116. pcli/tui/widgets/command_suggestions.py +33 -0
  117. pcli/tui/widgets/message_view.py +328 -0
  118. pcli/tui/widgets/paste_input.py +99 -0
  119. pcli/tui/widgets/paste_marker.py +69 -0
  120. pcli/tui/widgets/status_bar.py +133 -0
  121. pcli/tui/widgets/status_pane.py +58 -0
  122. pcli/util/__init__.py +0 -0
  123. pcli/util/ids.py +15 -0
  124. pcli/util/logging.py +18 -0
  125. pcli/util/text.py +10 -0
  126. pcli_agent-0.1.0.dist-info/METADATA +259 -0
  127. pcli_agent-0.1.0.dist-info/RECORD +130 -0
  128. pcli_agent-0.1.0.dist-info/WHEEL +4 -0
  129. pcli_agent-0.1.0.dist-info/entry_points.txt +2 -0
  130. pcli_agent-0.1.0.dist-info/licenses/LICENSE +21 -0
pcli/cost/tracker.py ADDED
@@ -0,0 +1,126 @@
1
+ """Per-turn/session/global cost aggregation."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import json
6
+ from collections import defaultdict
7
+ from datetime import UTC, datetime
8
+ from pathlib import Path
9
+ from typing import Literal
10
+
11
+ from pcli.config.paths import cost_ledger_file
12
+ from pcli.cost.pricing_table import PricingTable
13
+ from pcli.llm.models import Usage
14
+ from pcli.session.models import Session, TurnCost
15
+
16
+
17
+ class CostTracker:
18
+ def __init__(
19
+ self,
20
+ session: Session,
21
+ *,
22
+ pricing_table: PricingTable | None = None,
23
+ ledger_path: Path | None = None,
24
+ ) -> None:
25
+ self._session = session
26
+ self._pricing = pricing_table or PricingTable.load()
27
+ self._ledger_path = ledger_path or cost_ledger_file()
28
+
29
+ def record_turn(
30
+ self,
31
+ model: str,
32
+ usage: Usage,
33
+ *,
34
+ source: Literal["main", "subagent", "compaction", "memory"] = "main",
35
+ ) -> TurnCost:
36
+ cost_usd = self._pricing.cost_usd(
37
+ model,
38
+ prompt_tokens=usage.prompt_tokens,
39
+ completion_tokens=usage.completion_tokens,
40
+ cached_tokens=usage.cached_tokens or 0,
41
+ )
42
+ turn = TurnCost(
43
+ turn_index=len(self._session.cost.turns),
44
+ model=model,
45
+ usage=usage,
46
+ cost_usd=cost_usd,
47
+ estimated=usage.estimated,
48
+ source=source,
49
+ )
50
+ self._session.cost.turns.append(turn)
51
+ self._session.cost.session_total_usd += cost_usd
52
+ self._session.cost.total_tokens += usage.total_tokens
53
+ self._append_to_global_ledger(turn)
54
+ return turn
55
+
56
+ def _append_to_global_ledger(self, turn: TurnCost) -> None:
57
+ record = {
58
+ "session_id": self._session.id,
59
+ "turn_index": turn.turn_index,
60
+ "model": turn.model,
61
+ "prompt_tokens": turn.usage.prompt_tokens,
62
+ "completion_tokens": turn.usage.completion_tokens,
63
+ "cached_tokens": turn.usage.cached_tokens or 0,
64
+ "total_tokens": turn.usage.total_tokens,
65
+ "cost_usd": turn.cost_usd,
66
+ "estimated": turn.estimated,
67
+ "created_at": turn.created_at.isoformat(),
68
+ }
69
+ path = self._ledger_path
70
+ path.parent.mkdir(parents=True, exist_ok=True)
71
+ with path.open("a", encoding="utf-8") as fh:
72
+ fh.write(json.dumps(record) + "\n")
73
+
74
+
75
+ def global_cost_report(since: datetime | None = None, *, ledger_path: Path | None = None) -> dict:
76
+ path = ledger_path or cost_ledger_file()
77
+ total_cost = 0.0
78
+ total_tokens = 0
79
+ turn_count = 0
80
+ by_model: dict[str, float] = defaultdict(float)
81
+
82
+ if path.exists():
83
+ with path.open("r", encoding="utf-8") as fh:
84
+ for line in fh:
85
+ line = line.strip()
86
+ if not line:
87
+ continue
88
+ record = json.loads(line)
89
+ created_at = datetime.fromisoformat(record["created_at"])
90
+ if created_at.tzinfo is None:
91
+ created_at = created_at.replace(tzinfo=UTC)
92
+ if since is not None and created_at < since:
93
+ continue
94
+ total_cost += record["cost_usd"]
95
+ total_tokens += record["total_tokens"]
96
+ turn_count += 1
97
+ by_model[record["model"]] += record["cost_usd"]
98
+
99
+ return {
100
+ "total_cost_usd": total_cost,
101
+ "total_tokens": total_tokens,
102
+ "turn_count": turn_count,
103
+ "by_model": dict(by_model),
104
+ }
105
+
106
+
107
+ def cost_budget_reason(session: Session, max_session_cost_usd: float | None) -> str | None:
108
+ """None if there's no cap (max_session_cost_usd unset) or spend is still
109
+ under it; otherwise a formatted reason string for AgentLoop.run_turn's
110
+ budget_check to surface as a termination note - the one place that owns
111
+ this wording, so every call site (main loop, headless/scheduled runs,
112
+ subagents, memory extraction - see their own run_turn call sites) stays
113
+ consistent. session.cost.session_total_usd already includes subagent/
114
+ compaction/memory-extraction spend folded in (CostTracker.record_turn),
115
+ so this one check covers all of it regardless of which loop is asking."""
116
+ if max_session_cost_usd is None:
117
+ return None
118
+ spent = session.cost.session_total_usd
119
+ if spent < max_session_cost_usd:
120
+ return None
121
+ return (
122
+ f"Reached the session cost budget (${max_session_cost_usd:.2f}) - spent "
123
+ f"${spent:.4f} so far. Raise max_session_cost_usd (in the TUI: /budget <amount>; "
124
+ "otherwise PCLI_MAX_SESSION_COST_USD or config.toml) to continue this session, or "
125
+ "start a new one."
126
+ )
pcli/llm/__init__.py ADDED
File without changes
pcli/llm/client.py ADDED
@@ -0,0 +1,285 @@
1
+ """Async client for an OpenAI-compatible chat-completions gateway."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import asyncio
6
+ import random
7
+ from collections.abc import AsyncIterator
8
+ from typing import Any, Self
9
+
10
+ import httpx
11
+ from tenacity import retry, retry_if_exception_type, stop_after_attempt, wait_exponential_jitter
12
+
13
+ from pcli.config.settings import Settings
14
+ from pcli.cost.context_detect import detect_context_limit as _detect_context_limit
15
+ from pcli.llm.errors import GatewayError
16
+ from pcli.llm.models import ChatMessage, StreamEvent, ToolDefinition, Usage
17
+ from pcli.llm.streaming import parse_sse_stream
18
+
19
+
20
+ def _backoff_seconds(attempt: int) -> float:
21
+ return min(20.0, float(2**attempt)) + random.uniform(0, 1)
22
+
23
+
24
+ class GatewayClient:
25
+ def __init__(self, settings: Settings) -> None:
26
+ if not settings.is_configured():
27
+ raise GatewayError("Gateway URL and API key must be configured before use.")
28
+ self._settings = settings
29
+ headers = {"Content-Type": "application/json"}
30
+ if settings.gateway_api_key:
31
+ if settings.gateway_auth_header.lower() == "authorization":
32
+ headers["Authorization"] = f"Bearer {settings.gateway_api_key}"
33
+ else:
34
+ headers[settings.gateway_auth_header] = settings.gateway_api_key
35
+ self._client = httpx.AsyncClient(
36
+ base_url=settings.gateway_base_url.rstrip("/"),
37
+ headers=headers,
38
+ timeout=settings.effective_request_timeout_s,
39
+ )
40
+ # The constructor-level timeout above is only a fallback baseline —
41
+ # every actual request passes timeout=self._effective_timeout()
42
+ # explicitly (httpx supports a per-request override), so a setting
43
+ # change (e.g. via the /timeout command) takes effect on the very
44
+ # next request instead of requiring GatewayClient to be rebuilt.
45
+
46
+ def _effective_timeout(self) -> float:
47
+ return self._settings.effective_request_timeout_s
48
+
49
+ async def aclose(self) -> None:
50
+ await self._client.aclose()
51
+
52
+ async def __aenter__(self) -> Self:
53
+ return self
54
+
55
+ async def __aexit__(self, *exc_info: object) -> None:
56
+ await self.aclose()
57
+
58
+ def _network_error_hint(self, exc: httpx.HTTPError) -> str | None:
59
+ """Turns a raw httpx transport failure into something the user can
60
+ actually act on: which setting to change, and to what — grounded in
61
+ a real debugged case where a local model streaming at ~5 tokens/sec
62
+ tripped the 120s default mid-response, and pcli's only trace of it
63
+ was an opaque "Gateway error" with the underlying exception text."""
64
+ timeout = self._settings.effective_request_timeout_s
65
+ suggestion = max(600, int(timeout * 3))
66
+ gateway_kind = "local API gateway" if self._settings.is_local_api() else "gateway"
67
+
68
+ if isinstance(exc, httpx.ConnectTimeout):
69
+ return (
70
+ f"Couldn't connect to the {gateway_kind} within {timeout:g}s "
71
+ f"(request_timeout_s={timeout:g}). Check it's actually running at "
72
+ f"{self._settings.gateway_base_url!r} — or, if it's just slow to accept "
73
+ f"connections (e.g. still loading a model), set a larger request_timeout_s "
74
+ f"(e.g. {suggestion}) via PCLI_REQUEST_TIMEOUT_S or config.toml."
75
+ )
76
+ if isinstance(exc, httpx.TimeoutException):
77
+ local_note = (
78
+ "Local models are often much slower than hosted ones — " if self._settings.is_local_api() else ""
79
+ )
80
+ return (
81
+ f"Read timed out on the {gateway_kind} (request_timeout_s={timeout:g}). "
82
+ f"{local_note}Set a larger request_timeout_s (e.g. {suggestion}) via "
83
+ "PCLI_REQUEST_TIMEOUT_S or config.toml."
84
+ )
85
+ if isinstance(exc, httpx.ConnectError):
86
+ return (
87
+ f"Couldn't reach the {gateway_kind} at {self._settings.gateway_base_url!r} — "
88
+ "check that it's running and that gateway_base_url is correct."
89
+ )
90
+ return None
91
+
92
+ def _http_status_hint(self, status_code: int, body: str = "") -> str | None:
93
+ # Checked before the status-code-specific branches below: OpenAI-
94
+ # compatible gateways (including llama.cpp/LM Studio) report a
95
+ # context-length overflow as a 400 with wording that varies by
96
+ # backend, so this is a text match on the body rather than a status
97
+ # code alone. Worth surfacing as its own hint rather than a raw
98
+ # gateway error string, since the cause isn't obvious from a bare
99
+ # "invalid_request_error" — this is also the one context-length
100
+ # failure mode pcli's own turn-based auto-compaction can't catch: a
101
+ # subagent's own nested tool-calling loop has no compaction of its
102
+ # own (see tools/_nested_agent.py), so a long subagent task can hit
103
+ # this mid-turn with no prior warning.
104
+ body_lower = body.lower()
105
+ if status_code == 400 and (
106
+ "context_length_exceeded" in body_lower
107
+ or "context length" in body_lower
108
+ or "context window" in body_lower
109
+ or ("maximum" in body_lower and "token" in body_lower)
110
+ ):
111
+ return (
112
+ "This looks like a context-length overflow — the request (conversation history "
113
+ "plus tool results) is larger than the model can accept in one call. If this is "
114
+ "the main conversation, /context-limit sets the context window pcli assumes for "
115
+ "auto-compaction; if it's a subagent's own task, its conversation has no "
116
+ "compaction of its own, so try splitting the task into smaller, narrower steps."
117
+ )
118
+ if status_code in (401, 403):
119
+ if self._settings.gateway_api_key:
120
+ return "The gateway rejected the configured gateway_api_key — check it's correct and hasn't expired."
121
+ return (
122
+ "No gateway_api_key is configured and this gateway appears to require one — "
123
+ "set PCLI_GATEWAY_API_KEY or gateway_api_key in config.toml."
124
+ )
125
+ if status_code == 429:
126
+ return (
127
+ f"Rate-limited by the gateway. pcli already retries with backoff (up to "
128
+ f"max_retries={self._settings.max_retries}) — if this keeps happening, raise "
129
+ "max_retries or reduce request frequency."
130
+ )
131
+ if status_code >= 500:
132
+ return "This is usually transient on the gateway's side — trying again shortly often helps."
133
+ return None
134
+
135
+ def _build_payload(
136
+ self,
137
+ messages: list[ChatMessage],
138
+ *,
139
+ model: str | None,
140
+ tools: list[ToolDefinition] | None,
141
+ stream: bool,
142
+ temperature: float | None,
143
+ max_tokens: int | None = None,
144
+ ) -> dict[str, Any]:
145
+ payload: dict[str, Any] = {
146
+ "model": model or self._settings.default_model,
147
+ "messages": [m.to_wire() for m in messages],
148
+ "stream": stream,
149
+ }
150
+ if tools:
151
+ payload["tools"] = [t.model_dump() for t in tools]
152
+ if temperature is not None:
153
+ payload["temperature"] = temperature
154
+ if max_tokens is not None:
155
+ payload["max_tokens"] = max_tokens
156
+ if stream:
157
+ payload["stream_options"] = {"include_usage": True}
158
+ return payload
159
+
160
+ async def chat_stream(
161
+ self,
162
+ messages: list[ChatMessage],
163
+ *,
164
+ model: str | None = None,
165
+ tools: list[ToolDefinition] | None = None,
166
+ temperature: float | None = None,
167
+ max_tokens: int | None = None,
168
+ ) -> AsyncIterator[StreamEvent]:
169
+ """Streams chat events. Retries (with backoff) only apply to failures that
170
+ happen before any event has been yielded — once partial output has reached
171
+ the caller, retrying the whole request would duplicate it, so a mid-stream
172
+ failure is raised immediately instead.
173
+
174
+ `max_tokens`, when given, is the dynamic per-request cap computed by
175
+ cost/context.py's compute_max_response_tokens — see AgentLoop, which
176
+ is what actually sets it.
177
+ """
178
+ payload = self._build_payload(
179
+ messages, model=model, tools=tools, stream=True, temperature=temperature,
180
+ max_tokens=max_tokens,
181
+ )
182
+ max_attempts = max(1, self._settings.max_retries)
183
+ last_error: GatewayError | None = None
184
+
185
+ for attempt in range(1, max_attempts + 1):
186
+ started = False
187
+ try:
188
+ async with self._client.stream(
189
+ "POST", "/chat/completions", json=payload, timeout=self._effective_timeout()
190
+ ) as response:
191
+ if response.status_code >= 400:
192
+ body = await response.aread()
193
+ body_text = body.decode(errors="replace")
194
+ raise GatewayError.from_http_status(
195
+ response.status_code,
196
+ body_text,
197
+ hint=self._http_status_hint(response.status_code, body_text),
198
+ )
199
+ async for event in parse_sse_stream(response.aiter_lines()):
200
+ started = True
201
+ yield event
202
+ return
203
+ except GatewayError as exc:
204
+ last_error = exc
205
+ if not exc.retryable or started or attempt == max_attempts:
206
+ raise
207
+ except httpx.HTTPError as exc:
208
+ last_error = GatewayError.from_network_error(str(exc), hint=self._network_error_hint(exc))
209
+ if started or attempt == max_attempts:
210
+ raise last_error
211
+
212
+ await asyncio.sleep(_backoff_seconds(attempt))
213
+
214
+ if last_error:
215
+ raise last_error
216
+
217
+ async def collect(
218
+ self,
219
+ messages: list[ChatMessage],
220
+ *,
221
+ model: str | None = None,
222
+ tools: list[ToolDefinition] | None = None,
223
+ temperature: float | None = None,
224
+ max_tokens: int | None = None,
225
+ ) -> tuple[ChatMessage, Usage]:
226
+ """Consumes chat_stream and returns the final assembled message + usage."""
227
+ from pcli.llm.models import ToolCall, ToolCallCompleteEvent, UsageEvent
228
+
229
+ text_parts: list[str] = []
230
+ tool_calls: list[ToolCall] = []
231
+ usage = Usage()
232
+
233
+ async for event in self.chat_stream(
234
+ messages, model=model, tools=tools, temperature=temperature, max_tokens=max_tokens
235
+ ):
236
+ if event.kind == "text_delta":
237
+ text_parts.append(event.text)
238
+ elif isinstance(event, ToolCallCompleteEvent):
239
+ tool_calls = event.tool_calls
240
+ elif isinstance(event, UsageEvent):
241
+ usage = event.usage
242
+
243
+ message = ChatMessage(
244
+ role="assistant",
245
+ content="".join(text_parts) or None,
246
+ tool_calls=tool_calls or None,
247
+ )
248
+ return message, usage
249
+
250
+ @retry(
251
+ stop=stop_after_attempt(3),
252
+ wait=wait_exponential_jitter(initial=1, max=10),
253
+ retry=retry_if_exception_type(httpx.HTTPError),
254
+ reraise=True,
255
+ )
256
+ async def health_check(self) -> bool:
257
+ """Best-effort reachability probe against the gateway's models endpoint."""
258
+ response = await self._client.get("/models", timeout=self._effective_timeout())
259
+ return response.status_code < 500
260
+
261
+ async def list_models(self) -> list[str]:
262
+ """Fetches available model IDs from the gateway's OpenAI-compatible
263
+ `GET /models` endpoint (`{"data": [{"id": "..."}, ...]}`)."""
264
+ try:
265
+ response = await self._client.get("/models", timeout=self._effective_timeout())
266
+ except httpx.HTTPError as exc:
267
+ raise GatewayError.from_network_error(str(exc), hint=self._network_error_hint(exc)) from exc
268
+ if response.status_code >= 400:
269
+ raise GatewayError.from_http_status(
270
+ response.status_code,
271
+ response.text,
272
+ hint=self._http_status_hint(response.status_code, response.text),
273
+ )
274
+ payload = response.json()
275
+ entries = payload.get("data", []) if isinstance(payload, dict) else []
276
+ ids = [entry["id"] for entry in entries if isinstance(entry, dict) and "id" in entry]
277
+ return sorted(ids)
278
+
279
+ async def detect_context_limit(self, model: str) -> int | None:
280
+ """Best-effort: queries this gateway directly for model's real
281
+ context window, trying several known backend-specific extensions
282
+ (see cost/context_detect.py — no single standard covers this).
283
+ None if nothing answered, including for hosted-only gateways that
284
+ expose no such information at all. Never raises."""
285
+ return await _detect_context_limit(self._client, model)
pcli/llm/errors.py ADDED
@@ -0,0 +1,37 @@
1
+ """Error hierarchy for the gateway client, with retry classification."""
2
+
3
+ from __future__ import annotations
4
+
5
+
6
+ class GatewayError(Exception):
7
+ """Raised for any failure talking to the LLM gateway."""
8
+
9
+ def __init__(
10
+ self,
11
+ message: str,
12
+ *,
13
+ status_code: int | None = None,
14
+ retryable: bool = False,
15
+ ) -> None:
16
+ super().__init__(message)
17
+ self.message = message
18
+ self.status_code = status_code
19
+ self.retryable = retryable
20
+
21
+ @classmethod
22
+ def from_http_status(cls, status_code: int, message: str, *, hint: str | None = None) -> GatewayError:
23
+ retryable = status_code == 429 or status_code >= 500
24
+ full_message = f"{message} {hint}" if hint else message
25
+ return cls(full_message, status_code=status_code, retryable=retryable)
26
+
27
+ @classmethod
28
+ def from_network_error(cls, message: str, *, hint: str | None = None) -> GatewayError:
29
+ # hint is folded directly into .message (not kept as a separate
30
+ # attribute) so every existing call site that already displays
31
+ # str(exc)/exc.message — chat.py's turn/compaction/models handlers,
32
+ # subagent_tool.py's "Subagent failed: {exc}", ... — picks up
33
+ # actionable config guidance automatically, with no per-site change
34
+ # needed. See GatewayClient._network_error_hint/_http_status_hint
35
+ # for what gets suggested and why.
36
+ full_message = f"{message} {hint}" if hint else message
37
+ return cls(full_message, status_code=None, retryable=True)
pcli/llm/models.py ADDED
@@ -0,0 +1,100 @@
1
+ """Pydantic models for the OpenAI-compatible chat-completions wire format."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Any, Literal
6
+
7
+ from pydantic import BaseModel
8
+
9
+ Role = Literal["system", "user", "assistant", "tool"]
10
+
11
+
12
+ class ToolCallFunction(BaseModel):
13
+ name: str
14
+ arguments: str = ""
15
+
16
+
17
+ class ToolCall(BaseModel):
18
+ id: str
19
+ type: Literal["function"] = "function"
20
+ function: ToolCallFunction
21
+
22
+
23
+ class ChatMessage(BaseModel):
24
+ role: Role
25
+ content: str | None = None
26
+ tool_calls: list[ToolCall] | None = None
27
+ tool_call_id: str | None = None
28
+ name: str | None = None
29
+
30
+ def to_wire(self) -> dict[str, Any]:
31
+ data: dict[str, Any] = {"role": self.role}
32
+ if self.content is not None:
33
+ data["content"] = self.content
34
+ if self.tool_calls:
35
+ data["tool_calls"] = [tc.model_dump() for tc in self.tool_calls]
36
+ if self.tool_call_id:
37
+ data["tool_call_id"] = self.tool_call_id
38
+ if self.name:
39
+ data["name"] = self.name
40
+ return data
41
+
42
+
43
+ class ToolDefinition(BaseModel):
44
+ """A single entry in the OpenAI-style `tools=[...]` request payload."""
45
+
46
+ type: Literal["function"] = "function"
47
+ function: dict[str, Any]
48
+
49
+
50
+ class Usage(BaseModel):
51
+ prompt_tokens: int = 0
52
+ completion_tokens: int = 0
53
+ total_tokens: int = 0
54
+ cached_tokens: int | None = None
55
+ estimated: bool = False
56
+
57
+
58
+ class TextDelta(BaseModel):
59
+ kind: Literal["text_delta"] = "text_delta"
60
+ text: str
61
+
62
+
63
+ class ReasoningDelta(BaseModel):
64
+ """A fragment of a reasoning/"thinking" model's chain-of-thought,
65
+ streamed under `delta.reasoning_content` (or `delta.reasoning`) by
66
+ OpenAI-compatible servers that expose it as a channel separate from the
67
+ final answer (`delta.content`) — e.g. vLLM/SGLang/LM Studio serving
68
+ DeepSeek-R1-style or Nemotron "thinking" models. Never folded into the
69
+ assistant message's actual content; see agent/loop.py."""
70
+
71
+ kind: Literal["reasoning_delta"] = "reasoning_delta"
72
+ text: str
73
+
74
+
75
+ class ToolCallDelta(BaseModel):
76
+ kind: Literal["tool_call_delta"] = "tool_call_delta"
77
+ index: int
78
+ id: str | None = None
79
+ name: str | None = None
80
+ arguments_fragment: str = ""
81
+
82
+
83
+ class ToolCallCompleteEvent(BaseModel):
84
+ kind: Literal["tool_call_complete"] = "tool_call_complete"
85
+ tool_calls: list[ToolCall]
86
+
87
+
88
+ class UsageEvent(BaseModel):
89
+ kind: Literal["usage"] = "usage"
90
+ usage: Usage
91
+
92
+
93
+ class FinishEvent(BaseModel):
94
+ kind: Literal["finish"] = "finish"
95
+ reason: str | None = None
96
+
97
+
98
+ StreamEvent = (
99
+ TextDelta | ReasoningDelta | ToolCallDelta | ToolCallCompleteEvent | UsageEvent | FinishEvent
100
+ )
pcli/llm/streaming.py ADDED
@@ -0,0 +1,108 @@
1
+ """Parses an OpenAI-compatible chat-completions SSE stream into StreamEvents.
2
+
3
+ Tool-call arguments arrive fragmented across multiple chunks, keyed by the
4
+ `index` field within `delta.tool_calls`; fragments must be accumulated until
5
+ `finish_reason == "tool_calls"` before the full arguments JSON is usable.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import json
11
+ from collections.abc import AsyncIterator
12
+
13
+ from pcli.llm.models import (
14
+ FinishEvent,
15
+ ReasoningDelta,
16
+ StreamEvent,
17
+ TextDelta,
18
+ ToolCall,
19
+ ToolCallCompleteEvent,
20
+ ToolCallDelta,
21
+ ToolCallFunction,
22
+ Usage,
23
+ UsageEvent,
24
+ )
25
+
26
+
27
+ class _ToolCallAccumulator:
28
+ def __init__(self) -> None:
29
+ self.id: str | None = None
30
+ self.name: str | None = None
31
+ self.arguments = ""
32
+
33
+ def to_tool_call(self) -> ToolCall:
34
+ return ToolCall(
35
+ id=self.id or "",
36
+ function=ToolCallFunction(name=self.name or "", arguments=self.arguments),
37
+ )
38
+
39
+
40
+ async def parse_sse_stream(lines: AsyncIterator[str]) -> AsyncIterator[StreamEvent]:
41
+ accumulators: dict[int, _ToolCallAccumulator] = {}
42
+
43
+ async for raw_line in lines:
44
+ line = raw_line.strip()
45
+ if not line or not line.startswith("data:"):
46
+ continue
47
+ payload = line[len("data:") :].strip()
48
+ if payload == "[DONE]":
49
+ break
50
+ try:
51
+ chunk = json.loads(payload)
52
+ except json.JSONDecodeError:
53
+ continue
54
+
55
+ usage = chunk.get("usage")
56
+ if usage:
57
+ yield UsageEvent(
58
+ usage=Usage(
59
+ prompt_tokens=usage.get("prompt_tokens", 0),
60
+ completion_tokens=usage.get("completion_tokens", 0),
61
+ total_tokens=usage.get("total_tokens", 0),
62
+ cached_tokens=(usage.get("prompt_tokens_details") or {}).get("cached_tokens"),
63
+ )
64
+ )
65
+
66
+ choices = chunk.get("choices") or []
67
+ if not choices:
68
+ continue
69
+ choice = choices[0]
70
+ delta = choice.get("delta") or {}
71
+
72
+ content = delta.get("content")
73
+ if content:
74
+ yield TextDelta(text=content)
75
+
76
+ # Reasoning/"thinking" models (DeepSeek-R1-style, Nemotron "detailed
77
+ # thinking", ...) served through vLLM/SGLang/LM Studio stream their
78
+ # chain-of-thought under a separate field, never under `content` —
79
+ # a model that spends its whole response reasoning without ever
80
+ # transitioning to `content` previously vanished here entirely: no
81
+ # text, no tool call, no error, just a silently empty turn. Servers
82
+ # aren't consistent about the key name, so both are checked.
83
+ reasoning = delta.get("reasoning_content") or delta.get("reasoning")
84
+ if reasoning:
85
+ yield ReasoningDelta(text=reasoning)
86
+
87
+ for tc in delta.get("tool_calls") or []:
88
+ index = tc.get("index", 0)
89
+ acc = accumulators.setdefault(index, _ToolCallAccumulator())
90
+ if tc.get("id"):
91
+ acc.id = tc["id"]
92
+ fn = tc.get("function") or {}
93
+ if fn.get("name"):
94
+ acc.name = fn["name"]
95
+ frag = fn.get("arguments") or ""
96
+ if frag:
97
+ acc.arguments += frag
98
+ yield ToolCallDelta(
99
+ index=index, id=tc.get("id"), name=fn.get("name"), arguments_fragment=frag
100
+ )
101
+
102
+ finish_reason = choice.get("finish_reason")
103
+ if finish_reason:
104
+ if finish_reason == "tool_calls" and accumulators:
105
+ completed = [accumulators[i].to_tool_call() for i in sorted(accumulators)]
106
+ yield ToolCallCompleteEvent(tool_calls=completed)
107
+ accumulators.clear()
108
+ yield FinishEvent(reason=finish_reason)
File without changes