paimon 0.2.3__tar.gz → 0.2.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {paimon-0.2.3/paimon.egg-info → paimon-0.2.4}/PKG-INFO +6 -2
- {paimon-0.2.3 → paimon-0.2.4}/README.md +4 -0
- {paimon-0.2.3 → paimon-0.2.4}/README.zh-CN.md +4 -0
- {paimon-0.2.3 → paimon-0.2.4}/paimon/agent.py +262 -55
- {paimon-0.2.3 → paimon-0.2.4}/paimon/cli.py +10 -0
- {paimon-0.2.3 → paimon-0.2.4}/paimon/commands.py +58 -102
- {paimon-0.2.3 → paimon-0.2.4}/paimon/compaction.py +40 -3
- {paimon-0.2.3 → paimon-0.2.4}/paimon/config.py +95 -11
- {paimon-0.2.3 → paimon-0.2.4}/paimon/headless.py +36 -13
- {paimon-0.2.3 → paimon-0.2.4}/paimon/jobs.py +2 -3
- {paimon-0.2.3 → paimon-0.2.4}/paimon/llm.py +16 -1
- {paimon-0.2.3 → paimon-0.2.4}/paimon/login.py +9 -2
- {paimon-0.2.3 → paimon-0.2.4}/paimon/pane.py +1 -1
- {paimon-0.2.3 → paimon-0.2.4}/paimon/prompt.py +15 -5
- {paimon-0.2.3 → paimon-0.2.4}/paimon/retry.py +27 -0
- {paimon-0.2.3 → paimon-0.2.4}/paimon/session.py +152 -12
- {paimon-0.2.3 → paimon-0.2.4}/paimon/skill/SKILL.md +5 -2
- paimon-0.2.4/paimon/telemetry.py +160 -0
- {paimon-0.2.3 → paimon-0.2.4}/paimon/tools.py +554 -15
- {paimon-0.2.3 → paimon-0.2.4}/paimon/ui.py +33 -7
- {paimon-0.2.3 → paimon-0.2.4/paimon.egg-info}/PKG-INFO +6 -2
- {paimon-0.2.3 → paimon-0.2.4}/paimon.egg-info/SOURCES.txt +1 -0
- paimon-0.2.4/paimon.egg-info/requires.txt +3 -0
- {paimon-0.2.3 → paimon-0.2.4}/paimon.egg-info/scm_file_list.json +3 -0
- paimon-0.2.4/paimon.egg-info/scm_version.json +8 -0
- {paimon-0.2.3 → paimon-0.2.4}/pyproject.toml +1 -1
- paimon-0.2.3/paimon.egg-info/requires.txt +0 -3
- paimon-0.2.3/paimon.egg-info/scm_version.json +0 -8
- {paimon-0.2.3 → paimon-0.2.4}/LICENSE +0 -0
- {paimon-0.2.3 → paimon-0.2.4}/MANIFEST.in +0 -0
- {paimon-0.2.3 → paimon-0.2.4}/paimon/__init__.py +0 -0
- {paimon-0.2.3 → paimon-0.2.4}/paimon/__main__.py +0 -0
- {paimon-0.2.3 → paimon-0.2.4}/paimon/app.py +0 -0
- {paimon-0.2.3 → paimon-0.2.4}/paimon/app.tcss +0 -0
- {paimon-0.2.3 → paimon-0.2.4}/paimon/aside.py +0 -0
- {paimon-0.2.3 → paimon-0.2.4}/paimon/diff.py +0 -0
- {paimon-0.2.3 → paimon-0.2.4}/paimon/errors.py +0 -0
- {paimon-0.2.3 → paimon-0.2.4}/paimon/lockfile.py +0 -0
- {paimon-0.2.3 → paimon-0.2.4}/paimon/mentions.py +0 -0
- {paimon-0.2.3 → paimon-0.2.4}/paimon/model_windows.py +0 -0
- {paimon-0.2.3 → paimon-0.2.4}/paimon/supervisor.py +0 -0
- {paimon-0.2.3 → paimon-0.2.4}/paimon/tabs.py +0 -0
- {paimon-0.2.3 → paimon-0.2.4}/paimon/taskpane.py +0 -0
- {paimon-0.2.3 → paimon-0.2.4}/paimon.egg-info/dependency_links.txt +0 -0
- {paimon-0.2.3 → paimon-0.2.4}/paimon.egg-info/entry_points.txt +0 -0
- {paimon-0.2.3 → paimon-0.2.4}/paimon.egg-info/top_level.txt +0 -0
- {paimon-0.2.3 → paimon-0.2.4}/setup.cfg +0 -0
|
@@ -1,11 +1,11 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: paimon
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.4
|
|
4
4
|
Summary: A minimal code agent built on pydantic-ai + textual
|
|
5
5
|
Requires-Python: >=3.10
|
|
6
6
|
Description-Content-Type: text/markdown
|
|
7
7
|
License-File: LICENSE
|
|
8
|
-
Requires-Dist: pydantic-ai-slim[anthropic,openai]>=2.
|
|
8
|
+
Requires-Dist: pydantic-ai-slim[anthropic,openai]>=2.31.0
|
|
9
9
|
Requires-Dist: textual>=8.2.7
|
|
10
10
|
Requires-Dist: textual-serve>=1.1.3
|
|
11
11
|
Dynamic: license-file
|
|
@@ -110,3 +110,7 @@ paimon --profile work # a separately configured account
|
|
|
110
110
|
Each profile keeps its model settings in `~/.config/paimon/<name>/config.json`, written by the first launch or by `paimon login`. Sessions live in `~/.local/share/paimon/sessions/`. File changes render nicer if [delta](https://github.com/dandavison/delta) is installed.
|
|
111
111
|
|
|
112
112
|
Read and edit modes run a small set of clearly read-only commands (`ls`, `cat`, `git status`, …) without asking; `--strict` turns that off. **This is a guardrail against agent mistakes, not a security boundary.** For real isolation, run Paimon inside a container or VM.
|
|
113
|
+
|
|
114
|
+
## Telemetry
|
|
115
|
+
|
|
116
|
+
Each launch sends one anonymous event to Google Analytics: a random install id, the launch mode, the version, the OS name and the configured provider and model name. Nothing from your sessions, prompts, files or credentials is included. Set `PAIMON_NO_TELEMETRY=1` or `DO_NOT_TRACK=1` to turn it off.
|
|
@@ -98,3 +98,7 @@ paimon --profile work # a separately configured account
|
|
|
98
98
|
Each profile keeps its model settings in `~/.config/paimon/<name>/config.json`, written by the first launch or by `paimon login`. Sessions live in `~/.local/share/paimon/sessions/`. File changes render nicer if [delta](https://github.com/dandavison/delta) is installed.
|
|
99
99
|
|
|
100
100
|
Read and edit modes run a small set of clearly read-only commands (`ls`, `cat`, `git status`, …) without asking; `--strict` turns that off. **This is a guardrail against agent mistakes, not a security boundary.** For real isolation, run Paimon inside a container or VM.
|
|
101
|
+
|
|
102
|
+
## Telemetry
|
|
103
|
+
|
|
104
|
+
Each launch sends one anonymous event to Google Analytics: a random install id, the launch mode, the version, the OS name and the configured provider and model name. Nothing from your sessions, prompts, files or credentials is included. Set `PAIMON_NO_TELEMETRY=1` or `DO_NOT_TRACK=1` to turn it off.
|
|
@@ -98,3 +98,7 @@ paimon --profile work # 单独配置的另一个账号
|
|
|
98
98
|
每个 profile 的模型设置保存在 `~/.config/paimon/<name>/config.json`,由首次启动或 `paimon login` 写入。会话存放在 `~/.local/share/paimon/sessions/`。安装 [delta](https://github.com/dandavison/delta) 后文件改动的展示效果更好。
|
|
99
99
|
|
|
100
100
|
read 和 edit 模式会不经询问执行一小组明确只读的命令(`ls`、`cat`、`git status` 等),`--strict` 可以关掉。**这是防止 agent 失误的护栏,不是安全边界。** 需要真正的隔离时,请在容器或虚拟机中运行 Paimon。
|
|
101
|
+
|
|
102
|
+
## 遥测
|
|
103
|
+
|
|
104
|
+
每次启动会向 Google Analytics 发送一条匿名事件:随机生成的安装 id、启动方式、版本号、操作系统名称,以及配置的 provider 和模型名。会话、提示词、文件和密钥一概不上报。设置 `PAIMON_NO_TELEMETRY=1` 或 `DO_NOT_TRACK=1` 可以关闭。
|
|
@@ -107,6 +107,15 @@ class RequestStats:
|
|
|
107
107
|
cache_write_tokens: int = 0
|
|
108
108
|
|
|
109
109
|
|
|
110
|
+
@dataclass
|
|
111
|
+
class ToolBudgetExhausted:
|
|
112
|
+
"""The turn hit its tool-call budget: the remaining calls were refused
|
|
113
|
+
with an explicit result and the turn ends without another model request.
|
|
114
|
+
"""
|
|
115
|
+
|
|
116
|
+
limit: int
|
|
117
|
+
|
|
118
|
+
|
|
110
119
|
@dataclass
|
|
111
120
|
class TurnEnd:
|
|
112
121
|
pass
|
|
@@ -161,9 +170,9 @@ class AgentsNotice:
|
|
|
161
170
|
# on isinstance; the alias exists so a type checker can flag an unhandled one.
|
|
162
171
|
AgentEvent = (
|
|
163
172
|
TextDelta | ReasoningDelta | ToolStart | ToolEnd | TodosUpdate
|
|
164
|
-
| SessionHandoff | RequestStats |
|
|
165
|
-
| ContextCompactionFailed | ModelRetry | UserInput
|
|
166
|
-
| AgentsNotice
|
|
173
|
+
| SessionHandoff | RequestStats | ToolBudgetExhausted | TurnEnd
|
|
174
|
+
| ContextCompacted | ContextCompactionFailed | ModelRetry | UserInput
|
|
175
|
+
| CompactionNotice | AgentsNotice
|
|
167
176
|
)
|
|
168
177
|
|
|
169
178
|
|
|
@@ -184,12 +193,14 @@ def replay_events(messages: list[ModelMessage]) -> list[AgentEvent]:
|
|
|
184
193
|
"""Persisted messages replayed as the events a live ``Agent.run`` yields.
|
|
185
194
|
|
|
186
195
|
Lets a UI render resumed history through the same code path as live turns.
|
|
196
|
+
Tools run serially, so each call replays as its ToolStart immediately
|
|
197
|
+
followed by its ToolEnd (looked up in the next request), matching the live
|
|
198
|
+
order — not all starts of a batch first. The persisted ``outcome`` field
|
|
199
|
+
restores denied styling.
|
|
187
200
|
"""
|
|
188
201
|
events: list[AgentEvent] = []
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
rejected_todos: set[str] = set()
|
|
192
|
-
for message in messages:
|
|
202
|
+
consumed: set[str] = set() # tool returns already replayed with their call
|
|
203
|
+
for index, message in enumerate(messages):
|
|
193
204
|
if is_summary_message(message):
|
|
194
205
|
events.append(CompactionNotice())
|
|
195
206
|
continue
|
|
@@ -200,11 +211,20 @@ def replay_events(messages: list[ModelMessage]) -> list[AgentEvent]:
|
|
|
200
211
|
for part in message.parts:
|
|
201
212
|
if isinstance(part, UserPromptPart) and isinstance(part.content, str) and part.content:
|
|
202
213
|
events.append(UserInput(part.content))
|
|
203
|
-
elif isinstance(part, ToolReturnPart)
|
|
204
|
-
|
|
214
|
+
elif (isinstance(part, ToolReturnPart)
|
|
215
|
+
and part.tool_call_id not in consumed
|
|
216
|
+
and part.tool_name != "write_todos"):
|
|
217
|
+
# A return whose call is gone (compaction cut mid-turn):
|
|
218
|
+
# better an unpaired result than a silent hole.
|
|
205
219
|
events.append(ToolEnd(part.tool_call_id, part.tool_name,
|
|
206
|
-
str(part.content or "(no output)")
|
|
220
|
+
str(part.content or "(no output)"),
|
|
221
|
+
denied=part.outcome == "denied"))
|
|
207
222
|
elif isinstance(message, ModelResponse):
|
|
223
|
+
following = messages[index + 1] if index + 1 < len(messages) else None
|
|
224
|
+
returns: dict[str, ToolReturnPart] = {}
|
|
225
|
+
if isinstance(following, ModelRequest):
|
|
226
|
+
returns = {part.tool_call_id: part for part in following.parts
|
|
227
|
+
if isinstance(part, ToolReturnPart)}
|
|
208
228
|
for part in message.parts:
|
|
209
229
|
if isinstance(part, ThinkingPart) and part.content:
|
|
210
230
|
events.append(ReasoningDelta(part.content))
|
|
@@ -212,23 +232,37 @@ def replay_events(messages: list[ModelMessage]) -> list[AgentEvent]:
|
|
|
212
232
|
events.append(TextDelta(part.content))
|
|
213
233
|
elif isinstance(part, ToolCallPart):
|
|
214
234
|
args = _parse_args(part.args)
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
|
|
222
|
-
|
|
235
|
+
result = returns.get(part.tool_call_id)
|
|
236
|
+
if result is not None:
|
|
237
|
+
consumed.add(part.tool_call_id)
|
|
238
|
+
if part.tool_name == "write_todos":
|
|
239
|
+
todos = tools.normalize_todos(args.get("todos"))
|
|
240
|
+
# Live, a write_todos that actually ran yields only a
|
|
241
|
+
# TodosUpdate; one that was rejected or never executed
|
|
242
|
+
# yields a normal failed call.
|
|
243
|
+
if todos is not None and (result is None or result.outcome == "success"):
|
|
244
|
+
events.append(TodosUpdate(todos))
|
|
245
|
+
continue
|
|
246
|
+
events.append(ToolStart(part.tool_call_id, part.tool_name, args))
|
|
247
|
+
if result is not None:
|
|
248
|
+
events.append(ToolEnd(part.tool_call_id, part.tool_name,
|
|
249
|
+
str(result.content or "(no output)"),
|
|
250
|
+
denied=result.outcome == "denied"))
|
|
223
251
|
return events
|
|
224
252
|
|
|
225
253
|
|
|
226
254
|
def _strip_foreign_thinking(history: list[ModelMessage], model: Model) -> list[ModelMessage]:
|
|
227
|
-
"""
|
|
255
|
+
"""Adapt thinking parts produced by a different provider/model.
|
|
228
256
|
|
|
229
257
|
Preserved thinking is only valid replayed verbatim to the model that
|
|
230
|
-
produced it
|
|
231
|
-
|
|
258
|
+
produced it, so a foreign message's readable thinking is converted to a
|
|
259
|
+
plain text block — the rationale it carries is still context — while its
|
|
260
|
+
signatures and encrypted/redacted payloads are dropped, never sent to a
|
|
261
|
+
provider that did not produce them (cf. pi's transformMessages, which
|
|
262
|
+
converts rather than deletes readable thinking).
|
|
263
|
+
|
|
264
|
+
Only the request being built is touched; the persisted history keeps the
|
|
265
|
+
original parts.
|
|
232
266
|
"""
|
|
233
267
|
current = (model.system, model.model_name)
|
|
234
268
|
sanitized: list[ModelMessage] = []
|
|
@@ -236,9 +270,16 @@ def _strip_foreign_thinking(history: list[ModelMessage], model: Model) -> list[M
|
|
|
236
270
|
if (isinstance(message, ModelResponse)
|
|
237
271
|
and (message.provider_name, message.model_name) != current
|
|
238
272
|
and any(isinstance(part, ThinkingPart) for part in message.parts)):
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
273
|
+
parts = []
|
|
274
|
+
for part in message.parts:
|
|
275
|
+
if not isinstance(part, ThinkingPart):
|
|
276
|
+
parts.append(part)
|
|
277
|
+
elif part.content:
|
|
278
|
+
parts.append(TextPart(content=f"<thinking>\n{part.content}\n</thinking>"))
|
|
279
|
+
# else: encrypted or redacted thinking with nothing readable
|
|
280
|
+
if not parts:
|
|
281
|
+
continue # nothing a different provider can use
|
|
282
|
+
message = dataclasses.replace(message, parts=parts)
|
|
242
283
|
sanitized.append(message)
|
|
243
284
|
return sanitized
|
|
244
285
|
|
|
@@ -289,8 +330,9 @@ class Agent:
|
|
|
289
330
|
# None where nobody can type while a turn runs (headless, tests).
|
|
290
331
|
self.pending: Optional[PendingFn] = None
|
|
291
332
|
# Per-agent tool state, kept off the tool functions so one agent's
|
|
292
|
-
# shell overflow files stay invisible to the next one.
|
|
293
|
-
|
|
333
|
+
# shell overflow files stay invisible to the next one. The session
|
|
334
|
+
# rides along for the history tools.
|
|
335
|
+
self.tool_context = tools.ToolContext(session=session)
|
|
294
336
|
self.todos: list[dict] = []
|
|
295
337
|
self.session = session
|
|
296
338
|
self.system_prompt = system_prompt
|
|
@@ -300,6 +342,12 @@ class Agent:
|
|
|
300
342
|
self.tool_schemas = tools.schemas(self.toolset)
|
|
301
343
|
self._tool_definitions = tools.definitions(self.toolset)
|
|
302
344
|
self._cached_model: Optional[tuple[tuple, Model]] = None
|
|
345
|
+
# (history length, provider-reported tokens) after the last completed
|
|
346
|
+
# request: the authoritative context size at that point, used by
|
|
347
|
+
# count_context_tokens so the chars/4 heuristic only covers what was
|
|
348
|
+
# appended since. None until a request reports usage, and again after
|
|
349
|
+
# compaction reshapes the history.
|
|
350
|
+
self._usage_anchor: Optional[tuple[int, int]] = None
|
|
303
351
|
|
|
304
352
|
@classmethod
|
|
305
353
|
def open(cls, cwd: Optional[Path] = None, *, session: Optional[Session] = None,
|
|
@@ -313,7 +361,12 @@ class Agent:
|
|
|
313
361
|
|
|
314
362
|
``append_system_prompt`` is added to the end of a new session's system
|
|
315
363
|
prompt and persisted with it, so a resumed session keeps it. Resuming
|
|
316
|
-
with it set raises ``ValueError``: the persisted
|
|
364
|
+
with it set raises ``ValueError``: the persisted role is immutable.
|
|
365
|
+
The dynamic parts of the prompt (date, environment, AGENTS.md) are
|
|
366
|
+
rebuilt on resume — a session picked up months later must not keep
|
|
367
|
+
telling the model the old date or stale project rules — and when they
|
|
368
|
+
changed, the rebuilt snapshot is appended to the log, so it always
|
|
369
|
+
records the prompt each turn actually ran with.
|
|
317
370
|
``parent`` marks the new session as a subagent's, which keeps it out of
|
|
318
371
|
the session listings its parent shows up in.
|
|
319
372
|
|
|
@@ -334,14 +387,21 @@ class Agent:
|
|
|
334
387
|
# Agent is returned to unlock it, so the lock is released here.
|
|
335
388
|
try:
|
|
336
389
|
if is_new:
|
|
390
|
+
appended = append_system_prompt.strip() if append_system_prompt else None
|
|
337
391
|
system_prompt = build_system_prompt(cwd)
|
|
338
|
-
if
|
|
339
|
-
system_prompt += f"\n\n{
|
|
340
|
-
session.append_system_prompt(system_prompt)
|
|
392
|
+
if appended:
|
|
393
|
+
system_prompt += f"\n\n{appended}"
|
|
394
|
+
session.append_system_prompt(system_prompt, appended=appended)
|
|
341
395
|
else:
|
|
342
|
-
|
|
343
|
-
|
|
396
|
+
session.require_supported_format()
|
|
397
|
+
stored, appended = session.system_prompt_parts()
|
|
398
|
+
if stored is None:
|
|
344
399
|
raise SessionIncompleteError("Session does not contain a persisted system prompt")
|
|
400
|
+
system_prompt = build_system_prompt(cwd)
|
|
401
|
+
if appended:
|
|
402
|
+
system_prompt += f"\n\n{appended}"
|
|
403
|
+
if system_prompt != stored:
|
|
404
|
+
session.append_system_prompt(system_prompt, appended=appended)
|
|
345
405
|
return cls(session, system_prompt, cwd=cwd, confirm=confirm, mode=mode,
|
|
346
406
|
config=config, toolset=toolset, model_override=model_override)
|
|
347
407
|
except BaseException:
|
|
@@ -380,7 +440,8 @@ class Agent:
|
|
|
380
440
|
name = self.model_name
|
|
381
441
|
if not name:
|
|
382
442
|
raise NoModelError("No model configured; log in first")
|
|
383
|
-
|
|
443
|
+
api_base, api_key = self.config.provider_auth(name)
|
|
444
|
+
key = (name, api_base, api_key)
|
|
384
445
|
if self._cached_model is None or self._cached_model[0] != key:
|
|
385
446
|
self._cached_model = (key, build_model(*key))
|
|
386
447
|
return self._cached_model[1]
|
|
@@ -405,11 +466,11 @@ class Agent:
|
|
|
405
466
|
the recent window, and so returns None on a history short enough that
|
|
406
467
|
there is nothing to summarize.
|
|
407
468
|
"""
|
|
469
|
+
window = compaction.context_window(self.model_name,
|
|
470
|
+
self.config.compaction_context_window)
|
|
408
471
|
if not force:
|
|
409
472
|
if not self.config.compaction_enabled:
|
|
410
473
|
return None
|
|
411
|
-
window = compaction.context_window(self.model_name,
|
|
412
|
-
self.config.compaction_context_window)
|
|
413
474
|
# Nothing to compare against, so counting would be wasted work: an
|
|
414
475
|
# unknown window disables auto-compaction outright.
|
|
415
476
|
if window is None:
|
|
@@ -427,6 +488,7 @@ class Agent:
|
|
|
427
488
|
tokens_before=tokens_before,
|
|
428
489
|
tool_schemas=self.tool_schemas,
|
|
429
490
|
system_prompt=self.system_prompt,
|
|
491
|
+
window=window,
|
|
430
492
|
)
|
|
431
493
|
if result is None:
|
|
432
494
|
return None
|
|
@@ -435,17 +497,33 @@ class Agent:
|
|
|
435
497
|
# append-message invariant (see _append_message) still holds afterwards.
|
|
436
498
|
self.session.append_compaction(result.summary, result.kept_messages, result.tokens_before)
|
|
437
499
|
self.history = result.messages
|
|
500
|
+
# The provider's count described the old context; drop the anchor so
|
|
501
|
+
# the fresh estimate below covers the compacted shape.
|
|
502
|
+
self._usage_anchor = None
|
|
438
503
|
result.tokens_after = await self.count_context_tokens()
|
|
439
504
|
return result
|
|
440
505
|
|
|
441
506
|
async def count_context_tokens(self) -> int:
|
|
442
|
-
"""
|
|
443
|
-
|
|
444
|
-
|
|
445
|
-
|
|
446
|
-
|
|
447
|
-
|
|
507
|
+
"""Tokens of everything the next request would send.
|
|
508
|
+
|
|
509
|
+
Anchored on the provider-reported usage of the last completed request
|
|
510
|
+
when one exists — the provider's own count is authoritative and
|
|
511
|
+
already includes the system prompt and tool schemas — with the chars/4
|
|
512
|
+
heuristic applied only to messages appended since. Without an anchor
|
|
513
|
+
the heuristic covers everything.
|
|
514
|
+
|
|
515
|
+
Off the event loop: the count serializes history, which is hundreds of
|
|
516
|
+
kilobytes on a long session. It runs at the top of every model step,
|
|
517
|
+
and every agent shares one loop, so counting inline stalls every other
|
|
518
|
+
agent's streaming output for as long as it takes.
|
|
448
519
|
"""
|
|
520
|
+
anchor = self._usage_anchor
|
|
521
|
+
if anchor is not None and anchor[0] <= len(self.history):
|
|
522
|
+
index, tokens = anchor
|
|
523
|
+
suffix = list(self.history[index:])
|
|
524
|
+
if not suffix:
|
|
525
|
+
return tokens
|
|
526
|
+
return tokens + await asyncio.to_thread(compaction.count_tokens, suffix)
|
|
449
527
|
return await asyncio.to_thread(
|
|
450
528
|
compaction.count_tokens, list(self.history), self.tool_schemas, self.system_prompt)
|
|
451
529
|
|
|
@@ -493,11 +571,13 @@ class Agent:
|
|
|
493
571
|
yield ToolStart(call.tool_call_id, call.tool_name, args)
|
|
494
572
|
slot.content = ("Error: 'todos' must be an array of "
|
|
495
573
|
"{content, status} objects.")
|
|
574
|
+
slot.outcome = "failed"
|
|
496
575
|
persist()
|
|
497
576
|
yield ToolEnd(call.tool_call_id, call.tool_name, slot.content)
|
|
498
577
|
return
|
|
499
578
|
self.todos = todos
|
|
500
579
|
slot.content = tools.render_todos(self.todos)
|
|
580
|
+
slot.outcome = "success"
|
|
501
581
|
persist()
|
|
502
582
|
yield TodosUpdate(list(self.todos))
|
|
503
583
|
|
|
@@ -511,16 +591,19 @@ class Agent:
|
|
|
511
591
|
prompt_text = str(args.get("prompt") or "").strip()
|
|
512
592
|
if not prompt_text:
|
|
513
593
|
slot.content = "Error: prompt is required."
|
|
594
|
+
slot.outcome = "failed"
|
|
514
595
|
persist()
|
|
515
596
|
yield ToolEnd(call.tool_call_id, call.tool_name, slot.content)
|
|
516
597
|
return
|
|
517
598
|
if not await self._permitted(call.tool_name, args):
|
|
518
599
|
slot.content = "User denied this operation."
|
|
600
|
+
slot.outcome = "denied"
|
|
519
601
|
persist()
|
|
520
602
|
yield ToolEnd(call.tool_call_id, call.tool_name, slot.content, denied=True)
|
|
521
603
|
return
|
|
522
604
|
slot.content = ("Handoff accepted: this session ended here and a new "
|
|
523
605
|
"session continued with the provided prompt.")
|
|
606
|
+
slot.outcome = "success"
|
|
524
607
|
persist()
|
|
525
608
|
yield ToolEnd(call.tool_call_id, call.tool_name, slot.content)
|
|
526
609
|
yield SessionHandoff(prompt_text)
|
|
@@ -552,13 +635,16 @@ class Agent:
|
|
|
552
635
|
if self.supervisor is None:
|
|
553
636
|
slot.content = ("Error: this only works in the interactive UI; "
|
|
554
637
|
"do the work yourself instead.")
|
|
638
|
+
slot.outcome = "failed"
|
|
555
639
|
elif not await self._permitted(call.tool_name, args):
|
|
556
640
|
slot.content = "User denied this operation."
|
|
641
|
+
slot.outcome = "denied"
|
|
557
642
|
persist()
|
|
558
643
|
yield ToolEnd(call.tool_call_id, call.tool_name, slot.content, denied=True)
|
|
559
644
|
return
|
|
560
645
|
else:
|
|
561
646
|
slot.content = await self.supervisor.handle(call.tool_name, args, caller=self)
|
|
647
|
+
slot.outcome = "success"
|
|
562
648
|
persist()
|
|
563
649
|
yield ToolEnd(call.tool_call_id, call.tool_name, slot.content)
|
|
564
650
|
|
|
@@ -573,12 +659,20 @@ class Agent:
|
|
|
573
659
|
"stop_job": _run_supervised,
|
|
574
660
|
}
|
|
575
661
|
|
|
576
|
-
async def run(self, user_input: str, *, expand: bool = True
|
|
662
|
+
async def run(self, user_input: str, *, expand: bool = True,
|
|
663
|
+
max_tool_calls: Optional[int] = None) -> AsyncIterator[AgentEvent]:
|
|
577
664
|
"""Run one user turn to completion, yielding events along the way.
|
|
578
665
|
|
|
579
666
|
``expand=False`` skips @path expansion, for callers that assembled the
|
|
580
667
|
prompt themselves and must not have unrelated text rewritten (piped
|
|
581
668
|
stdin, where a line like ``@foo.py`` is data rather than a mention).
|
|
669
|
+
|
|
670
|
+
``max_tool_calls`` bounds the tool calls this turn may execute. It is
|
|
671
|
+
enforced here, where every ToolCallPart is dispatched — including the
|
|
672
|
+
agent-handled tools like write_todos that produce no ToolStart — so no
|
|
673
|
+
tool can slip past the budget. On the budget every remaining call gets
|
|
674
|
+
an explicit refusal persisted as its result, ToolBudgetExhausted is
|
|
675
|
+
yielded and the turn ends.
|
|
582
676
|
"""
|
|
583
677
|
prompt = expand_mentions(user_input, self.cwd) if expand else user_input
|
|
584
678
|
# Agents this session started report in here, at the top of the next
|
|
@@ -590,6 +684,44 @@ class Agent:
|
|
|
590
684
|
self._append_message(agents_message(summary))
|
|
591
685
|
yield AgentsNotice(summary)
|
|
592
686
|
self._append_message(ModelRequest(parts=[UserPromptPart(content=prompt)]))
|
|
687
|
+
|
|
688
|
+
# Every turn leaves exactly one terminal ``turn_end`` record in the
|
|
689
|
+
# session: success, error (with the exception), max_tool_calls or
|
|
690
|
+
# interrupted. The record is not a message, so it reaches audits and
|
|
691
|
+
# ``paimon log`` without an error or a partial answer ever being
|
|
692
|
+
# replayed to the model as a normal assistant response.
|
|
693
|
+
outcome_recorded = False
|
|
694
|
+
|
|
695
|
+
def record_outcome(outcome: str, *, error: Optional[str] = None,
|
|
696
|
+
partial_text: Optional[str] = None) -> None:
|
|
697
|
+
nonlocal outcome_recorded
|
|
698
|
+
if outcome_recorded:
|
|
699
|
+
return
|
|
700
|
+
outcome_recorded = True
|
|
701
|
+
self.session.append_meta("turn_end", outcome=outcome, error=error,
|
|
702
|
+
partial_text=partial_text)
|
|
703
|
+
|
|
704
|
+
try:
|
|
705
|
+
async for event in self._run(record_outcome, max_tool_calls):
|
|
706
|
+
yield event
|
|
707
|
+
# A session handoff is the one path that ends the loop without
|
|
708
|
+
# its own record; it ended the turn deliberately.
|
|
709
|
+
record_outcome("success")
|
|
710
|
+
except asyncio.CancelledError:
|
|
711
|
+
record_outcome("interrupted")
|
|
712
|
+
raise
|
|
713
|
+
except GeneratorExit:
|
|
714
|
+
# The consumer abandoned the turn mid-iteration.
|
|
715
|
+
record_outcome("interrupted")
|
|
716
|
+
raise
|
|
717
|
+
except Exception as exc:
|
|
718
|
+
record_outcome("error", error=f"{type(exc).__name__}: {exc}")
|
|
719
|
+
raise
|
|
720
|
+
|
|
721
|
+
async def _run(self, record_outcome: Callable[..., None],
|
|
722
|
+
max_tool_calls: Optional[int]) -> AsyncIterator[AgentEvent]:
|
|
723
|
+
"""The model/tool loop of one turn; ``run`` wraps it to guarantee the
|
|
724
|
+
terminal ``turn_end`` record whatever way the turn ends."""
|
|
593
725
|
# A compaction that failed for a transient reason is retried on the
|
|
594
726
|
# next step of this turn — the context only keeps growing, so giving
|
|
595
727
|
# up on the first rate limit disables the safety net exactly when it
|
|
@@ -597,6 +729,11 @@ class Agent:
|
|
|
597
729
|
# for the rest of the turn instead of paying for it every step.
|
|
598
730
|
compaction_off = False
|
|
599
731
|
compaction_failures = 0
|
|
732
|
+
calls_made = 0 # every dispatched ToolCallPart counts, whatever its kind
|
|
733
|
+
# A provider context-overflow error triggers one forced compaction and
|
|
734
|
+
# one retry per turn; a second overflow means compaction cannot make
|
|
735
|
+
# the request fit, and retrying again would loop.
|
|
736
|
+
overflow_retried = False
|
|
600
737
|
|
|
601
738
|
while True:
|
|
602
739
|
# Messages the user queued while this turn was already running.
|
|
@@ -616,6 +753,7 @@ class Agent:
|
|
|
616
753
|
compaction_failures += 1
|
|
617
754
|
compaction_off = (not retry.is_transient(exc)
|
|
618
755
|
or compaction_failures >= _MAX_COMPACTION_FAILURES)
|
|
756
|
+
self.session.append_meta("compaction_failed", error=str(exc))
|
|
619
757
|
yield ContextCompactionFailed(str(exc))
|
|
620
758
|
else:
|
|
621
759
|
compaction_failures = 0
|
|
@@ -623,10 +761,14 @@ class Agent:
|
|
|
623
761
|
yield ContextCompacted(compacted.tokens_before, compacted.tokens_after)
|
|
624
762
|
|
|
625
763
|
model = self._model()
|
|
626
|
-
|
|
627
|
-
|
|
628
|
-
|
|
629
|
-
|
|
764
|
+
|
|
765
|
+
def build_request() -> list[ModelMessage]:
|
|
766
|
+
return [
|
|
767
|
+
ModelRequest(parts=[SystemPromptPart(content=self.system_prompt)]),
|
|
768
|
+
*_strip_foreign_thinking(self.history, model), # noqa: B023 — rebuilt each step
|
|
769
|
+
]
|
|
770
|
+
|
|
771
|
+
request_messages = build_request()
|
|
630
772
|
parameters = ModelRequestParameters(
|
|
631
773
|
function_tools=self._tool_definitions, allow_text_output=True
|
|
632
774
|
)
|
|
@@ -677,29 +819,60 @@ class Agent:
|
|
|
677
819
|
usage.cache_write_tokens)
|
|
678
820
|
break
|
|
679
821
|
except asyncio.CancelledError:
|
|
680
|
-
# Interrupted mid-stream:
|
|
681
|
-
#
|
|
682
|
-
#
|
|
683
|
-
|
|
684
|
-
|
|
685
|
-
|
|
686
|
-
provider_name=model.system,
|
|
687
|
-
))
|
|
822
|
+
# Interrupted mid-stream: the partial text is persisted in
|
|
823
|
+
# the turn_end record for the log, but no ModelResponse is
|
|
824
|
+
# appended — a truncated answer replayed as a completed
|
|
825
|
+
# one misleads both the resumed UI and the model, and a
|
|
826
|
+
# history ending on the user request stays resumable.
|
|
827
|
+
record_outcome("interrupted", partial_text=content or None)
|
|
688
828
|
raise
|
|
689
829
|
except Exception as exc: # noqa: BLE001 — classified by retry.is_transient
|
|
690
830
|
attempt += 1
|
|
831
|
+
if (not started and not overflow_retried
|
|
832
|
+
and retry.is_context_overflow(exc)):
|
|
833
|
+
# The provider says the request exceeded the window:
|
|
834
|
+
# compact and retry once, with an audit record of the
|
|
835
|
+
# overflow that triggered it.
|
|
836
|
+
overflow_retried = True
|
|
837
|
+
self.session.append_meta("context_overflow",
|
|
838
|
+
error=f"{type(exc).__name__}: {exc}")
|
|
839
|
+
try:
|
|
840
|
+
compacted = await self._maybe_compact(force=True)
|
|
841
|
+
except Exception as compact_exc: # noqa: BLE001 — original error still raises below
|
|
842
|
+
self.session.append_meta("compaction_failed",
|
|
843
|
+
error=str(compact_exc))
|
|
844
|
+
compacted = None
|
|
845
|
+
if compacted:
|
|
846
|
+
yield ContextCompacted(compacted.tokens_before,
|
|
847
|
+
compacted.tokens_after)
|
|
848
|
+
request_messages = build_request()
|
|
849
|
+
continue
|
|
691
850
|
if started or attempt >= retry.MAX_ATTEMPTS or not retry.is_transient(exc):
|
|
851
|
+
# Recorded here rather than in run()'s outer boundary
|
|
852
|
+
# because only this scope still holds the partial text
|
|
853
|
+
# the user already saw streaming.
|
|
854
|
+
record_outcome("error", error=f"{type(exc).__name__}: {exc}",
|
|
855
|
+
partial_text=content or None)
|
|
692
856
|
raise
|
|
693
857
|
delay = retry.backoff(attempt)
|
|
858
|
+
self.session.append_meta("model_retry", attempt=attempt,
|
|
859
|
+
error=retry.describe(exc))
|
|
694
860
|
yield ModelRetry(attempt, retry.MAX_ATTEMPTS, delay, retry.describe(exc))
|
|
695
861
|
await asyncio.sleep(delay)
|
|
696
862
|
|
|
697
863
|
self._append_message(response)
|
|
864
|
+
usage = response.usage
|
|
865
|
+
if usage.input_tokens:
|
|
866
|
+
# The provider's own count of this request plus its response —
|
|
867
|
+
# the authoritative context size at this history length.
|
|
868
|
+
self._usage_anchor = (len(self.history),
|
|
869
|
+
usage.input_tokens + (usage.output_tokens or 0))
|
|
698
870
|
if stats is not None:
|
|
699
871
|
yield stats
|
|
700
872
|
|
|
701
873
|
calls = [part for part in response.parts if isinstance(part, ToolCallPart)]
|
|
702
874
|
if not calls:
|
|
875
|
+
record_outcome("success")
|
|
703
876
|
yield TurnEnd()
|
|
704
877
|
return
|
|
705
878
|
|
|
@@ -709,7 +882,7 @@ class Agent:
|
|
|
709
882
|
# keep this placeholder.
|
|
710
883
|
returns = [
|
|
711
884
|
ToolReturnPart(tool_name=call.tool_name, content="Interrupted by user.",
|
|
712
|
-
tool_call_id=call.tool_call_id)
|
|
885
|
+
tool_call_id=call.tool_call_id, outcome="interrupted")
|
|
713
886
|
for call in calls
|
|
714
887
|
]
|
|
715
888
|
tool_request = ModelRequest(parts=returns)
|
|
@@ -719,16 +892,44 @@ class Agent:
|
|
|
719
892
|
"""Re-persist the tool request with the slots filled so far."""
|
|
720
893
|
self._replace_message(record_id, tool_request)
|
|
721
894
|
|
|
895
|
+
budget_hit = False
|
|
722
896
|
for slot, call in zip(returns, calls):
|
|
723
897
|
args = _parse_args(call.args)
|
|
724
898
|
name = call.tool_name
|
|
725
899
|
|
|
900
|
+
# The budget check comes before any dispatch, agent-handled
|
|
901
|
+
# tools included: a refused call never executes, and its slot
|
|
902
|
+
# records why instead of a generic interrupted placeholder.
|
|
903
|
+
if max_tool_calls is not None and calls_made >= max_tool_calls:
|
|
904
|
+
budget_hit = True
|
|
905
|
+
yield ToolStart(call.tool_call_id, name, args)
|
|
906
|
+
slot.content = (f"Not executed: the run reached its tool call "
|
|
907
|
+
f"budget (max_tool_calls={max_tool_calls}).")
|
|
908
|
+
persist()
|
|
909
|
+
yield ToolEnd(call.tool_call_id, name, slot.content)
|
|
910
|
+
continue
|
|
911
|
+
calls_made += 1
|
|
912
|
+
|
|
726
913
|
# A name outside this agent's tool set is rejected before the
|
|
727
914
|
# agent-handled table below, so excluding write_todos or
|
|
728
915
|
# start_new_session from a toolset actually disables them.
|
|
729
916
|
if name not in self.toolset:
|
|
730
917
|
yield ToolStart(call.tool_call_id, name, args)
|
|
731
918
|
slot.content = f"Error: unknown tool {name!r}"
|
|
919
|
+
slot.outcome = "failed"
|
|
920
|
+
persist()
|
|
921
|
+
yield ToolEnd(call.tool_call_id, name, slot.content)
|
|
922
|
+
continue
|
|
923
|
+
|
|
924
|
+
# One schema validation for every call — before gating and
|
|
925
|
+
# before the agent-handled dispatch below — so a malformed
|
|
926
|
+
# argument is a tool error the model can react to, never an
|
|
927
|
+
# exception ending the turn (TOOLS-1).
|
|
928
|
+
invalid = tools.validate_args(name, args, self.toolset)
|
|
929
|
+
if invalid is not None:
|
|
930
|
+
yield ToolStart(call.tool_call_id, name, args)
|
|
931
|
+
slot.content = f"Error: invalid arguments for {name}: {invalid}"
|
|
932
|
+
slot.outcome = "failed"
|
|
732
933
|
persist()
|
|
733
934
|
yield ToolEnd(call.tool_call_id, name, slot.content)
|
|
734
935
|
continue
|
|
@@ -750,6 +951,12 @@ class Agent:
|
|
|
750
951
|
ctx=self.tool_context)
|
|
751
952
|
|
|
752
953
|
slot.content = result
|
|
954
|
+
slot.outcome = "denied" if denied else "success"
|
|
753
955
|
persist()
|
|
754
956
|
yield ToolEnd(call.tool_call_id, name, result, denied=denied)
|
|
957
|
+
|
|
958
|
+
if budget_hit:
|
|
959
|
+
record_outcome("max_tool_calls")
|
|
960
|
+
yield ToolBudgetExhausted(max_tool_calls)
|
|
961
|
+
return
|
|
755
962
|
# loop again so the model can react to tool results
|
|
@@ -7,6 +7,7 @@ from pathlib import Path
|
|
|
7
7
|
|
|
8
8
|
from . import commands
|
|
9
9
|
from . import headless as headless_mode
|
|
10
|
+
from . import telemetry
|
|
10
11
|
from .agent import Agent
|
|
11
12
|
from .app import PaimonApp
|
|
12
13
|
from .config import Config
|
|
@@ -20,6 +21,7 @@ def main() -> None:
|
|
|
20
21
|
# argument sets and none of the launch-mode logic below applies to them.
|
|
21
22
|
argv = sys.argv[1:]
|
|
22
23
|
if argv and argv[0] in commands.REGISTRY:
|
|
24
|
+
telemetry.record_launch(argv[0])
|
|
23
25
|
sys.exit(commands.REGISTRY[argv[0]](argv[1:]))
|
|
24
26
|
|
|
25
27
|
parser = argparse.ArgumentParser(
|
|
@@ -84,6 +86,9 @@ def main() -> None:
|
|
|
84
86
|
except ValueError as exc:
|
|
85
87
|
parser.error(str(exc))
|
|
86
88
|
except PaimonError as exc:
|
|
89
|
+
if args.output_format != "text":
|
|
90
|
+
# Machine formats promise init/result lines even for this failure.
|
|
91
|
+
sys.exit(headless_mode.fail(str(exc), args.output_format))
|
|
87
92
|
print(f"paimon: {exc}", file=sys.stderr)
|
|
88
93
|
sys.exit(1)
|
|
89
94
|
if args.strict:
|
|
@@ -114,6 +119,7 @@ def main() -> None:
|
|
|
114
119
|
if args.profile:
|
|
115
120
|
flags += ["--profile", args.profile]
|
|
116
121
|
command = shlex.join([sys.executable, "-m", "paimon", *flags])
|
|
122
|
+
telemetry.record_launch("web", model=args.model or config.model)
|
|
117
123
|
Server(command, port=args.port).serve()
|
|
118
124
|
return
|
|
119
125
|
|
|
@@ -153,6 +159,7 @@ def main() -> None:
|
|
|
153
159
|
piped = headless_mode.read_stdin() if piped_stdin else ""
|
|
154
160
|
if not (args.prompt or "").strip() and not piped.strip():
|
|
155
161
|
parser.error("nothing to do: pass a prompt to --print or pipe one on stdin")
|
|
162
|
+
telemetry.record_launch("headless", model=args.model or config.model)
|
|
156
163
|
sys.exit(headless_mode.run(
|
|
157
164
|
prompt=args.prompt or "", piped=piped, cwd=Path.cwd(), mode=args.mode,
|
|
158
165
|
session=resume_session, output_format=args.output_format, config=config,
|
|
@@ -162,6 +169,9 @@ def main() -> None:
|
|
|
162
169
|
|
|
163
170
|
if args.model:
|
|
164
171
|
config.model = args.model
|
|
172
|
+
# The --tui relaunch under --web was already counted by the server.
|
|
173
|
+
if not args.tui:
|
|
174
|
+
telemetry.record_launch("tui", model=config.model)
|
|
165
175
|
try:
|
|
166
176
|
agent = Agent.open(cwd=Path.cwd(), session=resume_session, mode=args.mode, config=config)
|
|
167
177
|
except SessionError as exc:
|