paimon 0.2.3__tar.gz → 0.2.4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. {paimon-0.2.3/paimon.egg-info → paimon-0.2.4}/PKG-INFO +6 -2
  2. {paimon-0.2.3 → paimon-0.2.4}/README.md +4 -0
  3. {paimon-0.2.3 → paimon-0.2.4}/README.zh-CN.md +4 -0
  4. {paimon-0.2.3 → paimon-0.2.4}/paimon/agent.py +262 -55
  5. {paimon-0.2.3 → paimon-0.2.4}/paimon/cli.py +10 -0
  6. {paimon-0.2.3 → paimon-0.2.4}/paimon/commands.py +58 -102
  7. {paimon-0.2.3 → paimon-0.2.4}/paimon/compaction.py +40 -3
  8. {paimon-0.2.3 → paimon-0.2.4}/paimon/config.py +95 -11
  9. {paimon-0.2.3 → paimon-0.2.4}/paimon/headless.py +36 -13
  10. {paimon-0.2.3 → paimon-0.2.4}/paimon/jobs.py +2 -3
  11. {paimon-0.2.3 → paimon-0.2.4}/paimon/llm.py +16 -1
  12. {paimon-0.2.3 → paimon-0.2.4}/paimon/login.py +9 -2
  13. {paimon-0.2.3 → paimon-0.2.4}/paimon/pane.py +1 -1
  14. {paimon-0.2.3 → paimon-0.2.4}/paimon/prompt.py +15 -5
  15. {paimon-0.2.3 → paimon-0.2.4}/paimon/retry.py +27 -0
  16. {paimon-0.2.3 → paimon-0.2.4}/paimon/session.py +152 -12
  17. {paimon-0.2.3 → paimon-0.2.4}/paimon/skill/SKILL.md +5 -2
  18. paimon-0.2.4/paimon/telemetry.py +160 -0
  19. {paimon-0.2.3 → paimon-0.2.4}/paimon/tools.py +554 -15
  20. {paimon-0.2.3 → paimon-0.2.4}/paimon/ui.py +33 -7
  21. {paimon-0.2.3 → paimon-0.2.4/paimon.egg-info}/PKG-INFO +6 -2
  22. {paimon-0.2.3 → paimon-0.2.4}/paimon.egg-info/SOURCES.txt +1 -0
  23. paimon-0.2.4/paimon.egg-info/requires.txt +3 -0
  24. {paimon-0.2.3 → paimon-0.2.4}/paimon.egg-info/scm_file_list.json +3 -0
  25. paimon-0.2.4/paimon.egg-info/scm_version.json +8 -0
  26. {paimon-0.2.3 → paimon-0.2.4}/pyproject.toml +1 -1
  27. paimon-0.2.3/paimon.egg-info/requires.txt +0 -3
  28. paimon-0.2.3/paimon.egg-info/scm_version.json +0 -8
  29. {paimon-0.2.3 → paimon-0.2.4}/LICENSE +0 -0
  30. {paimon-0.2.3 → paimon-0.2.4}/MANIFEST.in +0 -0
  31. {paimon-0.2.3 → paimon-0.2.4}/paimon/__init__.py +0 -0
  32. {paimon-0.2.3 → paimon-0.2.4}/paimon/__main__.py +0 -0
  33. {paimon-0.2.3 → paimon-0.2.4}/paimon/app.py +0 -0
  34. {paimon-0.2.3 → paimon-0.2.4}/paimon/app.tcss +0 -0
  35. {paimon-0.2.3 → paimon-0.2.4}/paimon/aside.py +0 -0
  36. {paimon-0.2.3 → paimon-0.2.4}/paimon/diff.py +0 -0
  37. {paimon-0.2.3 → paimon-0.2.4}/paimon/errors.py +0 -0
  38. {paimon-0.2.3 → paimon-0.2.4}/paimon/lockfile.py +0 -0
  39. {paimon-0.2.3 → paimon-0.2.4}/paimon/mentions.py +0 -0
  40. {paimon-0.2.3 → paimon-0.2.4}/paimon/model_windows.py +0 -0
  41. {paimon-0.2.3 → paimon-0.2.4}/paimon/supervisor.py +0 -0
  42. {paimon-0.2.3 → paimon-0.2.4}/paimon/tabs.py +0 -0
  43. {paimon-0.2.3 → paimon-0.2.4}/paimon/taskpane.py +0 -0
  44. {paimon-0.2.3 → paimon-0.2.4}/paimon.egg-info/dependency_links.txt +0 -0
  45. {paimon-0.2.3 → paimon-0.2.4}/paimon.egg-info/entry_points.txt +0 -0
  46. {paimon-0.2.3 → paimon-0.2.4}/paimon.egg-info/top_level.txt +0 -0
  47. {paimon-0.2.3 → paimon-0.2.4}/setup.cfg +0 -0
@@ -1,11 +1,11 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: paimon
3
- Version: 0.2.3
3
+ Version: 0.2.4
4
4
  Summary: A minimal code agent built on pydantic-ai + textual
5
5
  Requires-Python: >=3.10
6
6
  Description-Content-Type: text/markdown
7
7
  License-File: LICENSE
8
- Requires-Dist: pydantic-ai-slim[anthropic,openai]>=2.17.0
8
+ Requires-Dist: pydantic-ai-slim[anthropic,openai]>=2.31.0
9
9
  Requires-Dist: textual>=8.2.7
10
10
  Requires-Dist: textual-serve>=1.1.3
11
11
  Dynamic: license-file
@@ -110,3 +110,7 @@ paimon --profile work # a separately configured account
110
110
  Each profile keeps its model settings in `~/.config/paimon/<name>/config.json`, written by the first launch or by `paimon login`. Sessions live in `~/.local/share/paimon/sessions/`. File changes render nicer if [delta](https://github.com/dandavison/delta) is installed.
111
111
 
112
112
  Read and edit modes run a small set of clearly read-only commands (`ls`, `cat`, `git status`, …) without asking; `--strict` turns that off. **This is a guardrail against agent mistakes, not a security boundary.** For real isolation, run Paimon inside a container or VM.
113
+
114
+ ## Telemetry
115
+
116
+ Each launch sends one anonymous event to Google Analytics: a random install id, the launch mode, the version, the OS name and the configured provider and model name. Nothing from your sessions, prompts, files or credentials is included. Set `PAIMON_NO_TELEMETRY=1` or `DO_NOT_TRACK=1` to turn it off.
@@ -98,3 +98,7 @@ paimon --profile work # a separately configured account
98
98
  Each profile keeps its model settings in `~/.config/paimon/<name>/config.json`, written by the first launch or by `paimon login`. Sessions live in `~/.local/share/paimon/sessions/`. File changes render nicer if [delta](https://github.com/dandavison/delta) is installed.
99
99
 
100
100
  Read and edit modes run a small set of clearly read-only commands (`ls`, `cat`, `git status`, …) without asking; `--strict` turns that off. **This is a guardrail against agent mistakes, not a security boundary.** For real isolation, run Paimon inside a container or VM.
101
+
102
+ ## Telemetry
103
+
104
+ Each launch sends one anonymous event to Google Analytics: a random install id, the launch mode, the version, the OS name and the configured provider and model name. Nothing from your sessions, prompts, files or credentials is included. Set `PAIMON_NO_TELEMETRY=1` or `DO_NOT_TRACK=1` to turn it off.
@@ -98,3 +98,7 @@ paimon --profile work # 单独配置的另一个账号
98
98
  每个 profile 的模型设置保存在 `~/.config/paimon/<name>/config.json`,由首次启动或 `paimon login` 写入。会话存放在 `~/.local/share/paimon/sessions/`。安装 [delta](https://github.com/dandavison/delta) 后文件改动的展示效果更好。
99
99
 
100
100
  read 和 edit 模式会不经询问执行一小组明确只读的命令(`ls`、`cat`、`git status` 等),`--strict` 可以关掉。**这是防止 agent 失误的护栏,不是安全边界。** 需要真正的隔离时,请在容器或虚拟机中运行 Paimon。
101
+
102
+ ## 遥测
103
+
104
+ 每次启动会向 Google Analytics 发送一条匿名事件:随机生成的安装 id、启动方式、版本号、操作系统名称,以及配置的 provider 和模型名。会话、提示词、文件和密钥一概不上报。设置 `PAIMON_NO_TELEMETRY=1` 或 `DO_NOT_TRACK=1` 可以关闭。
@@ -107,6 +107,15 @@ class RequestStats:
107
107
  cache_write_tokens: int = 0
108
108
 
109
109
 
110
+ @dataclass
111
+ class ToolBudgetExhausted:
112
+ """The turn hit its tool-call budget: the remaining calls were refused
113
+ with an explicit result and the turn ends without another model request.
114
+ """
115
+
116
+ limit: int
117
+
118
+
110
119
  @dataclass
111
120
  class TurnEnd:
112
121
  pass
@@ -161,9 +170,9 @@ class AgentsNotice:
161
170
  # on isinstance; the alias exists so a type checker can flag an unhandled one.
162
171
  AgentEvent = (
163
172
  TextDelta | ReasoningDelta | ToolStart | ToolEnd | TodosUpdate
164
- | SessionHandoff | RequestStats | TurnEnd | ContextCompacted
165
- | ContextCompactionFailed | ModelRetry | UserInput | CompactionNotice
166
- | AgentsNotice
173
+ | SessionHandoff | RequestStats | ToolBudgetExhausted | TurnEnd
174
+ | ContextCompacted | ContextCompactionFailed | ModelRetry | UserInput
175
+ | CompactionNotice | AgentsNotice
167
176
  )
168
177
 
169
178
 
@@ -184,12 +193,14 @@ def replay_events(messages: list[ModelMessage]) -> list[AgentEvent]:
184
193
  """Persisted messages replayed as the events a live ``Agent.run`` yields.
185
194
 
186
195
  Lets a UI render resumed history through the same code path as live turns.
196
+ Tools run serially, so each call replays as its ToolStart immediately
197
+ followed by its ToolEnd (looked up in the next request), matching the live
198
+ order — not all starts of a batch first. The persisted ``outcome`` field
199
+ restores denied styling.
187
200
  """
188
201
  events: list[AgentEvent] = []
189
- # write_todos calls whose arguments never made a todo update: they were
190
- # rejected live as a normal tool error, so they replay as one.
191
- rejected_todos: set[str] = set()
192
- for message in messages:
202
+ consumed: set[str] = set() # tool returns already replayed with their call
203
+ for index, message in enumerate(messages):
193
204
  if is_summary_message(message):
194
205
  events.append(CompactionNotice())
195
206
  continue
@@ -200,11 +211,20 @@ def replay_events(messages: list[ModelMessage]) -> list[AgentEvent]:
200
211
  for part in message.parts:
201
212
  if isinstance(part, UserPromptPart) and isinstance(part.content, str) and part.content:
202
213
  events.append(UserInput(part.content))
203
- elif isinstance(part, ToolReturnPart) and (part.tool_name != "write_todos"
204
- or part.tool_call_id in rejected_todos):
214
+ elif (isinstance(part, ToolReturnPart)
215
+ and part.tool_call_id not in consumed
216
+ and part.tool_name != "write_todos"):
217
+ # A return whose call is gone (compaction cut mid-turn):
218
+ # better an unpaired result than a silent hole.
205
219
  events.append(ToolEnd(part.tool_call_id, part.tool_name,
206
- str(part.content or "(no output)")))
220
+ str(part.content or "(no output)"),
221
+ denied=part.outcome == "denied"))
207
222
  elif isinstance(message, ModelResponse):
223
+ following = messages[index + 1] if index + 1 < len(messages) else None
224
+ returns: dict[str, ToolReturnPart] = {}
225
+ if isinstance(following, ModelRequest):
226
+ returns = {part.tool_call_id: part for part in following.parts
227
+ if isinstance(part, ToolReturnPart)}
208
228
  for part in message.parts:
209
229
  if isinstance(part, ThinkingPart) and part.content:
210
230
  events.append(ReasoningDelta(part.content))
@@ -212,23 +232,37 @@ def replay_events(messages: list[ModelMessage]) -> list[AgentEvent]:
212
232
  events.append(TextDelta(part.content))
213
233
  elif isinstance(part, ToolCallPart):
214
234
  args = _parse_args(part.args)
215
- todos = (tools.normalize_todos(args.get("todos"))
216
- if part.tool_name == "write_todos" else None)
217
- if todos is not None:
218
- events.append(TodosUpdate(todos))
219
- else:
220
- if part.tool_name == "write_todos":
221
- rejected_todos.add(part.tool_call_id)
222
- events.append(ToolStart(part.tool_call_id, part.tool_name, args))
235
+ result = returns.get(part.tool_call_id)
236
+ if result is not None:
237
+ consumed.add(part.tool_call_id)
238
+ if part.tool_name == "write_todos":
239
+ todos = tools.normalize_todos(args.get("todos"))
240
+ # Live, a write_todos that actually ran yields only a
241
+ # TodosUpdate; one that was rejected or never executed
242
+ # yields a normal failed call.
243
+ if todos is not None and (result is None or result.outcome == "success"):
244
+ events.append(TodosUpdate(todos))
245
+ continue
246
+ events.append(ToolStart(part.tool_call_id, part.tool_name, args))
247
+ if result is not None:
248
+ events.append(ToolEnd(part.tool_call_id, part.tool_name,
249
+ str(result.content or "(no output)"),
250
+ denied=result.outcome == "denied"))
223
251
  return events
224
252
 
225
253
 
226
254
  def _strip_foreign_thinking(history: list[ModelMessage], model: Model) -> list[ModelMessage]:
227
- """Drop thinking parts produced by a different provider/model.
255
+ """Adapt thinking parts produced by a different provider/model.
228
256
 
229
257
  Preserved thinking is only valid replayed verbatim to the model that
230
- produced it; other endpoints may reject or misread it (cf. pi's
231
- transformMessages: same model keeps thinking, a changed model strips it).
258
+ produced it, so a foreign message's readable thinking is converted to a
259
+ plain text block — the rationale it carries is still context — while its
260
+ signatures and encrypted/redacted payloads are dropped, never sent to a
261
+ provider that did not produce them (cf. pi's transformMessages, which
262
+ converts rather than deletes readable thinking).
263
+
264
+ Only the request being built is touched; the persisted history keeps the
265
+ original parts.
232
266
  """
233
267
  current = (model.system, model.model_name)
234
268
  sanitized: list[ModelMessage] = []
@@ -236,9 +270,16 @@ def _strip_foreign_thinking(history: list[ModelMessage], model: Model) -> list[M
236
270
  if (isinstance(message, ModelResponse)
237
271
  and (message.provider_name, message.model_name) != current
238
272
  and any(isinstance(part, ThinkingPart) for part in message.parts)):
239
- message = dataclasses.replace(
240
- message, parts=[part for part in message.parts if not isinstance(part, ThinkingPart)]
241
- )
273
+ parts = []
274
+ for part in message.parts:
275
+ if not isinstance(part, ThinkingPart):
276
+ parts.append(part)
277
+ elif part.content:
278
+ parts.append(TextPart(content=f"<thinking>\n{part.content}\n</thinking>"))
279
+ # else: encrypted or redacted thinking with nothing readable
280
+ if not parts:
281
+ continue # nothing a different provider can use
282
+ message = dataclasses.replace(message, parts=parts)
242
283
  sanitized.append(message)
243
284
  return sanitized
244
285
 
@@ -289,8 +330,9 @@ class Agent:
289
330
  # None where nobody can type while a turn runs (headless, tests).
290
331
  self.pending: Optional[PendingFn] = None
291
332
  # Per-agent tool state, kept off the tool functions so one agent's
292
- # shell overflow files stay invisible to the next one.
293
- self.tool_context = tools.ToolContext()
333
+ # shell overflow files stay invisible to the next one. The session
334
+ # rides along for the history tools.
335
+ self.tool_context = tools.ToolContext(session=session)
294
336
  self.todos: list[dict] = []
295
337
  self.session = session
296
338
  self.system_prompt = system_prompt
@@ -300,6 +342,12 @@ class Agent:
300
342
  self.tool_schemas = tools.schemas(self.toolset)
301
343
  self._tool_definitions = tools.definitions(self.toolset)
302
344
  self._cached_model: Optional[tuple[tuple, Model]] = None
345
+ # (history length, provider-reported tokens) after the last completed
346
+ # request: the authoritative context size at that point, used by
347
+ # count_context_tokens so the chars/4 heuristic only covers what was
348
+ # appended since. None until a request reports usage, and again after
349
+ # compaction reshapes the history.
350
+ self._usage_anchor: Optional[tuple[int, int]] = None
303
351
 
304
352
  @classmethod
305
353
  def open(cls, cwd: Optional[Path] = None, *, session: Optional[Session] = None,
@@ -313,7 +361,12 @@ class Agent:
313
361
 
314
362
  ``append_system_prompt`` is added to the end of a new session's system
315
363
  prompt and persisted with it, so a resumed session keeps it. Resuming
316
- with it set raises ``ValueError``: the persisted prompt is immutable.
364
+ with it set raises ``ValueError``: the persisted role is immutable.
365
+ The dynamic parts of the prompt (date, environment, AGENTS.md) are
366
+ rebuilt on resume — a session picked up months later must not keep
367
+ telling the model the old date or stale project rules — and when they
368
+ changed, the rebuilt snapshot is appended to the log, so it always
369
+ records the prompt each turn actually ran with.
317
370
  ``parent`` marks the new session as a subagent's, which keeps it out of
318
371
  the session listings its parent shows up in.
319
372
 
@@ -334,14 +387,21 @@ class Agent:
334
387
  # Agent is returned to unlock it, so the lock is released here.
335
388
  try:
336
389
  if is_new:
390
+ appended = append_system_prompt.strip() if append_system_prompt else None
337
391
  system_prompt = build_system_prompt(cwd)
338
- if append_system_prompt:
339
- system_prompt += f"\n\n{append_system_prompt.strip()}"
340
- session.append_system_prompt(system_prompt)
392
+ if appended:
393
+ system_prompt += f"\n\n{appended}"
394
+ session.append_system_prompt(system_prompt, appended=appended)
341
395
  else:
342
- system_prompt = session.system_prompt()
343
- if system_prompt is None:
396
+ session.require_supported_format()
397
+ stored, appended = session.system_prompt_parts()
398
+ if stored is None:
344
399
  raise SessionIncompleteError("Session does not contain a persisted system prompt")
400
+ system_prompt = build_system_prompt(cwd)
401
+ if appended:
402
+ system_prompt += f"\n\n{appended}"
403
+ if system_prompt != stored:
404
+ session.append_system_prompt(system_prompt, appended=appended)
345
405
  return cls(session, system_prompt, cwd=cwd, confirm=confirm, mode=mode,
346
406
  config=config, toolset=toolset, model_override=model_override)
347
407
  except BaseException:
@@ -380,7 +440,8 @@ class Agent:
380
440
  name = self.model_name
381
441
  if not name:
382
442
  raise NoModelError("No model configured; log in first")
383
- key = (name, self.config.api_base, self.config.api_key)
443
+ api_base, api_key = self.config.provider_auth(name)
444
+ key = (name, api_base, api_key)
384
445
  if self._cached_model is None or self._cached_model[0] != key:
385
446
  self._cached_model = (key, build_model(*key))
386
447
  return self._cached_model[1]
@@ -405,11 +466,11 @@ class Agent:
405
466
  the recent window, and so returns None on a history short enough that
406
467
  there is nothing to summarize.
407
468
  """
469
+ window = compaction.context_window(self.model_name,
470
+ self.config.compaction_context_window)
408
471
  if not force:
409
472
  if not self.config.compaction_enabled:
410
473
  return None
411
- window = compaction.context_window(self.model_name,
412
- self.config.compaction_context_window)
413
474
  # Nothing to compare against, so counting would be wasted work: an
414
475
  # unknown window disables auto-compaction outright.
415
476
  if window is None:
@@ -427,6 +488,7 @@ class Agent:
427
488
  tokens_before=tokens_before,
428
489
  tool_schemas=self.tool_schemas,
429
490
  system_prompt=self.system_prompt,
491
+ window=window,
430
492
  )
431
493
  if result is None:
432
494
  return None
@@ -435,17 +497,33 @@ class Agent:
435
497
  # append-message invariant (see _append_message) still holds afterwards.
436
498
  self.session.append_compaction(result.summary, result.kept_messages, result.tokens_before)
437
499
  self.history = result.messages
500
+ # The provider's count described the old context; drop the anchor so
501
+ # the fresh estimate below covers the compacted shape.
502
+ self._usage_anchor = None
438
503
  result.tokens_after = await self.count_context_tokens()
439
504
  return result
440
505
 
441
506
  async def count_context_tokens(self) -> int:
442
- """Estimate the tokens of everything the next request would send.
443
-
444
- Off the event loop: the count serializes the whole history, which is
445
- hundreds of kilobytes on a long session. It runs at the top of every
446
- model step, and every agent shares one loop, so counting inline stalls
447
- every other agent's streaming output for as long as it takes.
507
+ """Tokens of everything the next request would send.
508
+
509
+ Anchored on the provider-reported usage of the last completed request
510
+ when one exists — the provider's own count is authoritative and
511
+ already includes the system prompt and tool schemas — with the chars/4
512
+ heuristic applied only to messages appended since. Without an anchor
513
+ the heuristic covers everything.
514
+
515
+ Off the event loop: the count serializes history, which is hundreds of
516
+ kilobytes on a long session. It runs at the top of every model step,
517
+ and every agent shares one loop, so counting inline stalls every other
518
+ agent's streaming output for as long as it takes.
448
519
  """
520
+ anchor = self._usage_anchor
521
+ if anchor is not None and anchor[0] <= len(self.history):
522
+ index, tokens = anchor
523
+ suffix = list(self.history[index:])
524
+ if not suffix:
525
+ return tokens
526
+ return tokens + await asyncio.to_thread(compaction.count_tokens, suffix)
449
527
  return await asyncio.to_thread(
450
528
  compaction.count_tokens, list(self.history), self.tool_schemas, self.system_prompt)
451
529
 
@@ -493,11 +571,13 @@ class Agent:
493
571
  yield ToolStart(call.tool_call_id, call.tool_name, args)
494
572
  slot.content = ("Error: 'todos' must be an array of "
495
573
  "{content, status} objects.")
574
+ slot.outcome = "failed"
496
575
  persist()
497
576
  yield ToolEnd(call.tool_call_id, call.tool_name, slot.content)
498
577
  return
499
578
  self.todos = todos
500
579
  slot.content = tools.render_todos(self.todos)
580
+ slot.outcome = "success"
501
581
  persist()
502
582
  yield TodosUpdate(list(self.todos))
503
583
 
@@ -511,16 +591,19 @@ class Agent:
511
591
  prompt_text = str(args.get("prompt") or "").strip()
512
592
  if not prompt_text:
513
593
  slot.content = "Error: prompt is required."
594
+ slot.outcome = "failed"
514
595
  persist()
515
596
  yield ToolEnd(call.tool_call_id, call.tool_name, slot.content)
516
597
  return
517
598
  if not await self._permitted(call.tool_name, args):
518
599
  slot.content = "User denied this operation."
600
+ slot.outcome = "denied"
519
601
  persist()
520
602
  yield ToolEnd(call.tool_call_id, call.tool_name, slot.content, denied=True)
521
603
  return
522
604
  slot.content = ("Handoff accepted: this session ended here and a new "
523
605
  "session continued with the provided prompt.")
606
+ slot.outcome = "success"
524
607
  persist()
525
608
  yield ToolEnd(call.tool_call_id, call.tool_name, slot.content)
526
609
  yield SessionHandoff(prompt_text)
@@ -552,13 +635,16 @@ class Agent:
552
635
  if self.supervisor is None:
553
636
  slot.content = ("Error: this only works in the interactive UI; "
554
637
  "do the work yourself instead.")
638
+ slot.outcome = "failed"
555
639
  elif not await self._permitted(call.tool_name, args):
556
640
  slot.content = "User denied this operation."
641
+ slot.outcome = "denied"
557
642
  persist()
558
643
  yield ToolEnd(call.tool_call_id, call.tool_name, slot.content, denied=True)
559
644
  return
560
645
  else:
561
646
  slot.content = await self.supervisor.handle(call.tool_name, args, caller=self)
647
+ slot.outcome = "success"
562
648
  persist()
563
649
  yield ToolEnd(call.tool_call_id, call.tool_name, slot.content)
564
650
 
@@ -573,12 +659,20 @@ class Agent:
573
659
  "stop_job": _run_supervised,
574
660
  }
575
661
 
576
- async def run(self, user_input: str, *, expand: bool = True) -> AsyncIterator[AgentEvent]:
662
+ async def run(self, user_input: str, *, expand: bool = True,
663
+ max_tool_calls: Optional[int] = None) -> AsyncIterator[AgentEvent]:
577
664
  """Run one user turn to completion, yielding events along the way.
578
665
 
579
666
  ``expand=False`` skips @path expansion, for callers that assembled the
580
667
  prompt themselves and must not have unrelated text rewritten (piped
581
668
  stdin, where a line like ``@foo.py`` is data rather than a mention).
669
+
670
+ ``max_tool_calls`` bounds the tool calls this turn may execute. It is
671
+ enforced here, where every ToolCallPart is dispatched — including the
672
+ agent-handled tools like write_todos that produce no ToolStart — so no
673
+ tool can slip past the budget. On the budget every remaining call gets
674
+ an explicit refusal persisted as its result, ToolBudgetExhausted is
675
+ yielded and the turn ends.
582
676
  """
583
677
  prompt = expand_mentions(user_input, self.cwd) if expand else user_input
584
678
  # Agents this session started report in here, at the top of the next
@@ -590,6 +684,44 @@ class Agent:
590
684
  self._append_message(agents_message(summary))
591
685
  yield AgentsNotice(summary)
592
686
  self._append_message(ModelRequest(parts=[UserPromptPart(content=prompt)]))
687
+
688
+ # Every turn leaves exactly one terminal ``turn_end`` record in the
689
+ # session: success, error (with the exception), max_tool_calls or
690
+ # interrupted. The record is not a message, so it reaches audits and
691
+ # ``paimon log`` without an error or a partial answer ever being
692
+ # replayed to the model as a normal assistant response.
693
+ outcome_recorded = False
694
+
695
+ def record_outcome(outcome: str, *, error: Optional[str] = None,
696
+ partial_text: Optional[str] = None) -> None:
697
+ nonlocal outcome_recorded
698
+ if outcome_recorded:
699
+ return
700
+ outcome_recorded = True
701
+ self.session.append_meta("turn_end", outcome=outcome, error=error,
702
+ partial_text=partial_text)
703
+
704
+ try:
705
+ async for event in self._run(record_outcome, max_tool_calls):
706
+ yield event
707
+ # A session handoff is the one path that ends the loop without
708
+ # its own record; it ended the turn deliberately.
709
+ record_outcome("success")
710
+ except asyncio.CancelledError:
711
+ record_outcome("interrupted")
712
+ raise
713
+ except GeneratorExit:
714
+ # The consumer abandoned the turn mid-iteration.
715
+ record_outcome("interrupted")
716
+ raise
717
+ except Exception as exc:
718
+ record_outcome("error", error=f"{type(exc).__name__}: {exc}")
719
+ raise
720
+
721
+ async def _run(self, record_outcome: Callable[..., None],
722
+ max_tool_calls: Optional[int]) -> AsyncIterator[AgentEvent]:
723
+ """The model/tool loop of one turn; ``run`` wraps it to guarantee the
724
+ terminal ``turn_end`` record whatever way the turn ends."""
593
725
  # A compaction that failed for a transient reason is retried on the
594
726
  # next step of this turn — the context only keeps growing, so giving
595
727
  # up on the first rate limit disables the safety net exactly when it
@@ -597,6 +729,11 @@ class Agent:
597
729
  # for the rest of the turn instead of paying for it every step.
598
730
  compaction_off = False
599
731
  compaction_failures = 0
732
+ calls_made = 0 # every dispatched ToolCallPart counts, whatever its kind
733
+ # A provider context-overflow error triggers one forced compaction and
734
+ # one retry per turn; a second overflow means compaction cannot make
735
+ # the request fit, and retrying again would loop.
736
+ overflow_retried = False
600
737
 
601
738
  while True:
602
739
  # Messages the user queued while this turn was already running.
@@ -616,6 +753,7 @@ class Agent:
616
753
  compaction_failures += 1
617
754
  compaction_off = (not retry.is_transient(exc)
618
755
  or compaction_failures >= _MAX_COMPACTION_FAILURES)
756
+ self.session.append_meta("compaction_failed", error=str(exc))
619
757
  yield ContextCompactionFailed(str(exc))
620
758
  else:
621
759
  compaction_failures = 0
@@ -623,10 +761,14 @@ class Agent:
623
761
  yield ContextCompacted(compacted.tokens_before, compacted.tokens_after)
624
762
 
625
763
  model = self._model()
626
- request_messages: list[ModelMessage] = [
627
- ModelRequest(parts=[SystemPromptPart(content=self.system_prompt)]),
628
- *_strip_foreign_thinking(self.history, model),
629
- ]
764
+
765
+ def build_request() -> list[ModelMessage]:
766
+ return [
767
+ ModelRequest(parts=[SystemPromptPart(content=self.system_prompt)]),
768
+ *_strip_foreign_thinking(self.history, model), # noqa: B023 — rebuilt each step
769
+ ]
770
+
771
+ request_messages = build_request()
630
772
  parameters = ModelRequestParameters(
631
773
  function_tools=self._tool_definitions, allow_text_output=True
632
774
  )
@@ -677,29 +819,60 @@ class Agent:
677
819
  usage.cache_write_tokens)
678
820
  break
679
821
  except asyncio.CancelledError:
680
- # Interrupted mid-stream: keep partial text but drop incomplete
681
- # tool calls and thinking (Z.ai-style preserved thinking must be
682
- # replayed complete or not at all) so the history stays valid.
683
- self._append_message(ModelResponse(
684
- parts=[TextPart(content=content or "(interrupted)")],
685
- model_name=model.model_name,
686
- provider_name=model.system,
687
- ))
822
+ # Interrupted mid-stream: the partial text is persisted in
823
+ # the turn_end record for the log, but no ModelResponse is
824
+ # appended — a truncated answer replayed as a completed
825
+ # one misleads both the resumed UI and the model, and a
826
+ # history ending on the user request stays resumable.
827
+ record_outcome("interrupted", partial_text=content or None)
688
828
  raise
689
829
  except Exception as exc: # noqa: BLE001 — classified by retry.is_transient
690
830
  attempt += 1
831
+ if (not started and not overflow_retried
832
+ and retry.is_context_overflow(exc)):
833
+ # The provider says the request exceeded the window:
834
+ # compact and retry once, with an audit record of the
835
+ # overflow that triggered it.
836
+ overflow_retried = True
837
+ self.session.append_meta("context_overflow",
838
+ error=f"{type(exc).__name__}: {exc}")
839
+ try:
840
+ compacted = await self._maybe_compact(force=True)
841
+ except Exception as compact_exc: # noqa: BLE001 — original error still raises below
842
+ self.session.append_meta("compaction_failed",
843
+ error=str(compact_exc))
844
+ compacted = None
845
+ if compacted:
846
+ yield ContextCompacted(compacted.tokens_before,
847
+ compacted.tokens_after)
848
+ request_messages = build_request()
849
+ continue
691
850
  if started or attempt >= retry.MAX_ATTEMPTS or not retry.is_transient(exc):
851
+ # Recorded here rather than in run()'s outer boundary
852
+ # because only this scope still holds the partial text
853
+ # the user already saw streaming.
854
+ record_outcome("error", error=f"{type(exc).__name__}: {exc}",
855
+ partial_text=content or None)
692
856
  raise
693
857
  delay = retry.backoff(attempt)
858
+ self.session.append_meta("model_retry", attempt=attempt,
859
+ error=retry.describe(exc))
694
860
  yield ModelRetry(attempt, retry.MAX_ATTEMPTS, delay, retry.describe(exc))
695
861
  await asyncio.sleep(delay)
696
862
 
697
863
  self._append_message(response)
864
+ usage = response.usage
865
+ if usage.input_tokens:
866
+ # The provider's own count of this request plus its response —
867
+ # the authoritative context size at this history length.
868
+ self._usage_anchor = (len(self.history),
869
+ usage.input_tokens + (usage.output_tokens or 0))
698
870
  if stats is not None:
699
871
  yield stats
700
872
 
701
873
  calls = [part for part in response.parts if isinstance(part, ToolCallPart)]
702
874
  if not calls:
875
+ record_outcome("success")
703
876
  yield TurnEnd()
704
877
  return
705
878
 
@@ -709,7 +882,7 @@ class Agent:
709
882
  # keep this placeholder.
710
883
  returns = [
711
884
  ToolReturnPart(tool_name=call.tool_name, content="Interrupted by user.",
712
- tool_call_id=call.tool_call_id)
885
+ tool_call_id=call.tool_call_id, outcome="interrupted")
713
886
  for call in calls
714
887
  ]
715
888
  tool_request = ModelRequest(parts=returns)
@@ -719,16 +892,44 @@ class Agent:
719
892
  """Re-persist the tool request with the slots filled so far."""
720
893
  self._replace_message(record_id, tool_request)
721
894
 
895
+ budget_hit = False
722
896
  for slot, call in zip(returns, calls):
723
897
  args = _parse_args(call.args)
724
898
  name = call.tool_name
725
899
 
900
+ # The budget check comes before any dispatch, agent-handled
901
+ # tools included: a refused call never executes, and its slot
902
+ # records why instead of a generic interrupted placeholder.
903
+ if max_tool_calls is not None and calls_made >= max_tool_calls:
904
+ budget_hit = True
905
+ yield ToolStart(call.tool_call_id, name, args)
906
+ slot.content = (f"Not executed: the run reached its tool call "
907
+ f"budget (max_tool_calls={max_tool_calls}).")
908
+ persist()
909
+ yield ToolEnd(call.tool_call_id, name, slot.content)
910
+ continue
911
+ calls_made += 1
912
+
726
913
  # A name outside this agent's tool set is rejected before the
727
914
  # agent-handled table below, so excluding write_todos or
728
915
  # start_new_session from a toolset actually disables them.
729
916
  if name not in self.toolset:
730
917
  yield ToolStart(call.tool_call_id, name, args)
731
918
  slot.content = f"Error: unknown tool {name!r}"
919
+ slot.outcome = "failed"
920
+ persist()
921
+ yield ToolEnd(call.tool_call_id, name, slot.content)
922
+ continue
923
+
924
+ # One schema validation for every call — before gating and
925
+ # before the agent-handled dispatch below — so a malformed
926
+ # argument is a tool error the model can react to, never an
927
+ # exception ending the turn (TOOLS-1).
928
+ invalid = tools.validate_args(name, args, self.toolset)
929
+ if invalid is not None:
930
+ yield ToolStart(call.tool_call_id, name, args)
931
+ slot.content = f"Error: invalid arguments for {name}: {invalid}"
932
+ slot.outcome = "failed"
732
933
  persist()
733
934
  yield ToolEnd(call.tool_call_id, name, slot.content)
734
935
  continue
@@ -750,6 +951,12 @@ class Agent:
750
951
  ctx=self.tool_context)
751
952
 
752
953
  slot.content = result
954
+ slot.outcome = "denied" if denied else "success"
753
955
  persist()
754
956
  yield ToolEnd(call.tool_call_id, name, result, denied=denied)
957
+
958
+ if budget_hit:
959
+ record_outcome("max_tool_calls")
960
+ yield ToolBudgetExhausted(max_tool_calls)
961
+ return
755
962
  # loop again so the model can react to tool results
@@ -7,6 +7,7 @@ from pathlib import Path
7
7
 
8
8
  from . import commands
9
9
  from . import headless as headless_mode
10
+ from . import telemetry
10
11
  from .agent import Agent
11
12
  from .app import PaimonApp
12
13
  from .config import Config
@@ -20,6 +21,7 @@ def main() -> None:
20
21
  # argument sets and none of the launch-mode logic below applies to them.
21
22
  argv = sys.argv[1:]
22
23
  if argv and argv[0] in commands.REGISTRY:
24
+ telemetry.record_launch(argv[0])
23
25
  sys.exit(commands.REGISTRY[argv[0]](argv[1:]))
24
26
 
25
27
  parser = argparse.ArgumentParser(
@@ -84,6 +86,9 @@ def main() -> None:
84
86
  except ValueError as exc:
85
87
  parser.error(str(exc))
86
88
  except PaimonError as exc:
89
+ if args.output_format != "text":
90
+ # Machine formats promise init/result lines even for this failure.
91
+ sys.exit(headless_mode.fail(str(exc), args.output_format))
87
92
  print(f"paimon: {exc}", file=sys.stderr)
88
93
  sys.exit(1)
89
94
  if args.strict:
@@ -114,6 +119,7 @@ def main() -> None:
114
119
  if args.profile:
115
120
  flags += ["--profile", args.profile]
116
121
  command = shlex.join([sys.executable, "-m", "paimon", *flags])
122
+ telemetry.record_launch("web", model=args.model or config.model)
117
123
  Server(command, port=args.port).serve()
118
124
  return
119
125
 
@@ -153,6 +159,7 @@ def main() -> None:
153
159
  piped = headless_mode.read_stdin() if piped_stdin else ""
154
160
  if not (args.prompt or "").strip() and not piped.strip():
155
161
  parser.error("nothing to do: pass a prompt to --print or pipe one on stdin")
162
+ telemetry.record_launch("headless", model=args.model or config.model)
156
163
  sys.exit(headless_mode.run(
157
164
  prompt=args.prompt or "", piped=piped, cwd=Path.cwd(), mode=args.mode,
158
165
  session=resume_session, output_format=args.output_format, config=config,
@@ -162,6 +169,9 @@ def main() -> None:
162
169
 
163
170
  if args.model:
164
171
  config.model = args.model
172
+ # The --tui relaunch under --web was already counted by the server.
173
+ if not args.tui:
174
+ telemetry.record_launch("tui", model=config.model)
165
175
  try:
166
176
  agent = Agent.open(cwd=Path.cwd(), session=resume_session, mode=args.mode, config=config)
167
177
  except SessionError as exc: