paimon 0.2.0__tar.gz → 0.2.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. paimon-0.2.2/MANIFEST.in +7 -0
  2. {paimon-0.2.0 → paimon-0.2.2}/PKG-INFO +10 -4
  3. {paimon-0.2.0 → paimon-0.2.2}/README.md +9 -3
  4. paimon-0.2.2/README.zh-CN.md +104 -0
  5. {paimon-0.2.0 → paimon-0.2.2}/paimon/agent.py +191 -48
  6. paimon-0.2.2/paimon/app.py +538 -0
  7. {paimon-0.2.0 → paimon-0.2.2}/paimon/app.tcss +30 -4
  8. {paimon-0.2.0 → paimon-0.2.2}/paimon/cli.py +8 -5
  9. {paimon-0.2.0 → paimon-0.2.2}/paimon/commands.py +10 -4
  10. {paimon-0.2.0 → paimon-0.2.2}/paimon/compaction.py +57 -16
  11. {paimon-0.2.0 → paimon-0.2.2}/paimon/config.py +12 -10
  12. {paimon-0.2.0 → paimon-0.2.2}/paimon/headless.py +15 -1
  13. paimon-0.2.2/paimon/jobs.py +433 -0
  14. {paimon-0.2.0 → paimon-0.2.2}/paimon/llm.py +13 -0
  15. paimon-0.2.2/paimon/model_windows.py +2852 -0
  16. paimon-0.2.2/paimon/pane.py +877 -0
  17. {paimon-0.2.0 → paimon-0.2.2}/paimon/retry.py +10 -2
  18. {paimon-0.2.0 → paimon-0.2.2}/paimon/session.py +82 -13
  19. {paimon-0.2.0 → paimon-0.2.2}/paimon/skill/SKILL.md +6 -6
  20. paimon-0.2.2/paimon/supervisor.py +346 -0
  21. paimon-0.2.2/paimon/tabs.py +194 -0
  22. paimon-0.2.2/paimon/taskpane.py +179 -0
  23. paimon-0.2.2/paimon/tools.py +1462 -0
  24. {paimon-0.2.0 → paimon-0.2.2}/paimon/ui.py +18 -4
  25. {paimon-0.2.0 → paimon-0.2.2}/paimon.egg-info/PKG-INFO +10 -4
  26. {paimon-0.2.0 → paimon-0.2.2}/paimon.egg-info/SOURCES.txt +11 -16
  27. paimon-0.2.2/paimon.egg-info/scm_file_list.json +64 -0
  28. paimon-0.2.2/paimon.egg-info/scm_version.json +8 -0
  29. {paimon-0.2.0 → paimon-0.2.2}/pyproject.toml +4 -2
  30. paimon-0.2.0/paimon/app.py +0 -733
  31. paimon-0.2.0/paimon/tools.py +0 -727
  32. paimon-0.2.0/tests/test_agent.py +0 -420
  33. paimon-0.2.0/tests/test_app.py +0 -600
  34. paimon-0.2.0/tests/test_cli.py +0 -424
  35. paimon-0.2.0/tests/test_commands.py +0 -369
  36. paimon-0.2.0/tests/test_compaction.py +0 -120
  37. paimon-0.2.0/tests/test_config.py +0 -76
  38. paimon-0.2.0/tests/test_diff.py +0 -38
  39. paimon-0.2.0/tests/test_headless.py +0 -372
  40. paimon-0.2.0/tests/test_llm.py +0 -69
  41. paimon-0.2.0/tests/test_lockfile.py +0 -66
  42. paimon-0.2.0/tests/test_mentions.py +0 -94
  43. paimon-0.2.0/tests/test_reasoning.py +0 -157
  44. paimon-0.2.0/tests/test_retry.py +0 -147
  45. paimon-0.2.0/tests/test_session.py +0 -145
  46. paimon-0.2.0/tests/test_tools.py +0 -304
  47. {paimon-0.2.0 → paimon-0.2.2}/LICENSE +0 -0
  48. {paimon-0.2.0 → paimon-0.2.2}/paimon/__init__.py +0 -0
  49. {paimon-0.2.0 → paimon-0.2.2}/paimon/__main__.py +0 -0
  50. {paimon-0.2.0 → paimon-0.2.2}/paimon/diff.py +0 -0
  51. {paimon-0.2.0 → paimon-0.2.2}/paimon/lockfile.py +0 -0
  52. {paimon-0.2.0 → paimon-0.2.2}/paimon/login.py +0 -0
  53. {paimon-0.2.0 → paimon-0.2.2}/paimon/mentions.py +0 -0
  54. {paimon-0.2.0 → paimon-0.2.2}/paimon/prompt.py +0 -0
  55. {paimon-0.2.0 → paimon-0.2.2}/paimon.egg-info/dependency_links.txt +0 -0
  56. {paimon-0.2.0 → paimon-0.2.2}/paimon.egg-info/entry_points.txt +0 -0
  57. {paimon-0.2.0 → paimon-0.2.2}/paimon.egg-info/requires.txt +0 -0
  58. {paimon-0.2.0 → paimon-0.2.2}/paimon.egg-info/top_level.txt +0 -0
  59. {paimon-0.2.0 → paimon-0.2.2}/setup.cfg +0 -0
@@ -0,0 +1,7 @@
1
+ prune tests
2
+ prune docs
3
+ prune scripts
4
+ prune .github
5
+ exclude .gitignore
6
+ exclude .python-version
7
+ exclude uv.lock
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: paimon
3
- Version: 0.2.0
3
+ Version: 0.2.2
4
4
  Summary: A minimal code agent built on pydantic-ai + textual
5
5
  Requires-Python: >=3.10
6
6
  Description-Content-Type: text/markdown
@@ -38,7 +38,13 @@ uvx paimon
38
38
 
39
39
  The first launch asks for a provider, model, API base and key, and saves them to `~/.config/paimon/default/config.json`. Then just type what you want done.
40
40
 
41
- While it runs: `Shift+Tab` switches how much the agent may do on its own (**read**: ask before writing files or running commands, except clearly read-only ones like `ls` or `git status`, which run without asking, **edit**: edits inside the working directory go through, **yolo**: never ask), `Esc` interrupts the current turn, `Ctrl+P` opens the command palette (switch provider or profile, new, fork or resume session, show the model's thinking, compact the context), `Ctrl+C` quits.
41
+ While it runs: `Shift+Tab` switches how much the agent may do on its own (**read**: ask before writing files or running commands, except clearly read-only ones like `ls` or `git status`, which run without asking, **edit**: edits inside the working directory go through, **yolo**: never ask; the default), `Esc` interrupts the current turn, `Ctrl+P` opens the command palette (switch provider or profile, new, fork or resume session, show the model's thinking, compact the context), `Ctrl+C` quits.
42
+
43
+ `Ctrl+T` opens another session in a pane of its own, `Ctrl+W` closes one, `Ctrl+PageUp` and `Ctrl+PageDown` move between them, and `Ctrl+G` jumps to a pane waiting for permission.
44
+
45
+ Paimon can open panes itself: ask for two independent things and it starts a second agent in its own tab, with the same tools, working directory and permission mode. Its permission prompts appear in that tab, so `Ctrl+G` is how you unblock it. Those sessions belong to the one that started them, so they stay out of `paimon sessions` and end when it does.
46
+
47
+ It can also leave a command running in a tab of its own, for a dev server, a watcher or a long build that would otherwise hold up a turn. It asks first, every time, whatever the mode says about read-only commands. The tab streams the output and stops the command when you close it or quit. Programs that buffer their output when it is not going to a terminal print in blocks there rather than line by line; that is what a pipe costs, and Paimon does not emulate a terminal.
42
48
 
43
49
  Write `@path/to/file` in a prompt to hand a file to the agent.
44
50
 
@@ -76,7 +82,7 @@ paimon log a1b2c3 # what a session did, one line per event
76
82
  ## Other ways to run it
77
83
 
78
84
  ```bash
79
- paimon --mode edit # start in a less cautious permission mode
85
+ paimon --mode read # start in a more cautious permission mode (yolo is the default)
80
86
  paimon --strict # ask before every command, even read-only ones
81
87
  paimon --web # the same UI in a browser (--port, default 8000)
82
88
  paimon -p "what does cli.py do?" # one answer on stdout, no UI
@@ -85,7 +91,7 @@ paimon --model zai:glm-4.7 # this model for this run only
85
91
  paimon --profile work # a separately configured account
86
92
  ```
87
93
 
88
- `-p` never stops to ask, so anything the current mode would prompt for is refused instead (recognized read-only commands still run in read mode); pass `--mode edit` or `--mode yolo` if the run needs to change files. Add `--output-format result` for a single JSON object with the outcome (or `json` for one event per line), and `--timeout`/`--max-tool-calls` to bound an unattended run.
94
+ `-p` never stops to ask, and the default mode is `yolo`, so it can already write files and run commands; pass `--mode read` or `--mode edit` to keep the guardrails, in which case anything the mode would prompt for is refused instead (recognized read-only commands still run in read mode). Add `--output-format result` for a single JSON object with the outcome (or `json` for one event per line), and `--timeout`/`--max-tool-calls` to bound an unattended run.
89
95
 
90
96
  ## Configuration
91
97
 
@@ -26,7 +26,13 @@ uvx paimon
26
26
 
27
27
  The first launch asks for a provider, model, API base and key, and saves them to `~/.config/paimon/default/config.json`. Then just type what you want done.
28
28
 
29
- While it runs: `Shift+Tab` switches how much the agent may do on its own (**read**: ask before writing files or running commands, except clearly read-only ones like `ls` or `git status`, which run without asking, **edit**: edits inside the working directory go through, **yolo**: never ask), `Esc` interrupts the current turn, `Ctrl+P` opens the command palette (switch provider or profile, new, fork or resume session, show the model's thinking, compact the context), `Ctrl+C` quits.
29
+ While it runs: `Shift+Tab` switches how much the agent may do on its own (**read**: ask before writing files or running commands, except clearly read-only ones like `ls` or `git status`, which run without asking, **edit**: edits inside the working directory go through, **yolo**: never ask; the default), `Esc` interrupts the current turn, `Ctrl+P` opens the command palette (switch provider or profile, new, fork or resume session, show the model's thinking, compact the context), `Ctrl+C` quits.
30
+
31
+ `Ctrl+T` opens another session in a pane of its own, `Ctrl+W` closes one, `Ctrl+PageUp` and `Ctrl+PageDown` move between them, and `Ctrl+G` jumps to a pane waiting for permission.
32
+
33
+ Paimon can open panes itself: ask for two independent things and it starts a second agent in its own tab, with the same tools, working directory and permission mode. Its permission prompts appear in that tab, so `Ctrl+G` is how you unblock it. Those sessions belong to the one that started them, so they stay out of `paimon sessions` and end when it does.
34
+
35
+ It can also leave a command running in a tab of its own, for a dev server, a watcher or a long build that would otherwise hold up a turn. It asks first, every time, whatever the mode says about read-only commands. The tab streams the output and stops the command when you close it or quit. Programs that buffer their output when it is not going to a terminal print in blocks there rather than line by line; that is what a pipe costs, and Paimon does not emulate a terminal.
30
36
 
31
37
  Write `@path/to/file` in a prompt to hand a file to the agent.
32
38
 
@@ -64,7 +70,7 @@ paimon log a1b2c3 # what a session did, one line per event
64
70
  ## Other ways to run it
65
71
 
66
72
  ```bash
67
- paimon --mode edit # start in a less cautious permission mode
73
+ paimon --mode read # start in a more cautious permission mode (yolo is the default)
68
74
  paimon --strict # ask before every command, even read-only ones
69
75
  paimon --web # the same UI in a browser (--port, default 8000)
70
76
  paimon -p "what does cli.py do?" # one answer on stdout, no UI
@@ -73,7 +79,7 @@ paimon --model zai:glm-4.7 # this model for this run only
73
79
  paimon --profile work # a separately configured account
74
80
  ```
75
81
 
76
- `-p` never stops to ask, so anything the current mode would prompt for is refused instead (recognized read-only commands still run in read mode); pass `--mode edit` or `--mode yolo` if the run needs to change files. Add `--output-format result` for a single JSON object with the outcome (or `json` for one event per line), and `--timeout`/`--max-tool-calls` to bound an unattended run.
82
+ `-p` never stops to ask, and the default mode is `yolo`, so it can already write files and run commands; pass `--mode read` or `--mode edit` to keep the guardrails, in which case anything the mode would prompt for is refused instead (recognized read-only commands still run in read mode). Add `--output-format result` for a single JSON object with the outcome (or `json` for one event per line), and `--timeout`/`--max-tool-calls` to bound an unattended run.
77
83
 
78
84
  ## Configuration
79
85
 
@@ -0,0 +1,104 @@
1
+ # Paimon
2
+
3
+ ![Paimon](https://automaton-media.com/wp-content/uploads/2020/10/20201019-140524-header.jpg)
4
+
5
+ [English](README.md) | 简体中文
6
+
7
+ Paimon 是一个终端里的 coding agent。它读写当前目录下的文件、执行命令,在做任何改动前会先询问。它也支持无头运行,可以被更强的 agent 作为执行者调用。
8
+
9
+ ## 安装
10
+
11
+ ```bash
12
+ uv tool install paimon # 或者:pip install paimon
13
+ ```
14
+
15
+ ## 快速开始
16
+
17
+ ```bash
18
+ paimon
19
+ ```
20
+
21
+ 或者不安装直接运行:
22
+
23
+ ```bash
24
+ uvx paimon
25
+ ```
26
+
27
+ 首次启动会询问 provider、模型、API base 和 key,并保存到 `~/.config/paimon/default/config.json`。之后输入要完成的任务即可。
28
+
29
+ 运行时:`Shift+Tab` 切换 agent 的自主程度(**read**:写文件或执行命令前先询问,`ls`、`git status` 这类明确只读的命令除外,会直接执行,**edit**:工作目录内的编辑直接执行,**yolo**:从不询问,也是默认值),`Esc` 打断当前回合,`Ctrl+P` 打开命令面板(切换 provider 或 profile、新建、分叉或恢复会话、显示模型思考、压缩上下文),`Ctrl+C` 退出。
30
+
31
+ `Ctrl+T` 在新 pane 里打开另一个会话,`Ctrl+W` 关闭当前 pane,`Ctrl+PageUp` 和 `Ctrl+PageDown` 在 pane 之间切换,`Ctrl+G` 跳到正在等待授权的 pane。
32
+
33
+ Paimon 自己也能开 pane:让它同时做两件互不相干的事,它会在新 tab 里起第二个 agent,工具、工作目录和权限模式都和当前会话一样。它的授权确认弹在它自己的 tab 里,用 `Ctrl+G` 过去处理。这些会话属于开它们的那个会话,不会出现在 `paimon sessions` 里,也随它一起结束。
34
+
35
+ 它也能把一条命令留在单独的 tab 里跑,比如开发服务器、文件监视或者很长的构建,不占着当前回合。这类命令一律先确认,不管当前模式对只读命令怎么规定。tab 里流式显示输出,关掉 tab 或退出时命令随之停止。输出不是终端时很多程序会按块缓冲,所以 tab 里可能一阵子不出东西再一次性出来:这是用管道代替终端的代价,Paimon 不做终端模拟。
36
+
37
+ 在提示中写 `@path/to/file` 可以把文件提供给 agent。
38
+
39
+ ## 当作 subagent 使用
40
+
41
+ 前沿模型擅长制定计划和验收结果,中间的执行步骤往往比较机械。让 Paimon 使用成本较低的模型执行,由 Claude Code 或 Codex 制定计划并检查结果,只在必要的环节为前沿模型付费。用一个 profile 单独保存该模型的账号:
42
+
43
+ ```bash
44
+ paimon login --profile glm --model zai:glm-4.7 --api-key-env ZAI_API_KEY
45
+ paimon --profile glm -p "apply the plan in PLAN.md" --mode edit --output-format result
46
+ ```
47
+
48
+ 自带的 skill 会向调用方 agent 说明这套流程(先用 `paimon status --json` 检查、单次运行、读取唯一一行 result 对象、用其中的 `session_id` 续跑、用 `paimon log` 查看运行过程):
49
+
50
+ ```bash
51
+ paimon install-skill # 安装到 Claude Code(~/.claude/skills/paimon)
52
+ paimon install-skill --target codex # 安装到 Codex;--dest DIR 安装到任意目录
53
+ npx skills add aisk/paimon # 通过 skills.sh 安装同一个 skill
54
+ ```
55
+
56
+ ## 会话
57
+
58
+ 每次对话都会保存。退出时 Paimon 会打印恢复该会话的命令:
59
+
60
+ ```bash
61
+ paimon -r # 从当前目录的会话中选择
62
+ paimon -r a1b2c3 # 按 id 恢复
63
+ paimon -c # 恢复最近一个会话
64
+ paimon sessions # 列出会话(--json 输出机器可读格式)
65
+ paimon log a1b2c3 # 查看会话做了什么,每个事件一行
66
+ ```
67
+
68
+ `paimon log` 的每行输出都带一个稳定的序号;`--after SEQ`、`--turns N`、`--tail N` 缩小范围,`--json` 和 `--full` 输出原始记录。
69
+
70
+ ## 其他运行方式
71
+
72
+ ```bash
73
+ paimon --mode read # 以更谨慎的权限模式启动(默认为 yolo)
74
+ paimon --strict # 每条命令都先询问,包括只读命令
75
+ paimon --web # 在浏览器中使用同一套 UI(--port,默认 8000)
76
+ paimon -p "what does cli.py do?" # 直接在 stdout 输出回答,不启动 UI
77
+ cat log.txt | paimon -p "summarize this"
78
+ paimon --model zai:glm-4.7 # 仅本次运行使用该模型
79
+ paimon --profile work # 单独配置的另一个账号
80
+ ```
81
+
82
+ `-p` 不会停下来询问,默认模式是 `yolo`,因此已经可以修改文件和执行命令;传 `--mode read` 或 `--mode edit` 可以保留确认护栏,此时当前模式需要确认的操作会被直接拒绝(识别为只读的命令在 read 模式下仍会执行)。加 `--output-format result` 输出一个包含结果的 JSON 对象(`json` 则每行输出一个事件),用 `--timeout`/`--max-tool-calls` 为无人值守的运行设置上限。
83
+
84
+ ## 配置
85
+
86
+ `~/.config/paimon/<name>/config.json` 保存每个 profile 的模型设置(不传 `--profile` 时为 `default`)。两个可选配置项可以改变它的行为:自动放行只读命令,以及在接近上下文上限时原地总结长对话。
87
+
88
+ ```json
89
+ {
90
+ "safe_commands": false,
91
+ "compaction": {
92
+ "enabled": true,
93
+ "context_window": 128000,
94
+ "reserve_tokens": 16384,
95
+ "keep_recent_tokens": 20000
96
+ }
97
+ }
98
+ ```
99
+
100
+ `safe_commands`(默认 `true`)允许 read 和 edit 模式不经询问执行一小组固定的、明确只读的命令(`ls`、`cat`、`git status` 等);`--strict` 可在单次运行中关闭它。被识别的命令可以用 `&&`、`;` 或管道串联,也包括 `cd 目录 && …`(目录须留在工作目录内,且整条命令都以 `&&` 连接)。重定向、`$()`/反引号替换和后台 `&` 仍会询问。
101
+
102
+ **这是防止 agent 失误的护栏,不是安全边界。** 被识别的命令仍然通过 `PATH` 查找,仍可能顺着符号链接读到工作目录之外;而且哪怕是纯读取,也会把文件内容带进模型上下文,只读不等于保密安全。需要真正的隔离时,请在容器或虚拟机中运行 Paimon。
103
+
104
+ 会话存放在 `~/.local/share/paimon/sessions/`(`PAIMON_DATA_HOME` 可覆盖)。安装 [delta](https://github.com/dandavison/delta) 后文件改动的展示效果更好。
@@ -32,10 +32,20 @@ from pydantic_ai.models import Model, ModelRequestParameters
32
32
 
33
33
  from . import compaction, retry, tools
34
34
  from .config import Config
35
- from .llm import build_model
35
+ from .llm import build_model, user_agent
36
36
  from .mentions import expand_mentions
37
37
  from .prompt import build_system_prompt
38
- from .session import Session, SessionIncompleteError, is_summary_message
38
+ from .session import (
39
+ Session,
40
+ SessionIncompleteError,
41
+ agents_message,
42
+ agents_text,
43
+ is_agents_message,
44
+ is_summary_message,
45
+ )
46
+
47
+ # Transient compaction failures tolerated in one turn before it is left off.
48
+ _MAX_COMPACTION_FAILURES = 3
39
49
 
40
50
 
41
51
  # ---- Events yielded by Agent.run -------------------------------------------
@@ -128,12 +138,24 @@ class CompactionNotice:
128
138
  """A compaction checkpoint encountered while replaying history."""
129
139
 
130
140
 
141
+ @dataclass
142
+ class AgentsNotice:
143
+ """A status line about the agents this session started.
144
+
145
+ Not replay-only: it is written into the history at the top of a turn, so
146
+ it is both yielded live and rebuilt when the session is resumed.
147
+ """
148
+
149
+ text: str
150
+
151
+
131
152
  # Everything ``Agent.run`` and ``replay_events`` can yield. Renderers dispatch
132
153
  # on isinstance; the alias exists so a type checker can flag an unhandled one.
133
154
  AgentEvent = (
134
155
  TextDelta | ReasoningDelta | ToolStart | ToolEnd | TodosUpdate
135
156
  | SessionHandoff | RequestStats | TurnEnd | ContextCompacted
136
157
  | ContextCompactionFailed | ModelRetry | UserInput | CompactionNotice
158
+ | AgentsNotice
137
159
  )
138
160
 
139
161
 
@@ -156,15 +178,22 @@ def replay_events(messages: list[ModelMessage]) -> list[AgentEvent]:
156
178
  Lets a UI render resumed history through the same code path as live turns.
157
179
  """
158
180
  events: list[AgentEvent] = []
181
+ # write_todos calls whose arguments never made a todo update: they were
182
+ # rejected live as a normal tool error, so they replay as one.
183
+ rejected_todos: set[str] = set()
159
184
  for message in messages:
160
185
  if is_summary_message(message):
161
186
  events.append(CompactionNotice())
162
187
  continue
188
+ if is_agents_message(message):
189
+ events.append(AgentsNotice(agents_text(message)))
190
+ continue
163
191
  if isinstance(message, ModelRequest):
164
192
  for part in message.parts:
165
193
  if isinstance(part, UserPromptPart) and isinstance(part.content, str) and part.content:
166
194
  events.append(UserInput(part.content))
167
- elif isinstance(part, ToolReturnPart) and part.tool_name != "write_todos":
195
+ elif isinstance(part, ToolReturnPart) and (part.tool_name != "write_todos"
196
+ or part.tool_call_id in rejected_todos):
168
197
  events.append(ToolEnd(part.tool_call_id, part.tool_name,
169
198
  str(part.content or "(no output)")))
170
199
  elif isinstance(message, ModelResponse):
@@ -175,9 +204,13 @@ def replay_events(messages: list[ModelMessage]) -> list[AgentEvent]:
175
204
  events.append(TextDelta(part.content))
176
205
  elif isinstance(part, ToolCallPart):
177
206
  args = _parse_args(part.args)
178
- if part.tool_name == "write_todos":
179
- events.append(TodosUpdate(args.get("todos") or []))
207
+ todos = (tools.normalize_todos(args.get("todos"))
208
+ if part.tool_name == "write_todos" else None)
209
+ if todos is not None:
210
+ events.append(TodosUpdate(todos))
180
211
  else:
212
+ if part.tool_name == "write_todos":
213
+ rejected_todos.add(part.tool_call_id)
181
214
  events.append(ToolStart(part.tool_call_id, part.tool_name, args))
182
215
  return events
183
216
 
@@ -215,13 +248,26 @@ class Agent:
215
248
  """
216
249
 
217
250
  def __init__(self, session: Session, system_prompt: str, *, cwd: Optional[Path] = None,
218
- confirm: Optional[ConfirmFn] = None, mode: str = "read",
251
+ confirm: Optional[ConfirmFn] = None, mode: str = "yolo",
219
252
  config: Optional[Config] = None,
220
- toolset: Optional[dict[str, tools.Tool]] = None):
253
+ toolset: Optional[dict[str, tools.Tool]] = None,
254
+ model_override: Optional[str] = None):
221
255
  self.cwd = Path(cwd or Path.cwd())
222
256
  self.confirm = confirm
223
257
  self.mode = mode
224
258
  self.config = config or Config.load()
259
+ # Per-agent model choice. One Config instance is shared by every agent
260
+ # in the process, so writing config.model would repoint all of them;
261
+ # this overrides the model for this agent alone, credentials included
262
+ # (they come from the config either way).
263
+ self.model_override = model_override
264
+ # Set by the UI when this agent may start and talk to other agents.
265
+ # None everywhere else (headless, tests), where the agent tools refuse
266
+ # rather than pretend.
267
+ self.supervisor = None
268
+ # Per-agent tool state, kept off the tool functions so one agent's
269
+ # shell overflow files stay invisible to the next one.
270
+ self.tool_context = tools.ToolContext()
225
271
  self.todos: list[dict] = []
226
272
  self.session = session
227
273
  self.system_prompt = system_prompt
@@ -234,44 +280,62 @@ class Agent:
234
280
 
235
281
  @classmethod
236
282
  def open(cls, cwd: Optional[Path] = None, *, session: Optional[Session] = None,
237
- confirm: Optional[ConfirmFn] = None, mode: str = "read",
283
+ confirm: Optional[ConfirmFn] = None, mode: str = "yolo",
238
284
  config: Optional[Config] = None,
239
285
  append_system_prompt: Optional[str] = None,
240
- toolset: Optional[dict[str, tools.Tool]] = None) -> "Agent":
286
+ toolset: Optional[dict[str, tools.Tool]] = None,
287
+ model_override: Optional[str] = None,
288
+ parent: Optional[str] = None) -> "Agent":
241
289
  """Start a new session, or resume ``session``, and take its lock.
242
290
 
243
291
  ``append_system_prompt`` is added to the end of a new session's system
244
292
  prompt and persisted with it, so a resumed session keeps it. Resuming
245
293
  with it set raises ``ValueError``: the persisted prompt is immutable.
294
+ ``parent`` marks the new session as a subagent's, which keeps it out of
295
+ the session listings its parent shows up in.
246
296
 
247
- Raises ``SessionBusyError`` when another process holds the session and
248
- ``SessionIncompleteError`` when a resumed log has no system prompt
249
- snapshot — both ``SessionError``, and neither leaves a lock held.
297
+ Raises ``SessionBusyError`` when the session is already open (here or
298
+ in another process) and ``SessionIncompleteError`` when a resumed log
299
+ has no system prompt snapshot — both ``SessionError``, and neither
300
+ leaves a lock held.
250
301
  """
251
302
  cwd = Path(cwd or Path.cwd())
252
303
  if session is not None and append_system_prompt:
253
304
  raise ValueError("append_system_prompt only applies to a new session")
305
+ is_new = session is None
254
306
  if session is None:
255
- session = Session.create(cwd)
256
- session.lock()
257
- system_prompt = build_system_prompt(cwd)
258
- if append_system_prompt:
259
- system_prompt += f"\n\n{append_system_prompt.strip()}"
260
- session.append_system_prompt(system_prompt)
261
- else:
262
- session.lock()
263
- system_prompt = session.system_prompt()
264
- if system_prompt is None:
265
- session.unlock()
266
- raise SessionIncompleteError("Session does not contain a persisted system prompt")
267
- return cls(session, system_prompt, cwd=cwd, confirm=confirm, mode=mode,
268
- config=config, toolset=toolset)
307
+ session = Session.create(cwd, parent)
308
+ session.lock()
309
+ # Everything after the lock can fail — a full disk while writing the
310
+ # prompt, a message the current pydantic-ai cannot parse — and no
311
+ # Agent is returned to unlock it, so the lock is released here.
312
+ try:
313
+ if is_new:
314
+ system_prompt = build_system_prompt(cwd)
315
+ if append_system_prompt:
316
+ system_prompt += f"\n\n{append_system_prompt.strip()}"
317
+ session.append_system_prompt(system_prompt)
318
+ else:
319
+ system_prompt = session.system_prompt()
320
+ if system_prompt is None:
321
+ raise SessionIncompleteError("Session does not contain a persisted system prompt")
322
+ return cls(session, system_prompt, cwd=cwd, confirm=confirm, mode=mode,
323
+ config=config, toolset=toolset, model_override=model_override)
324
+ except BaseException:
325
+ session.unlock()
326
+ raise
327
+
328
+ @property
329
+ def model_name(self) -> Optional[str]:
330
+ """The model this agent talks to: its own override, else the config's."""
331
+ return self.model_override or self.config.model
269
332
 
270
333
  def _model(self) -> Model:
271
334
  """The configured model, rebuilt when login changes the config."""
272
- if not self.config.model:
335
+ name = self.model_name
336
+ if not name:
273
337
  raise RuntimeError("No model configured; log in first")
274
- key = (self.config.model, self.config.api_base, self.config.api_key)
338
+ key = (name, self.config.api_base, self.config.api_key)
275
339
  if self._cached_model is None or self._cached_model[0] != key:
276
340
  self._cached_model = (key, build_model(*key))
277
341
  return self._cached_model[1]
@@ -299,13 +363,17 @@ class Agent:
299
363
  if not force:
300
364
  if not self.config.compaction_enabled:
301
365
  return None
302
- window = compaction.context_window(self.config.model,
366
+ window = compaction.context_window(self.model_name,
303
367
  self.config.compaction_context_window)
304
- tokens_before = self.count_context_tokens()
368
+ # Nothing to compare against, so counting would be wasted work: an
369
+ # unknown window disables auto-compaction outright.
370
+ if window is None:
371
+ return None
372
+ tokens_before = await self.count_context_tokens()
305
373
  if not compaction.should_compact(tokens_before, window, self.config.compaction_reserve_tokens):
306
374
  return None
307
375
  else:
308
- tokens_before = self.count_context_tokens()
376
+ tokens_before = await self.count_context_tokens()
309
377
 
310
378
  result = await compaction.compact(
311
379
  self.history,
@@ -322,12 +390,19 @@ class Agent:
322
390
  # append-message invariant (see _append_message) still holds afterwards.
323
391
  self.session.append_compaction(result.summary, result.kept_messages, result.tokens_before)
324
392
  self.history = result.messages
325
- result.tokens_after = self.count_context_tokens()
393
+ result.tokens_after = await self.count_context_tokens()
326
394
  return result
327
395
 
328
- def count_context_tokens(self) -> int:
329
- """Estimate the tokens of everything the next request would send."""
330
- return compaction.count_tokens(self.history, self.tool_schemas, self.system_prompt)
396
+ async def count_context_tokens(self) -> int:
397
+ """Estimate the tokens of everything the next request would send.
398
+
399
+ Off the event loop: the count serializes the whole history, which is
400
+ hundreds of kilobytes on a long session. It runs at the top of every
401
+ model step, and every agent shares one loop, so counting inline stalls
402
+ every other agent's streaming output for as long as it takes.
403
+ """
404
+ return await asyncio.to_thread(
405
+ compaction.count_tokens, list(self.history), self.tool_schemas, self.system_prompt)
331
406
 
332
407
  async def compact_now(self) -> Optional[compaction.CompactionResult]:
333
408
  """Compact on demand; None when the history is too short to be worth it."""
@@ -343,8 +418,17 @@ class Agent:
343
418
  persist: Callable[[], None]) -> AsyncIterator[AgentEvent]:
344
419
  # Reports itself as TodosUpdate alone — no ToolStart/ToolEnd — which is
345
420
  # also the shape replay_events produces for it, so no renderer has to
346
- # special-case the name.
347
- self.todos = args.get("todos") or []
421
+ # special-case the name. A malformed argument is the exception: it
422
+ # renders as a normal failed tool call, live and on replay alike.
423
+ todos = tools.normalize_todos(args.get("todos"))
424
+ if todos is None:
425
+ yield ToolStart(call.tool_call_id, call.tool_name, args)
426
+ slot.content = ("Error: 'todos' must be an array of "
427
+ "{content, status} objects.")
428
+ persist()
429
+ yield ToolEnd(call.tool_call_id, call.tool_name, slot.content)
430
+ return
431
+ self.todos = todos
348
432
  slot.content = tools.render_todos(self.todos)
349
433
  persist()
350
434
  yield TodosUpdate(list(self.todos))
@@ -362,12 +446,7 @@ class Agent:
362
446
  persist()
363
447
  yield ToolEnd(call.tool_call_id, call.tool_name, slot.content)
364
448
  return
365
- needs_confirm = tools.gate(call.tool_name, args, self.mode, self.cwd,
366
- self.toolset) == "confirm"
367
- allowed = not needs_confirm or (
368
- await self.confirm(call.tool_name, args) if self.confirm else False
369
- )
370
- if not allowed:
449
+ if not await self._permitted(call.tool_name, args):
371
450
  slot.content = "User denied this operation."
372
451
  persist()
373
452
  yield ToolEnd(call.tool_call_id, call.tool_name, slot.content, denied=True)
@@ -378,9 +457,52 @@ class Agent:
378
457
  yield ToolEnd(call.tool_call_id, call.tool_name, slot.content)
379
458
  yield SessionHandoff(prompt_text)
380
459
 
460
+ async def _permitted(self, name: str, args: dict) -> bool:
461
+ """Gate a tool the loop runs itself, the way run_tool gates the rest.
462
+
463
+ The enforcement point for everything in _AGENT_HANDLED: without a
464
+ confirm hook a call that needs one is denied, so a headless agent
465
+ cannot walk around the permission mode here either.
466
+ """
467
+ if tools.gate(name, args, self.mode, self.cwd, self.toolset,
468
+ safe_commands=self.config.safe_commands,
469
+ ctx=self.tool_context) != "confirm":
470
+ return True
471
+ return await self.confirm(name, args) if self.confirm else False
472
+
473
+ async def _run_supervised(self, call: ToolCallPart, args: dict, slot: ToolReturnPart,
474
+ persist: Callable[[], None]) -> AsyncIterator[AgentEvent]:
475
+ """The job tools: starting agents and commands, and reporting on both.
476
+
477
+ They act on the pool of jobs the UI is running, which no stateless tool
478
+ function can reach, and the supervisor is the one thing that knows
479
+ whether a given job is busy \u2014 so they are dispatched from here. Most of
480
+ them are not gated at all (they only reach jobs this same agent
481
+ started); run_background is, because it starts a process.
482
+ """
483
+ yield ToolStart(call.tool_call_id, call.tool_name, args)
484
+ if self.supervisor is None:
485
+ slot.content = ("Error: this only works in the interactive UI; "
486
+ "do the work yourself instead.")
487
+ elif not await self._permitted(call.tool_name, args):
488
+ slot.content = "User denied this operation."
489
+ persist()
490
+ yield ToolEnd(call.tool_call_id, call.tool_name, slot.content, denied=True)
491
+ return
492
+ else:
493
+ slot.content = await self.supervisor.handle(call.tool_name, args, caller=self)
494
+ persist()
495
+ yield ToolEnd(call.tool_call_id, call.tool_name, slot.content)
496
+
381
497
  _AGENT_HANDLED = {
382
498
  "write_todos": _run_write_todos,
383
499
  "start_new_session": _run_start_new_session,
500
+ "spawn_agent": _run_supervised,
501
+ "send_to_agent": _run_supervised,
502
+ "run_background": _run_supervised,
503
+ "read_job": _run_supervised,
504
+ "wait_for_job": _run_supervised,
505
+ "stop_job": _run_supervised,
384
506
  }
385
507
 
386
508
  async def run(self, user_input: str, *, expand: bool = True) -> AsyncIterator[AgentEvent]:
@@ -391,17 +513,34 @@ class Agent:
391
513
  stdin, where a line like ``@foo.py`` is data rather than a mention).
392
514
  """
393
515
  prompt = expand_mentions(user_input, self.cwd) if expand else user_input
516
+ # Agents this session started report in here, at the top of the next
517
+ # turn, rather than by interrupting whatever the user is typing. It has
518
+ # to be a persisted message: an event the model never sees would defeat
519
+ # the point, which is to get it to call read_job.
520
+ summary = self.supervisor.status_summary(self) if self.supervisor is not None else None
521
+ if summary:
522
+ self._append_message(agents_message(summary))
523
+ yield AgentsNotice(summary)
394
524
  self._append_message(ModelRequest(parts=[UserPromptPart(content=prompt)]))
395
- compaction_failed = False
525
+ # A compaction that failed for a transient reason is retried on the
526
+ # next step of this turn — the context only keeps growing, so giving
527
+ # up on the first rate limit disables the safety net exactly when it
528
+ # is needed. Anything else, and repeated transient failures, stop it
529
+ # for the rest of the turn instead of paying for it every step.
530
+ compaction_off = False
531
+ compaction_failures = 0
396
532
 
397
533
  while True:
398
- if not compaction_failed:
534
+ if not compaction_off:
399
535
  try:
400
536
  compacted = await self._maybe_compact()
401
537
  except Exception as exc: # noqa: BLE001 - the normal request may still fit
402
- compaction_failed = True
538
+ compaction_failures += 1
539
+ compaction_off = (not retry.is_transient(exc)
540
+ or compaction_failures >= _MAX_COMPACTION_FAILURES)
403
541
  yield ContextCompactionFailed(str(exc))
404
542
  else:
543
+ compaction_failures = 0
405
544
  if compacted:
406
545
  yield ContextCompacted(compacted.tokens_before, compacted.tokens_after)
407
546
 
@@ -423,7 +562,10 @@ class Agent:
423
562
  first_event_at: Optional[float] = None
424
563
  try:
425
564
  async with model_request_stream(
426
- model, request_messages, model_request_parameters=parameters
565
+ model,
566
+ request_messages,
567
+ model_settings={"extra_headers": {"User-Agent": user_agent()}},
568
+ model_request_parameters=parameters,
427
569
  ) as stream:
428
570
  async for event in stream:
429
571
  if first_event_at is None:
@@ -523,7 +665,8 @@ class Agent:
523
665
  yield ToolStart(call.tool_call_id, name, args)
524
666
  result, denied = await tools.run_tool(name, args, self.cwd, self.mode,
525
667
  self.confirm, self.toolset,
526
- safe_commands=self.config.safe_commands)
668
+ safe_commands=self.config.safe_commands,
669
+ ctx=self.tool_context)
527
670
 
528
671
  slot.content = result
529
672
  persist()