paimon 0.2.0__tar.gz → 0.2.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- paimon-0.2.2/MANIFEST.in +7 -0
- {paimon-0.2.0 → paimon-0.2.2}/PKG-INFO +10 -4
- {paimon-0.2.0 → paimon-0.2.2}/README.md +9 -3
- paimon-0.2.2/README.zh-CN.md +104 -0
- {paimon-0.2.0 → paimon-0.2.2}/paimon/agent.py +191 -48
- paimon-0.2.2/paimon/app.py +538 -0
- {paimon-0.2.0 → paimon-0.2.2}/paimon/app.tcss +30 -4
- {paimon-0.2.0 → paimon-0.2.2}/paimon/cli.py +8 -5
- {paimon-0.2.0 → paimon-0.2.2}/paimon/commands.py +10 -4
- {paimon-0.2.0 → paimon-0.2.2}/paimon/compaction.py +57 -16
- {paimon-0.2.0 → paimon-0.2.2}/paimon/config.py +12 -10
- {paimon-0.2.0 → paimon-0.2.2}/paimon/headless.py +15 -1
- paimon-0.2.2/paimon/jobs.py +433 -0
- {paimon-0.2.0 → paimon-0.2.2}/paimon/llm.py +13 -0
- paimon-0.2.2/paimon/model_windows.py +2852 -0
- paimon-0.2.2/paimon/pane.py +877 -0
- {paimon-0.2.0 → paimon-0.2.2}/paimon/retry.py +10 -2
- {paimon-0.2.0 → paimon-0.2.2}/paimon/session.py +82 -13
- {paimon-0.2.0 → paimon-0.2.2}/paimon/skill/SKILL.md +6 -6
- paimon-0.2.2/paimon/supervisor.py +346 -0
- paimon-0.2.2/paimon/tabs.py +194 -0
- paimon-0.2.2/paimon/taskpane.py +179 -0
- paimon-0.2.2/paimon/tools.py +1462 -0
- {paimon-0.2.0 → paimon-0.2.2}/paimon/ui.py +18 -4
- {paimon-0.2.0 → paimon-0.2.2}/paimon.egg-info/PKG-INFO +10 -4
- {paimon-0.2.0 → paimon-0.2.2}/paimon.egg-info/SOURCES.txt +11 -16
- paimon-0.2.2/paimon.egg-info/scm_file_list.json +64 -0
- paimon-0.2.2/paimon.egg-info/scm_version.json +8 -0
- {paimon-0.2.0 → paimon-0.2.2}/pyproject.toml +4 -2
- paimon-0.2.0/paimon/app.py +0 -733
- paimon-0.2.0/paimon/tools.py +0 -727
- paimon-0.2.0/tests/test_agent.py +0 -420
- paimon-0.2.0/tests/test_app.py +0 -600
- paimon-0.2.0/tests/test_cli.py +0 -424
- paimon-0.2.0/tests/test_commands.py +0 -369
- paimon-0.2.0/tests/test_compaction.py +0 -120
- paimon-0.2.0/tests/test_config.py +0 -76
- paimon-0.2.0/tests/test_diff.py +0 -38
- paimon-0.2.0/tests/test_headless.py +0 -372
- paimon-0.2.0/tests/test_llm.py +0 -69
- paimon-0.2.0/tests/test_lockfile.py +0 -66
- paimon-0.2.0/tests/test_mentions.py +0 -94
- paimon-0.2.0/tests/test_reasoning.py +0 -157
- paimon-0.2.0/tests/test_retry.py +0 -147
- paimon-0.2.0/tests/test_session.py +0 -145
- paimon-0.2.0/tests/test_tools.py +0 -304
- {paimon-0.2.0 → paimon-0.2.2}/LICENSE +0 -0
- {paimon-0.2.0 → paimon-0.2.2}/paimon/__init__.py +0 -0
- {paimon-0.2.0 → paimon-0.2.2}/paimon/__main__.py +0 -0
- {paimon-0.2.0 → paimon-0.2.2}/paimon/diff.py +0 -0
- {paimon-0.2.0 → paimon-0.2.2}/paimon/lockfile.py +0 -0
- {paimon-0.2.0 → paimon-0.2.2}/paimon/login.py +0 -0
- {paimon-0.2.0 → paimon-0.2.2}/paimon/mentions.py +0 -0
- {paimon-0.2.0 → paimon-0.2.2}/paimon/prompt.py +0 -0
- {paimon-0.2.0 → paimon-0.2.2}/paimon.egg-info/dependency_links.txt +0 -0
- {paimon-0.2.0 → paimon-0.2.2}/paimon.egg-info/entry_points.txt +0 -0
- {paimon-0.2.0 → paimon-0.2.2}/paimon.egg-info/requires.txt +0 -0
- {paimon-0.2.0 → paimon-0.2.2}/paimon.egg-info/top_level.txt +0 -0
- {paimon-0.2.0 → paimon-0.2.2}/setup.cfg +0 -0
paimon-0.2.2/MANIFEST.in
ADDED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: paimon
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.2
|
|
4
4
|
Summary: A minimal code agent built on pydantic-ai + textual
|
|
5
5
|
Requires-Python: >=3.10
|
|
6
6
|
Description-Content-Type: text/markdown
|
|
@@ -38,7 +38,13 @@ uvx paimon
|
|
|
38
38
|
|
|
39
39
|
The first launch asks for a provider, model, API base and key, and saves them to `~/.config/paimon/default/config.json`. Then just type what you want done.
|
|
40
40
|
|
|
41
|
-
While it runs: `Shift+Tab` switches how much the agent may do on its own (**read**: ask before writing files or running commands, except clearly read-only ones like `ls` or `git status`, which run without asking, **edit**: edits inside the working directory go through, **yolo**: never ask), `Esc` interrupts the current turn, `Ctrl+P` opens the command palette (switch provider or profile, new, fork or resume session, show the model's thinking, compact the context), `Ctrl+C` quits.
|
|
41
|
+
While it runs: `Shift+Tab` switches how much the agent may do on its own (**read**: ask before writing files or running commands, except clearly read-only ones like `ls` or `git status`, which run without asking, **edit**: edits inside the working directory go through, **yolo**: never ask; the default), `Esc` interrupts the current turn, `Ctrl+P` opens the command palette (switch provider or profile, new, fork or resume session, show the model's thinking, compact the context), `Ctrl+C` quits.
|
|
42
|
+
|
|
43
|
+
`Ctrl+T` opens another session in a pane of its own, `Ctrl+W` closes one, `Ctrl+PageUp` and `Ctrl+PageDown` move between them, and `Ctrl+G` jumps to a pane waiting for permission.
|
|
44
|
+
|
|
45
|
+
Paimon can open panes itself: ask for two independent things and it starts a second agent in its own tab, with the same tools, working directory and permission mode. Its permission prompts appear in that tab, so `Ctrl+G` is how you unblock it. Those sessions belong to the one that started them, so they stay out of `paimon sessions` and end when it does.
|
|
46
|
+
|
|
47
|
+
It can also leave a command running in a tab of its own, for a dev server, a watcher or a long build that would otherwise hold up a turn. It asks first, every time, whatever the mode says about read-only commands. The tab streams the output and stops the command when you close it or quit. Programs that buffer their output when it is not going to a terminal print in blocks there rather than line by line; that is what a pipe costs, and Paimon does not emulate a terminal.
|
|
42
48
|
|
|
43
49
|
Write `@path/to/file` in a prompt to hand a file to the agent.
|
|
44
50
|
|
|
@@ -76,7 +82,7 @@ paimon log a1b2c3 # what a session did, one line per event
|
|
|
76
82
|
## Other ways to run it
|
|
77
83
|
|
|
78
84
|
```bash
|
|
79
|
-
paimon --mode
|
|
85
|
+
paimon --mode read # start in a more cautious permission mode (yolo is the default)
|
|
80
86
|
paimon --strict # ask before every command, even read-only ones
|
|
81
87
|
paimon --web # the same UI in a browser (--port, default 8000)
|
|
82
88
|
paimon -p "what does cli.py do?" # one answer on stdout, no UI
|
|
@@ -85,7 +91,7 @@ paimon --model zai:glm-4.7 # this model for this run only
|
|
|
85
91
|
paimon --profile work # a separately configured account
|
|
86
92
|
```
|
|
87
93
|
|
|
88
|
-
`-p` never stops to ask,
|
|
94
|
+
`-p` never stops to ask, and the default mode is `yolo`, so it can already write files and run commands; pass `--mode read` or `--mode edit` to keep the guardrails, in which case anything the mode would prompt for is refused instead (recognized read-only commands still run in read mode). Add `--output-format result` for a single JSON object with the outcome (or `json` for one event per line), and `--timeout`/`--max-tool-calls` to bound an unattended run.
|
|
89
95
|
|
|
90
96
|
## Configuration
|
|
91
97
|
|
|
@@ -26,7 +26,13 @@ uvx paimon
|
|
|
26
26
|
|
|
27
27
|
The first launch asks for a provider, model, API base and key, and saves them to `~/.config/paimon/default/config.json`. Then just type what you want done.
|
|
28
28
|
|
|
29
|
-
While it runs: `Shift+Tab` switches how much the agent may do on its own (**read**: ask before writing files or running commands, except clearly read-only ones like `ls` or `git status`, which run without asking, **edit**: edits inside the working directory go through, **yolo**: never ask), `Esc` interrupts the current turn, `Ctrl+P` opens the command palette (switch provider or profile, new, fork or resume session, show the model's thinking, compact the context), `Ctrl+C` quits.
|
|
29
|
+
While it runs: `Shift+Tab` switches how much the agent may do on its own (**read**: ask before writing files or running commands, except clearly read-only ones like `ls` or `git status`, which run without asking, **edit**: edits inside the working directory go through, **yolo**: never ask; the default), `Esc` interrupts the current turn, `Ctrl+P` opens the command palette (switch provider or profile, new, fork or resume session, show the model's thinking, compact the context), `Ctrl+C` quits.
|
|
30
|
+
|
|
31
|
+
`Ctrl+T` opens another session in a pane of its own, `Ctrl+W` closes one, `Ctrl+PageUp` and `Ctrl+PageDown` move between them, and `Ctrl+G` jumps to a pane waiting for permission.
|
|
32
|
+
|
|
33
|
+
Paimon can open panes itself: ask for two independent things and it starts a second agent in its own tab, with the same tools, working directory and permission mode. Its permission prompts appear in that tab, so `Ctrl+G` is how you unblock it. Those sessions belong to the one that started them, so they stay out of `paimon sessions` and end when it does.
|
|
34
|
+
|
|
35
|
+
It can also leave a command running in a tab of its own, for a dev server, a watcher or a long build that would otherwise hold up a turn. It asks first, every time, whatever the mode says about read-only commands. The tab streams the output and stops the command when you close it or quit. Programs that buffer their output when it is not going to a terminal print in blocks there rather than line by line; that is what a pipe costs, and Paimon does not emulate a terminal.
|
|
30
36
|
|
|
31
37
|
Write `@path/to/file` in a prompt to hand a file to the agent.
|
|
32
38
|
|
|
@@ -64,7 +70,7 @@ paimon log a1b2c3 # what a session did, one line per event
|
|
|
64
70
|
## Other ways to run it
|
|
65
71
|
|
|
66
72
|
```bash
|
|
67
|
-
paimon --mode
|
|
73
|
+
paimon --mode read # start in a more cautious permission mode (yolo is the default)
|
|
68
74
|
paimon --strict # ask before every command, even read-only ones
|
|
69
75
|
paimon --web # the same UI in a browser (--port, default 8000)
|
|
70
76
|
paimon -p "what does cli.py do?" # one answer on stdout, no UI
|
|
@@ -73,7 +79,7 @@ paimon --model zai:glm-4.7 # this model for this run only
|
|
|
73
79
|
paimon --profile work # a separately configured account
|
|
74
80
|
```
|
|
75
81
|
|
|
76
|
-
`-p` never stops to ask,
|
|
82
|
+
`-p` never stops to ask, and the default mode is `yolo`, so it can already write files and run commands; pass `--mode read` or `--mode edit` to keep the guardrails, in which case anything the mode would prompt for is refused instead (recognized read-only commands still run in read mode). Add `--output-format result` for a single JSON object with the outcome (or `json` for one event per line), and `--timeout`/`--max-tool-calls` to bound an unattended run.
|
|
77
83
|
|
|
78
84
|
## Configuration
|
|
79
85
|
|
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
# Paimon
|
|
2
|
+
|
|
3
|
+

|
|
4
|
+
|
|
5
|
+
[English](README.md) | 简体中文
|
|
6
|
+
|
|
7
|
+
Paimon 是一个终端里的 coding agent。它读写当前目录下的文件、执行命令,在做任何改动前会先询问。它也支持无头运行,可以被更强的 agent 作为执行者调用。
|
|
8
|
+
|
|
9
|
+
## 安装
|
|
10
|
+
|
|
11
|
+
```bash
|
|
12
|
+
uv tool install paimon # 或者:pip install paimon
|
|
13
|
+
```
|
|
14
|
+
|
|
15
|
+
## 快速开始
|
|
16
|
+
|
|
17
|
+
```bash
|
|
18
|
+
paimon
|
|
19
|
+
```
|
|
20
|
+
|
|
21
|
+
或者不安装直接运行:
|
|
22
|
+
|
|
23
|
+
```bash
|
|
24
|
+
uvx paimon
|
|
25
|
+
```
|
|
26
|
+
|
|
27
|
+
首次启动会询问 provider、模型、API base 和 key,并保存到 `~/.config/paimon/default/config.json`。之后输入要完成的任务即可。
|
|
28
|
+
|
|
29
|
+
运行时:`Shift+Tab` 切换 agent 的自主程度(**read**:写文件或执行命令前先询问,`ls`、`git status` 这类明确只读的命令除外,会直接执行,**edit**:工作目录内的编辑直接执行,**yolo**:从不询问,也是默认值),`Esc` 打断当前回合,`Ctrl+P` 打开命令面板(切换 provider 或 profile、新建、分叉或恢复会话、显示模型思考、压缩上下文),`Ctrl+C` 退出。
|
|
30
|
+
|
|
31
|
+
`Ctrl+T` 在新 pane 里打开另一个会话,`Ctrl+W` 关闭当前 pane,`Ctrl+PageUp` 和 `Ctrl+PageDown` 在 pane 之间切换,`Ctrl+G` 跳到正在等待授权的 pane。
|
|
32
|
+
|
|
33
|
+
Paimon 自己也能开 pane:让它同时做两件互不相干的事,它会在新 tab 里起第二个 agent,工具、工作目录和权限模式都和当前会话一样。它的授权确认弹在它自己的 tab 里,用 `Ctrl+G` 过去处理。这些会话属于开它们的那个会话,不会出现在 `paimon sessions` 里,也随它一起结束。
|
|
34
|
+
|
|
35
|
+
它也能把一条命令留在单独的 tab 里跑,比如开发服务器、文件监视或者很长的构建,不占着当前回合。这类命令一律先确认,不管当前模式对只读命令怎么规定。tab 里流式显示输出,关掉 tab 或退出时命令随之停止。输出不是终端时很多程序会按块缓冲,所以 tab 里可能一阵子不出东西再一次性出来:这是用管道代替终端的代价,Paimon 不做终端模拟。
|
|
36
|
+
|
|
37
|
+
在提示中写 `@path/to/file` 可以把文件提供给 agent。
|
|
38
|
+
|
|
39
|
+
## 当作 subagent 使用
|
|
40
|
+
|
|
41
|
+
前沿模型擅长制定计划和验收结果,中间的执行步骤往往比较机械。让 Paimon 使用成本较低的模型执行,由 Claude Code 或 Codex 制定计划并检查结果,只在必要的环节为前沿模型付费。用一个 profile 单独保存该模型的账号:
|
|
42
|
+
|
|
43
|
+
```bash
|
|
44
|
+
paimon login --profile glm --model zai:glm-4.7 --api-key-env ZAI_API_KEY
|
|
45
|
+
paimon --profile glm -p "apply the plan in PLAN.md" --mode edit --output-format result
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
自带的 skill 会向调用方 agent 说明这套流程(先用 `paimon status --json` 检查、单次运行、读取唯一一行 result 对象、用其中的 `session_id` 续跑、用 `paimon log` 查看运行过程):
|
|
49
|
+
|
|
50
|
+
```bash
|
|
51
|
+
paimon install-skill # 安装到 Claude Code(~/.claude/skills/paimon)
|
|
52
|
+
paimon install-skill --target codex # 安装到 Codex;--dest DIR 安装到任意目录
|
|
53
|
+
npx skills add aisk/paimon # 通过 skills.sh 安装同一个 skill
|
|
54
|
+
```
|
|
55
|
+
|
|
56
|
+
## 会话
|
|
57
|
+
|
|
58
|
+
每次对话都会保存。退出时 Paimon 会打印恢复该会话的命令:
|
|
59
|
+
|
|
60
|
+
```bash
|
|
61
|
+
paimon -r # 从当前目录的会话中选择
|
|
62
|
+
paimon -r a1b2c3 # 按 id 恢复
|
|
63
|
+
paimon -c # 恢复最近一个会话
|
|
64
|
+
paimon sessions # 列出会话(--json 输出机器可读格式)
|
|
65
|
+
paimon log a1b2c3 # 查看会话做了什么,每个事件一行
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
`paimon log` 的每行输出都带一个稳定的序号;`--after SEQ`、`--turns N`、`--tail N` 缩小范围,`--json` 和 `--full` 输出原始记录。
|
|
69
|
+
|
|
70
|
+
## 其他运行方式
|
|
71
|
+
|
|
72
|
+
```bash
|
|
73
|
+
paimon --mode read # 以更谨慎的权限模式启动(默认为 yolo)
|
|
74
|
+
paimon --strict # 每条命令都先询问,包括只读命令
|
|
75
|
+
paimon --web # 在浏览器中使用同一套 UI(--port,默认 8000)
|
|
76
|
+
paimon -p "what does cli.py do?" # 直接在 stdout 输出回答,不启动 UI
|
|
77
|
+
cat log.txt | paimon -p "summarize this"
|
|
78
|
+
paimon --model zai:glm-4.7 # 仅本次运行使用该模型
|
|
79
|
+
paimon --profile work # 单独配置的另一个账号
|
|
80
|
+
```
|
|
81
|
+
|
|
82
|
+
`-p` 不会停下来询问,默认模式是 `yolo`,因此已经可以修改文件和执行命令;传 `--mode read` 或 `--mode edit` 可以保留确认护栏,此时当前模式需要确认的操作会被直接拒绝(识别为只读的命令在 read 模式下仍会执行)。加 `--output-format result` 输出一个包含结果的 JSON 对象(`json` 则每行输出一个事件),用 `--timeout`/`--max-tool-calls` 为无人值守的运行设置上限。
|
|
83
|
+
|
|
84
|
+
## 配置
|
|
85
|
+
|
|
86
|
+
`~/.config/paimon/<name>/config.json` 保存每个 profile 的模型设置(不传 `--profile` 时为 `default`)。两个可选配置项可以改变它的行为:自动放行只读命令,以及在接近上下文上限时原地总结长对话。
|
|
87
|
+
|
|
88
|
+
```json
|
|
89
|
+
{
|
|
90
|
+
"safe_commands": false,
|
|
91
|
+
"compaction": {
|
|
92
|
+
"enabled": true,
|
|
93
|
+
"context_window": 128000,
|
|
94
|
+
"reserve_tokens": 16384,
|
|
95
|
+
"keep_recent_tokens": 20000
|
|
96
|
+
}
|
|
97
|
+
}
|
|
98
|
+
```
|
|
99
|
+
|
|
100
|
+
`safe_commands`(默认 `true`)允许 read 和 edit 模式不经询问执行一小组固定的、明确只读的命令(`ls`、`cat`、`git status` 等);`--strict` 可在单次运行中关闭它。被识别的命令可以用 `&&`、`;` 或管道串联,也包括 `cd 目录 && …`(目录须留在工作目录内,且整条命令都以 `&&` 连接)。重定向、`$()`/反引号替换和后台 `&` 仍会询问。
|
|
101
|
+
|
|
102
|
+
**这是防止 agent 失误的护栏,不是安全边界。** 被识别的命令仍然通过 `PATH` 查找,仍可能顺着符号链接读到工作目录之外;而且哪怕是纯读取,也会把文件内容带进模型上下文,只读不等于保密安全。需要真正的隔离时,请在容器或虚拟机中运行 Paimon。
|
|
103
|
+
|
|
104
|
+
会话存放在 `~/.local/share/paimon/sessions/`(`PAIMON_DATA_HOME` 可覆盖)。安装 [delta](https://github.com/dandavison/delta) 后文件改动的展示效果更好。
|
|
@@ -32,10 +32,20 @@ from pydantic_ai.models import Model, ModelRequestParameters
|
|
|
32
32
|
|
|
33
33
|
from . import compaction, retry, tools
|
|
34
34
|
from .config import Config
|
|
35
|
-
from .llm import build_model
|
|
35
|
+
from .llm import build_model, user_agent
|
|
36
36
|
from .mentions import expand_mentions
|
|
37
37
|
from .prompt import build_system_prompt
|
|
38
|
-
from .session import
|
|
38
|
+
from .session import (
|
|
39
|
+
Session,
|
|
40
|
+
SessionIncompleteError,
|
|
41
|
+
agents_message,
|
|
42
|
+
agents_text,
|
|
43
|
+
is_agents_message,
|
|
44
|
+
is_summary_message,
|
|
45
|
+
)
|
|
46
|
+
|
|
47
|
+
# Transient compaction failures tolerated in one turn before it is left off.
|
|
48
|
+
_MAX_COMPACTION_FAILURES = 3
|
|
39
49
|
|
|
40
50
|
|
|
41
51
|
# ---- Events yielded by Agent.run -------------------------------------------
|
|
@@ -128,12 +138,24 @@ class CompactionNotice:
|
|
|
128
138
|
"""A compaction checkpoint encountered while replaying history."""
|
|
129
139
|
|
|
130
140
|
|
|
141
|
+
@dataclass
|
|
142
|
+
class AgentsNotice:
|
|
143
|
+
"""A status line about the agents this session started.
|
|
144
|
+
|
|
145
|
+
Not replay-only: it is written into the history at the top of a turn, so
|
|
146
|
+
it is both yielded live and rebuilt when the session is resumed.
|
|
147
|
+
"""
|
|
148
|
+
|
|
149
|
+
text: str
|
|
150
|
+
|
|
151
|
+
|
|
131
152
|
# Everything ``Agent.run`` and ``replay_events`` can yield. Renderers dispatch
|
|
132
153
|
# on isinstance; the alias exists so a type checker can flag an unhandled one.
|
|
133
154
|
AgentEvent = (
|
|
134
155
|
TextDelta | ReasoningDelta | ToolStart | ToolEnd | TodosUpdate
|
|
135
156
|
| SessionHandoff | RequestStats | TurnEnd | ContextCompacted
|
|
136
157
|
| ContextCompactionFailed | ModelRetry | UserInput | CompactionNotice
|
|
158
|
+
| AgentsNotice
|
|
137
159
|
)
|
|
138
160
|
|
|
139
161
|
|
|
@@ -156,15 +178,22 @@ def replay_events(messages: list[ModelMessage]) -> list[AgentEvent]:
|
|
|
156
178
|
Lets a UI render resumed history through the same code path as live turns.
|
|
157
179
|
"""
|
|
158
180
|
events: list[AgentEvent] = []
|
|
181
|
+
# write_todos calls whose arguments never made a todo update: they were
|
|
182
|
+
# rejected live as a normal tool error, so they replay as one.
|
|
183
|
+
rejected_todos: set[str] = set()
|
|
159
184
|
for message in messages:
|
|
160
185
|
if is_summary_message(message):
|
|
161
186
|
events.append(CompactionNotice())
|
|
162
187
|
continue
|
|
188
|
+
if is_agents_message(message):
|
|
189
|
+
events.append(AgentsNotice(agents_text(message)))
|
|
190
|
+
continue
|
|
163
191
|
if isinstance(message, ModelRequest):
|
|
164
192
|
for part in message.parts:
|
|
165
193
|
if isinstance(part, UserPromptPart) and isinstance(part.content, str) and part.content:
|
|
166
194
|
events.append(UserInput(part.content))
|
|
167
|
-
elif isinstance(part, ToolReturnPart) and part.tool_name != "write_todos"
|
|
195
|
+
elif isinstance(part, ToolReturnPart) and (part.tool_name != "write_todos"
|
|
196
|
+
or part.tool_call_id in rejected_todos):
|
|
168
197
|
events.append(ToolEnd(part.tool_call_id, part.tool_name,
|
|
169
198
|
str(part.content or "(no output)")))
|
|
170
199
|
elif isinstance(message, ModelResponse):
|
|
@@ -175,9 +204,13 @@ def replay_events(messages: list[ModelMessage]) -> list[AgentEvent]:
|
|
|
175
204
|
events.append(TextDelta(part.content))
|
|
176
205
|
elif isinstance(part, ToolCallPart):
|
|
177
206
|
args = _parse_args(part.args)
|
|
178
|
-
|
|
179
|
-
|
|
207
|
+
todos = (tools.normalize_todos(args.get("todos"))
|
|
208
|
+
if part.tool_name == "write_todos" else None)
|
|
209
|
+
if todos is not None:
|
|
210
|
+
events.append(TodosUpdate(todos))
|
|
180
211
|
else:
|
|
212
|
+
if part.tool_name == "write_todos":
|
|
213
|
+
rejected_todos.add(part.tool_call_id)
|
|
181
214
|
events.append(ToolStart(part.tool_call_id, part.tool_name, args))
|
|
182
215
|
return events
|
|
183
216
|
|
|
@@ -215,13 +248,26 @@ class Agent:
|
|
|
215
248
|
"""
|
|
216
249
|
|
|
217
250
|
def __init__(self, session: Session, system_prompt: str, *, cwd: Optional[Path] = None,
|
|
218
|
-
confirm: Optional[ConfirmFn] = None, mode: str = "
|
|
251
|
+
confirm: Optional[ConfirmFn] = None, mode: str = "yolo",
|
|
219
252
|
config: Optional[Config] = None,
|
|
220
|
-
toolset: Optional[dict[str, tools.Tool]] = None
|
|
253
|
+
toolset: Optional[dict[str, tools.Tool]] = None,
|
|
254
|
+
model_override: Optional[str] = None):
|
|
221
255
|
self.cwd = Path(cwd or Path.cwd())
|
|
222
256
|
self.confirm = confirm
|
|
223
257
|
self.mode = mode
|
|
224
258
|
self.config = config or Config.load()
|
|
259
|
+
# Per-agent model choice. One Config instance is shared by every agent
|
|
260
|
+
# in the process, so writing config.model would repoint all of them;
|
|
261
|
+
# this overrides the model for this agent alone, credentials included
|
|
262
|
+
# (they come from the config either way).
|
|
263
|
+
self.model_override = model_override
|
|
264
|
+
# Set by the UI when this agent may start and talk to other agents.
|
|
265
|
+
# None everywhere else (headless, tests), where the agent tools refuse
|
|
266
|
+
# rather than pretend.
|
|
267
|
+
self.supervisor = None
|
|
268
|
+
# Per-agent tool state, kept off the tool functions so one agent's
|
|
269
|
+
# shell overflow files stay invisible to the next one.
|
|
270
|
+
self.tool_context = tools.ToolContext()
|
|
225
271
|
self.todos: list[dict] = []
|
|
226
272
|
self.session = session
|
|
227
273
|
self.system_prompt = system_prompt
|
|
@@ -234,44 +280,62 @@ class Agent:
|
|
|
234
280
|
|
|
235
281
|
@classmethod
|
|
236
282
|
def open(cls, cwd: Optional[Path] = None, *, session: Optional[Session] = None,
|
|
237
|
-
confirm: Optional[ConfirmFn] = None, mode: str = "
|
|
283
|
+
confirm: Optional[ConfirmFn] = None, mode: str = "yolo",
|
|
238
284
|
config: Optional[Config] = None,
|
|
239
285
|
append_system_prompt: Optional[str] = None,
|
|
240
|
-
toolset: Optional[dict[str, tools.Tool]] = None
|
|
286
|
+
toolset: Optional[dict[str, tools.Tool]] = None,
|
|
287
|
+
model_override: Optional[str] = None,
|
|
288
|
+
parent: Optional[str] = None) -> "Agent":
|
|
241
289
|
"""Start a new session, or resume ``session``, and take its lock.
|
|
242
290
|
|
|
243
291
|
``append_system_prompt`` is added to the end of a new session's system
|
|
244
292
|
prompt and persisted with it, so a resumed session keeps it. Resuming
|
|
245
293
|
with it set raises ``ValueError``: the persisted prompt is immutable.
|
|
294
|
+
``parent`` marks the new session as a subagent's, which keeps it out of
|
|
295
|
+
the session listings its parent shows up in.
|
|
246
296
|
|
|
247
|
-
Raises ``SessionBusyError`` when
|
|
248
|
-
``SessionIncompleteError`` when a resumed log
|
|
249
|
-
snapshot — both ``SessionError``, and neither
|
|
297
|
+
Raises ``SessionBusyError`` when the session is already open (here or
|
|
298
|
+
in another process) and ``SessionIncompleteError`` when a resumed log
|
|
299
|
+
has no system prompt snapshot — both ``SessionError``, and neither
|
|
300
|
+
leaves a lock held.
|
|
250
301
|
"""
|
|
251
302
|
cwd = Path(cwd or Path.cwd())
|
|
252
303
|
if session is not None and append_system_prompt:
|
|
253
304
|
raise ValueError("append_system_prompt only applies to a new session")
|
|
305
|
+
is_new = session is None
|
|
254
306
|
if session is None:
|
|
255
|
-
session = Session.create(cwd)
|
|
256
|
-
|
|
257
|
-
|
|
258
|
-
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
|
|
264
|
-
|
|
265
|
-
session.
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
|
|
307
|
+
session = Session.create(cwd, parent)
|
|
308
|
+
session.lock()
|
|
309
|
+
# Everything after the lock can fail — a full disk while writing the
|
|
310
|
+
# prompt, a message the current pydantic-ai cannot parse — and no
|
|
311
|
+
# Agent is returned to unlock it, so the lock is released here.
|
|
312
|
+
try:
|
|
313
|
+
if is_new:
|
|
314
|
+
system_prompt = build_system_prompt(cwd)
|
|
315
|
+
if append_system_prompt:
|
|
316
|
+
system_prompt += f"\n\n{append_system_prompt.strip()}"
|
|
317
|
+
session.append_system_prompt(system_prompt)
|
|
318
|
+
else:
|
|
319
|
+
system_prompt = session.system_prompt()
|
|
320
|
+
if system_prompt is None:
|
|
321
|
+
raise SessionIncompleteError("Session does not contain a persisted system prompt")
|
|
322
|
+
return cls(session, system_prompt, cwd=cwd, confirm=confirm, mode=mode,
|
|
323
|
+
config=config, toolset=toolset, model_override=model_override)
|
|
324
|
+
except BaseException:
|
|
325
|
+
session.unlock()
|
|
326
|
+
raise
|
|
327
|
+
|
|
328
|
+
@property
|
|
329
|
+
def model_name(self) -> Optional[str]:
|
|
330
|
+
"""The model this agent talks to: its own override, else the config's."""
|
|
331
|
+
return self.model_override or self.config.model
|
|
269
332
|
|
|
270
333
|
def _model(self) -> Model:
|
|
271
334
|
"""The configured model, rebuilt when login changes the config."""
|
|
272
|
-
|
|
335
|
+
name = self.model_name
|
|
336
|
+
if not name:
|
|
273
337
|
raise RuntimeError("No model configured; log in first")
|
|
274
|
-
key = (
|
|
338
|
+
key = (name, self.config.api_base, self.config.api_key)
|
|
275
339
|
if self._cached_model is None or self._cached_model[0] != key:
|
|
276
340
|
self._cached_model = (key, build_model(*key))
|
|
277
341
|
return self._cached_model[1]
|
|
@@ -299,13 +363,17 @@ class Agent:
|
|
|
299
363
|
if not force:
|
|
300
364
|
if not self.config.compaction_enabled:
|
|
301
365
|
return None
|
|
302
|
-
window = compaction.context_window(self.
|
|
366
|
+
window = compaction.context_window(self.model_name,
|
|
303
367
|
self.config.compaction_context_window)
|
|
304
|
-
|
|
368
|
+
# Nothing to compare against, so counting would be wasted work: an
|
|
369
|
+
# unknown window disables auto-compaction outright.
|
|
370
|
+
if window is None:
|
|
371
|
+
return None
|
|
372
|
+
tokens_before = await self.count_context_tokens()
|
|
305
373
|
if not compaction.should_compact(tokens_before, window, self.config.compaction_reserve_tokens):
|
|
306
374
|
return None
|
|
307
375
|
else:
|
|
308
|
-
tokens_before = self.count_context_tokens()
|
|
376
|
+
tokens_before = await self.count_context_tokens()
|
|
309
377
|
|
|
310
378
|
result = await compaction.compact(
|
|
311
379
|
self.history,
|
|
@@ -322,12 +390,19 @@ class Agent:
|
|
|
322
390
|
# append-message invariant (see _append_message) still holds afterwards.
|
|
323
391
|
self.session.append_compaction(result.summary, result.kept_messages, result.tokens_before)
|
|
324
392
|
self.history = result.messages
|
|
325
|
-
result.tokens_after = self.count_context_tokens()
|
|
393
|
+
result.tokens_after = await self.count_context_tokens()
|
|
326
394
|
return result
|
|
327
395
|
|
|
328
|
-
def count_context_tokens(self) -> int:
|
|
329
|
-
"""Estimate the tokens of everything the next request would send.
|
|
330
|
-
|
|
396
|
+
async def count_context_tokens(self) -> int:
|
|
397
|
+
"""Estimate the tokens of everything the next request would send.
|
|
398
|
+
|
|
399
|
+
Off the event loop: the count serializes the whole history, which is
|
|
400
|
+
hundreds of kilobytes on a long session. It runs at the top of every
|
|
401
|
+
model step, and every agent shares one loop, so counting inline stalls
|
|
402
|
+
every other agent's streaming output for as long as it takes.
|
|
403
|
+
"""
|
|
404
|
+
return await asyncio.to_thread(
|
|
405
|
+
compaction.count_tokens, list(self.history), self.tool_schemas, self.system_prompt)
|
|
331
406
|
|
|
332
407
|
async def compact_now(self) -> Optional[compaction.CompactionResult]:
|
|
333
408
|
"""Compact on demand; None when the history is too short to be worth it."""
|
|
@@ -343,8 +418,17 @@ class Agent:
|
|
|
343
418
|
persist: Callable[[], None]) -> AsyncIterator[AgentEvent]:
|
|
344
419
|
# Reports itself as TodosUpdate alone — no ToolStart/ToolEnd — which is
|
|
345
420
|
# also the shape replay_events produces for it, so no renderer has to
|
|
346
|
-
# special-case the name.
|
|
347
|
-
|
|
421
|
+
# special-case the name. A malformed argument is the exception: it
|
|
422
|
+
# renders as a normal failed tool call, live and on replay alike.
|
|
423
|
+
todos = tools.normalize_todos(args.get("todos"))
|
|
424
|
+
if todos is None:
|
|
425
|
+
yield ToolStart(call.tool_call_id, call.tool_name, args)
|
|
426
|
+
slot.content = ("Error: 'todos' must be an array of "
|
|
427
|
+
"{content, status} objects.")
|
|
428
|
+
persist()
|
|
429
|
+
yield ToolEnd(call.tool_call_id, call.tool_name, slot.content)
|
|
430
|
+
return
|
|
431
|
+
self.todos = todos
|
|
348
432
|
slot.content = tools.render_todos(self.todos)
|
|
349
433
|
persist()
|
|
350
434
|
yield TodosUpdate(list(self.todos))
|
|
@@ -362,12 +446,7 @@ class Agent:
|
|
|
362
446
|
persist()
|
|
363
447
|
yield ToolEnd(call.tool_call_id, call.tool_name, slot.content)
|
|
364
448
|
return
|
|
365
|
-
|
|
366
|
-
self.toolset) == "confirm"
|
|
367
|
-
allowed = not needs_confirm or (
|
|
368
|
-
await self.confirm(call.tool_name, args) if self.confirm else False
|
|
369
|
-
)
|
|
370
|
-
if not allowed:
|
|
449
|
+
if not await self._permitted(call.tool_name, args):
|
|
371
450
|
slot.content = "User denied this operation."
|
|
372
451
|
persist()
|
|
373
452
|
yield ToolEnd(call.tool_call_id, call.tool_name, slot.content, denied=True)
|
|
@@ -378,9 +457,52 @@ class Agent:
|
|
|
378
457
|
yield ToolEnd(call.tool_call_id, call.tool_name, slot.content)
|
|
379
458
|
yield SessionHandoff(prompt_text)
|
|
380
459
|
|
|
460
|
+
async def _permitted(self, name: str, args: dict) -> bool:
|
|
461
|
+
"""Gate a tool the loop runs itself, the way run_tool gates the rest.
|
|
462
|
+
|
|
463
|
+
The enforcement point for everything in _AGENT_HANDLED: without a
|
|
464
|
+
confirm hook a call that needs one is denied, so a headless agent
|
|
465
|
+
cannot walk around the permission mode here either.
|
|
466
|
+
"""
|
|
467
|
+
if tools.gate(name, args, self.mode, self.cwd, self.toolset,
|
|
468
|
+
safe_commands=self.config.safe_commands,
|
|
469
|
+
ctx=self.tool_context) != "confirm":
|
|
470
|
+
return True
|
|
471
|
+
return await self.confirm(name, args) if self.confirm else False
|
|
472
|
+
|
|
473
|
+
async def _run_supervised(self, call: ToolCallPart, args: dict, slot: ToolReturnPart,
|
|
474
|
+
persist: Callable[[], None]) -> AsyncIterator[AgentEvent]:
|
|
475
|
+
"""The job tools: starting agents and commands, and reporting on both.
|
|
476
|
+
|
|
477
|
+
They act on the pool of jobs the UI is running, which no stateless tool
|
|
478
|
+
function can reach, and the supervisor is the one thing that knows
|
|
479
|
+
whether a given job is busy \u2014 so they are dispatched from here. Most of
|
|
480
|
+
them are not gated at all (they only reach jobs this same agent
|
|
481
|
+
started); run_background is, because it starts a process.
|
|
482
|
+
"""
|
|
483
|
+
yield ToolStart(call.tool_call_id, call.tool_name, args)
|
|
484
|
+
if self.supervisor is None:
|
|
485
|
+
slot.content = ("Error: this only works in the interactive UI; "
|
|
486
|
+
"do the work yourself instead.")
|
|
487
|
+
elif not await self._permitted(call.tool_name, args):
|
|
488
|
+
slot.content = "User denied this operation."
|
|
489
|
+
persist()
|
|
490
|
+
yield ToolEnd(call.tool_call_id, call.tool_name, slot.content, denied=True)
|
|
491
|
+
return
|
|
492
|
+
else:
|
|
493
|
+
slot.content = await self.supervisor.handle(call.tool_name, args, caller=self)
|
|
494
|
+
persist()
|
|
495
|
+
yield ToolEnd(call.tool_call_id, call.tool_name, slot.content)
|
|
496
|
+
|
|
381
497
|
_AGENT_HANDLED = {
|
|
382
498
|
"write_todos": _run_write_todos,
|
|
383
499
|
"start_new_session": _run_start_new_session,
|
|
500
|
+
"spawn_agent": _run_supervised,
|
|
501
|
+
"send_to_agent": _run_supervised,
|
|
502
|
+
"run_background": _run_supervised,
|
|
503
|
+
"read_job": _run_supervised,
|
|
504
|
+
"wait_for_job": _run_supervised,
|
|
505
|
+
"stop_job": _run_supervised,
|
|
384
506
|
}
|
|
385
507
|
|
|
386
508
|
async def run(self, user_input: str, *, expand: bool = True) -> AsyncIterator[AgentEvent]:
|
|
@@ -391,17 +513,34 @@ class Agent:
|
|
|
391
513
|
stdin, where a line like ``@foo.py`` is data rather than a mention).
|
|
392
514
|
"""
|
|
393
515
|
prompt = expand_mentions(user_input, self.cwd) if expand else user_input
|
|
516
|
+
# Agents this session started report in here, at the top of the next
|
|
517
|
+
# turn, rather than by interrupting whatever the user is typing. It has
|
|
518
|
+
# to be a persisted message: an event the model never sees would defeat
|
|
519
|
+
# the point, which is to get it to call read_job.
|
|
520
|
+
summary = self.supervisor.status_summary(self) if self.supervisor is not None else None
|
|
521
|
+
if summary:
|
|
522
|
+
self._append_message(agents_message(summary))
|
|
523
|
+
yield AgentsNotice(summary)
|
|
394
524
|
self._append_message(ModelRequest(parts=[UserPromptPart(content=prompt)]))
|
|
395
|
-
|
|
525
|
+
# A compaction that failed for a transient reason is retried on the
|
|
526
|
+
# next step of this turn — the context only keeps growing, so giving
|
|
527
|
+
# up on the first rate limit disables the safety net exactly when it
|
|
528
|
+
# is needed. Anything else, and repeated transient failures, stop it
|
|
529
|
+
# for the rest of the turn instead of paying for it every step.
|
|
530
|
+
compaction_off = False
|
|
531
|
+
compaction_failures = 0
|
|
396
532
|
|
|
397
533
|
while True:
|
|
398
|
-
if not
|
|
534
|
+
if not compaction_off:
|
|
399
535
|
try:
|
|
400
536
|
compacted = await self._maybe_compact()
|
|
401
537
|
except Exception as exc: # noqa: BLE001 - the normal request may still fit
|
|
402
|
-
|
|
538
|
+
compaction_failures += 1
|
|
539
|
+
compaction_off = (not retry.is_transient(exc)
|
|
540
|
+
or compaction_failures >= _MAX_COMPACTION_FAILURES)
|
|
403
541
|
yield ContextCompactionFailed(str(exc))
|
|
404
542
|
else:
|
|
543
|
+
compaction_failures = 0
|
|
405
544
|
if compacted:
|
|
406
545
|
yield ContextCompacted(compacted.tokens_before, compacted.tokens_after)
|
|
407
546
|
|
|
@@ -423,7 +562,10 @@ class Agent:
|
|
|
423
562
|
first_event_at: Optional[float] = None
|
|
424
563
|
try:
|
|
425
564
|
async with model_request_stream(
|
|
426
|
-
model,
|
|
565
|
+
model,
|
|
566
|
+
request_messages,
|
|
567
|
+
model_settings={"extra_headers": {"User-Agent": user_agent()}},
|
|
568
|
+
model_request_parameters=parameters,
|
|
427
569
|
) as stream:
|
|
428
570
|
async for event in stream:
|
|
429
571
|
if first_event_at is None:
|
|
@@ -523,7 +665,8 @@ class Agent:
|
|
|
523
665
|
yield ToolStart(call.tool_call_id, name, args)
|
|
524
666
|
result, denied = await tools.run_tool(name, args, self.cwd, self.mode,
|
|
525
667
|
self.confirm, self.toolset,
|
|
526
|
-
safe_commands=self.config.safe_commands
|
|
668
|
+
safe_commands=self.config.safe_commands,
|
|
669
|
+
ctx=self.tool_context)
|
|
527
670
|
|
|
528
671
|
slot.content = result
|
|
529
672
|
persist()
|