thwip-cli 1.2.0__tar.gz → 1.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {thwip_cli-1.2.0 → thwip_cli-1.4.0}/PKG-INFO +51 -10
- {thwip_cli-1.2.0 → thwip_cli-1.4.0}/README.md +50 -9
- thwip_cli-1.4.0/docs/verification.md +58 -0
- {thwip_cli-1.2.0 → thwip_cli-1.4.0}/pyproject.toml +1 -1
- thwip_cli-1.4.0/tests/test_audit_regressions.py +205 -0
- {thwip_cli-1.2.0 → thwip_cli-1.4.0}/tests/test_cli.py +27 -2
- thwip_cli-1.4.0/tests/test_compatible_streaming.py +75 -0
- thwip_cli-1.4.0/tests/test_native_agents.py +198 -0
- thwip_cli-1.4.0/tests/test_native_launcher.py +69 -0
- thwip_cli-1.4.0/tests/test_native_print.py +234 -0
- thwip_cli-1.4.0/tests/test_native_rpc.py +51 -0
- thwip_cli-1.4.0/tests/test_repair_verification.py +177 -0
- {thwip_cli-1.2.0 → thwip_cli-1.4.0}/tests/test_session.py +41 -0
- thwip_cli-1.4.0/tests/test_utils.py +17 -0
- {thwip_cli-1.2.0 → thwip_cli-1.4.0}/thwip/__init__.py +1 -1
- {thwip_cli-1.2.0 → thwip_cli-1.4.0}/thwip/agents/__init__.py +42 -0
- {thwip_cli-1.2.0 → thwip_cli-1.4.0}/thwip/agents/base.py +15 -1
- thwip_cli-1.4.0/thwip/agents/chat_messages.py +20 -0
- {thwip_cli-1.2.0 → thwip_cli-1.4.0}/thwip/agents/claude_agent.py +6 -1
- {thwip_cli-1.2.0 → thwip_cli-1.4.0}/thwip/agents/deepseek_agent.py +2 -1
- {thwip_cli-1.2.0 → thwip_cli-1.4.0}/thwip/agents/google_agent.py +5 -1
- {thwip_cli-1.2.0 → thwip_cli-1.4.0}/thwip/agents/groq_agent.py +5 -3
- thwip_cli-1.4.0/thwip/agents/native_agent.py +281 -0
- thwip_cli-1.4.0/thwip/agents/native_common.py +102 -0
- thwip_cli-1.4.0/thwip/agents/native_print.py +327 -0
- thwip_cli-1.4.0/thwip/agents/native_rpc.py +86 -0
- {thwip_cli-1.2.0 → thwip_cli-1.4.0}/thwip/agents/openai_agent.py +8 -1
- {thwip_cli-1.2.0 → thwip_cli-1.4.0}/thwip/agents/openrouter_agent.py +2 -1
- {thwip_cli-1.2.0 → thwip_cli-1.4.0}/thwip/cli.py +284 -57
- {thwip_cli-1.2.0 → thwip_cli-1.4.0}/thwip/config.py +27 -0
- {thwip_cli-1.2.0 → thwip_cli-1.4.0}/thwip/detector.py +16 -10
- {thwip_cli-1.2.0 → thwip_cli-1.4.0}/thwip/handoff.py +3 -1
- {thwip_cli-1.2.0 → thwip_cli-1.4.0}/thwip/limits.py +33 -4
- {thwip_cli-1.2.0 → thwip_cli-1.4.0}/thwip/session.py +29 -1
- {thwip_cli-1.2.0 → thwip_cli-1.4.0}/thwip/shortcuts.py +1 -0
- {thwip_cli-1.2.0 → thwip_cli-1.4.0}/thwip/theme.py +17 -11
- {thwip_cli-1.2.0 → thwip_cli-1.4.0}/thwip/tools/__init__.py +22 -1
- {thwip_cli-1.2.0 → thwip_cli-1.4.0}/thwip/tools/code_runner.py +4 -6
- {thwip_cli-1.2.0 → thwip_cli-1.4.0}/thwip/tools/file_editor.py +1 -1
- {thwip_cli-1.2.0 → thwip_cli-1.4.0}/thwip/tools/terminal.py +35 -5
- {thwip_cli-1.2.0 → thwip_cli-1.4.0}/thwip/utils.py +1 -1
- {thwip_cli-1.2.0 → thwip_cli-1.4.0}/uv.lock +1 -1
- {thwip_cli-1.2.0 → thwip_cli-1.4.0}/.github/workflows/publish.yml +0 -0
- {thwip_cli-1.2.0 → thwip_cli-1.4.0}/.gitignore +0 -0
- {thwip_cli-1.2.0 → thwip_cli-1.4.0}/LICENSE +0 -0
- {thwip_cli-1.2.0 → thwip_cli-1.4.0}/docs/handoff-research.md +0 -0
- {thwip_cli-1.2.0 → thwip_cli-1.4.0}/install.sh +0 -0
- {thwip_cli-1.2.0 → thwip_cli-1.4.0}/tests/test_agents.py +0 -0
- {thwip_cli-1.2.0 → thwip_cli-1.4.0}/tests/test_config.py +0 -0
- {thwip_cli-1.2.0 → thwip_cli-1.4.0}/tests/test_detector.py +0 -0
- {thwip_cli-1.2.0 → thwip_cli-1.4.0}/tests/test_handoff.py +0 -0
- {thwip_cli-1.2.0 → thwip_cli-1.4.0}/tests/test_tools.py +0 -0
- {thwip_cli-1.2.0 → thwip_cli-1.4.0}/thwip/__main__.py +0 -0
- {thwip_cli-1.2.0 → thwip_cli-1.4.0}/thwip/agents/ollama_agent.py +0 -0
- {thwip_cli-1.2.0 → thwip_cli-1.4.0}/thwip/tools/git_ops.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: thwip-cli
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.4.0
|
|
4
4
|
Summary: Universal coding agent multiplexer: detect, switch, and route between AI coding agents seamlessly
|
|
5
5
|
Project-URL: Homepage, https://github.com/tanmayhutt/thwip-cli
|
|
6
6
|
Project-URL: Repository, https://github.com/tanmayhutt/thwip-cli
|
|
@@ -46,6 +46,7 @@ Description-Content-Type: text/markdown
|
|
|
46
46
|
## Features
|
|
47
47
|
|
|
48
48
|
- **Auto-Detection**: Discovers installed AI coding agents (Claude Code, Antigravity, Gemini CLI, OpenAI/Codex, Aider, Copilot, Cursor, Windsurf, Cline, Ollama) and configured credentials.
|
|
49
|
+
- **Existing Sign-ins**: Chats through installed Codex, Claude Code, and Antigravity CLIs using their own logins and live model lists, with no API key required.
|
|
49
50
|
- **Context Portability**: Switch providers with stored conversational text. Working files remain in the selected local project; they are not automatically uploaded.
|
|
50
51
|
- **Handoff Preview**: Inspect text continuity, omitted state, capability changes, approximate context pressure, and a text fingerprint before switching. Runs locally without model calls.
|
|
51
52
|
- **Dynamic UI**: Terminal interface adapts its status bar, capabilities, and theme based on the active provider.
|
|
@@ -81,6 +82,8 @@ Start the interactive terminal in your current project directory:
|
|
|
81
82
|
|
|
82
83
|
```bash
|
|
83
84
|
thwip
|
|
85
|
+
thwip --project ~/code/my-app # open a specific project
|
|
86
|
+
thwip --version
|
|
84
87
|
```
|
|
85
88
|
|
|
86
89
|
---
|
|
@@ -91,12 +94,13 @@ thwip
|
|
|
91
94
|
|:---|:---|
|
|
92
95
|
| `/switch [agent] [model]` | Switch active agent or model mid-conversation |
|
|
93
96
|
| `/handoff [agent] [model]` | Preview a target locally without switching or sending data |
|
|
97
|
+
| `/native codex` | Save and leave Thwip for the installed Codex CLI itself (launcher, no context transfer) |
|
|
94
98
|
| `/agents` | Show all detected coding agents, company status, and capabilities |
|
|
95
99
|
| `/models [agent]` | List available models for current or target agent |
|
|
96
100
|
| `/key [provider]` | Enter an API key securely without placing it in prompt history |
|
|
97
101
|
| `/status` | Display current session, project, and token stats |
|
|
98
102
|
| `/limits` | View token usage, quota, and spend metrics |
|
|
99
|
-
| `/detect` | Re-scan
|
|
103
|
+
| `/detect` | Re-scan installed agents and reconnect CLI sign-ins |
|
|
100
104
|
| `/session save [name]` | Save current chat session |
|
|
101
105
|
| `/session load <name>` | Load a previously saved session |
|
|
102
106
|
| `/session list` | List all saved sessions |
|
|
@@ -105,13 +109,50 @@ thwip
|
|
|
105
109
|
| `/cost` | Show estimated session and cumulative cost |
|
|
106
110
|
| `/project [path]` | View or change project working directory |
|
|
107
111
|
| `Ctrl + S` | Quick switch agent prompt |
|
|
108
|
-
|
|
109
|
-
Short aliases are available for frequent commands: `/a`, `/m`, `/s`, `/k`, `/g`, and `/t`.
|
|
110
112
|
| `Ctrl + T` | Show agent status |
|
|
111
113
|
| `Ctrl + H` | View history |
|
|
112
114
|
| `/quit` | Exit thwip |
|
|
113
115
|
|
|
114
|
-
|
|
116
|
+
Short aliases are available for frequent commands: `/a`, `/m`, `/s`, `/sw`, `/k`, `/g`, and `/t`.
|
|
117
|
+
|
|
118
|
+
### Existing CLI sign-ins (no API key needed)
|
|
119
|
+
|
|
120
|
+
If Codex, Claude Code, or the Antigravity CLI is installed and signed in, Thwip
|
|
121
|
+
connects to it at startup and uses that sign-in for chat. No API key is copied or
|
|
122
|
+
required, and Thwip never reads the CLI's stored credentials. A configured direct
|
|
123
|
+
API key for the same provider always takes precedence over the native connection.
|
|
124
|
+
|
|
125
|
+
| Provider | Installed CLI | Transport | Model list |
|
|
126
|
+
|:---|:---|:---|:---|
|
|
127
|
+
| OpenAI | `codex` | Codex App Server (JSON-RPC over stdio) | Live from `model/list` |
|
|
128
|
+
| Anthropic | `claude` | Claude Code print mode (`stream-json`) | Aliases `fable`, `opus`, `sonnet`, `haiku`; explicit IDs pass through |
|
|
129
|
+
| Google | `agy` (Antigravity CLI) or `gemini` | Antigravity print mode (`stream-json`) or Gemini ACP | Live from `agy models` or the ACP session |
|
|
130
|
+
|
|
131
|
+
`/models` refreshes the catalog supplied by each CLI. An explicit model ID that is
|
|
132
|
+
not in the catalog is passed to the CLI for validation instead of being rejected by
|
|
133
|
+
a bundled list. A model available in a desktop app may still be absent from the
|
|
134
|
+
installed CLI's catalog or account access.
|
|
135
|
+
|
|
136
|
+
Native connections are read-only by default. Codex starts with a read-only sandbox
|
|
137
|
+
and asks before operations outside it; approval requests appear in Thwip with the
|
|
138
|
+
command or file list and default to denial. Claude Code and the Antigravity CLI run
|
|
139
|
+
in their non-interactive print modes, where tools that would need an approval are
|
|
140
|
+
declined by the CLI itself. Each turn sends the portable text conversation to a
|
|
141
|
+
fresh native session, so native reasoning and tool state do not carry between turns.
|
|
142
|
+
Ctrl+C interrupts the current turn and stops the child process; the unanswered
|
|
143
|
+
message is removed so it can be re-sent or handed to another provider.
|
|
144
|
+
|
|
145
|
+
Native billing and usage limits are managed by each CLI account. Thwip records the
|
|
146
|
+
token counts the CLIs report but does not estimate cost for them. Codex and Claude
|
|
147
|
+
Code also report their account usage windows (5 hour and 7 day) after each response;
|
|
148
|
+
`/limits` and `/status` show the used percentage and reset time. When a CLI reports
|
|
149
|
+
an exhausted usage limit, the standard failover prompt offers the other connected
|
|
150
|
+
providers.
|
|
151
|
+
|
|
152
|
+
`/native codex` remains available as a launcher: it saves the Thwip session and
|
|
153
|
+
replaces Thwip with the Codex CLI itself in the selected project. Conversation
|
|
154
|
+
history is not transferred by the launcher; use `/session load` after restarting
|
|
155
|
+
Thwip to resume.
|
|
115
156
|
|
|
116
157
|
## Auditable handoffs
|
|
117
158
|
|
|
@@ -150,11 +191,11 @@ See [research and prior art](docs/handoff-research.md) for the differentiation r
|
|
|
150
191
|
|
|
151
192
|
| Company | Agent | Capabilities |
|
|
152
193
|
|:---|:---|:---|
|
|
153
|
-
| Anthropic | Claude API (Fable 5, Opus 5, Sonnet 5, Haiku 4.5) | Chat, File Edit, Code Run, Terminal, Git |
|
|
154
|
-
| Google | Gemini API (3.1 Pro Preview, 3.7 Flash, 3.5 Flash-Lite) | Chat, File Edit, Code Run, Terminal, Git |
|
|
155
|
-
| OpenAI | OpenAI API (GPT-5.6 Sol, Terra, Luna) | Chat, File Edit, Code Run, Terminal, Git |
|
|
194
|
+
| Anthropic | Claude Code sign-in, or Claude API (Fable 5, Opus 5, Sonnet 5, Haiku 4.5) | Chat, File Edit, Code Run, Terminal, Git |
|
|
195
|
+
| Google | Antigravity or Gemini CLI sign-in, or Gemini API (3.1 Pro Preview, 3.7 Flash, 3.5 Flash-Lite) | Chat, File Edit, Code Run, Terminal, Git |
|
|
196
|
+
| OpenAI | Codex CLI sign-in, or OpenAI API (GPT-5.6 Sol, Terra, Luna) | Chat, File Edit, Code Run, Terminal, Git |
|
|
156
197
|
| DeepSeek | DeepSeek V3 / R1 Reasoner | Chat, File Edit, Code Run, Reasoning |
|
|
157
|
-
| Groq | Llama 3.3
|
|
198
|
+
| Groq | GPT-OSS 120B (default); Llama 3.3 for eligible enterprise accounts only | Chat, File Edit, Code Run |
|
|
158
199
|
| Ollama | Local Models (Llama 3.3, Qwen Coder, DeepSeek R1) | Chat, File Edit, Code Run (Local, Offline) |
|
|
159
200
|
| OpenRouter | Multi-Company Models | Gateway Routing |
|
|
160
201
|
|
|
@@ -164,7 +205,7 @@ See [research and prior art](docs/handoff-research.md) for the differentiation r
|
|
|
164
205
|
|
|
165
206
|
thwip auto-detects existing API keys from environment variables and existing agent configs (`~/.claude.json`, `~/.gemini/config.json`). Configuration can also be set manually:
|
|
166
207
|
|
|
167
|
-
Installed
|
|
208
|
+
Installed CLI sign-ins and API access are separate paths. A signed-in Codex, Claude Code, or Antigravity CLI is used directly through its own protocol (see above). The direct SDK adapters for Anthropic, Google, OpenAI, DeepSeek, Groq, and OpenRouter require a provider API key. Ollama needs no key when its local server is running.
|
|
168
209
|
|
|
169
210
|
```toml
|
|
170
211
|
[defaults]
|
|
@@ -9,6 +9,7 @@
|
|
|
9
9
|
## Features
|
|
10
10
|
|
|
11
11
|
- **Auto-Detection**: Discovers installed AI coding agents (Claude Code, Antigravity, Gemini CLI, OpenAI/Codex, Aider, Copilot, Cursor, Windsurf, Cline, Ollama) and configured credentials.
|
|
12
|
+
- **Existing Sign-ins**: Chats through installed Codex, Claude Code, and Antigravity CLIs using their own logins and live model lists, with no API key required.
|
|
12
13
|
- **Context Portability**: Switch providers with stored conversational text. Working files remain in the selected local project; they are not automatically uploaded.
|
|
13
14
|
- **Handoff Preview**: Inspect text continuity, omitted state, capability changes, approximate context pressure, and a text fingerprint before switching. Runs locally without model calls.
|
|
14
15
|
- **Dynamic UI**: Terminal interface adapts its status bar, capabilities, and theme based on the active provider.
|
|
@@ -44,6 +45,8 @@ Start the interactive terminal in your current project directory:
|
|
|
44
45
|
|
|
45
46
|
```bash
|
|
46
47
|
thwip
|
|
48
|
+
thwip --project ~/code/my-app # open a specific project
|
|
49
|
+
thwip --version
|
|
47
50
|
```
|
|
48
51
|
|
|
49
52
|
---
|
|
@@ -54,12 +57,13 @@ thwip
|
|
|
54
57
|
|:---|:---|
|
|
55
58
|
| `/switch [agent] [model]` | Switch active agent or model mid-conversation |
|
|
56
59
|
| `/handoff [agent] [model]` | Preview a target locally without switching or sending data |
|
|
60
|
+
| `/native codex` | Save and leave Thwip for the installed Codex CLI itself (launcher, no context transfer) |
|
|
57
61
|
| `/agents` | Show all detected coding agents, company status, and capabilities |
|
|
58
62
|
| `/models [agent]` | List available models for current or target agent |
|
|
59
63
|
| `/key [provider]` | Enter an API key securely without placing it in prompt history |
|
|
60
64
|
| `/status` | Display current session, project, and token stats |
|
|
61
65
|
| `/limits` | View token usage, quota, and spend metrics |
|
|
62
|
-
| `/detect` | Re-scan
|
|
66
|
+
| `/detect` | Re-scan installed agents and reconnect CLI sign-ins |
|
|
63
67
|
| `/session save [name]` | Save current chat session |
|
|
64
68
|
| `/session load <name>` | Load a previously saved session |
|
|
65
69
|
| `/session list` | List all saved sessions |
|
|
@@ -68,13 +72,50 @@ thwip
|
|
|
68
72
|
| `/cost` | Show estimated session and cumulative cost |
|
|
69
73
|
| `/project [path]` | View or change project working directory |
|
|
70
74
|
| `Ctrl + S` | Quick switch agent prompt |
|
|
71
|
-
|
|
72
|
-
Short aliases are available for frequent commands: `/a`, `/m`, `/s`, `/k`, `/g`, and `/t`.
|
|
73
75
|
| `Ctrl + T` | Show agent status |
|
|
74
76
|
| `Ctrl + H` | View history |
|
|
75
77
|
| `/quit` | Exit thwip |
|
|
76
78
|
|
|
77
|
-
|
|
79
|
+
Short aliases are available for frequent commands: `/a`, `/m`, `/s`, `/sw`, `/k`, `/g`, and `/t`.
|
|
80
|
+
|
|
81
|
+
### Existing CLI sign-ins (no API key needed)
|
|
82
|
+
|
|
83
|
+
If Codex, Claude Code, or the Antigravity CLI is installed and signed in, Thwip
|
|
84
|
+
connects to it at startup and uses that sign-in for chat. No API key is copied or
|
|
85
|
+
required, and Thwip never reads the CLI's stored credentials. A configured direct
|
|
86
|
+
API key for the same provider always takes precedence over the native connection.
|
|
87
|
+
|
|
88
|
+
| Provider | Installed CLI | Transport | Model list |
|
|
89
|
+
|:---|:---|:---|:---|
|
|
90
|
+
| OpenAI | `codex` | Codex App Server (JSON-RPC over stdio) | Live from `model/list` |
|
|
91
|
+
| Anthropic | `claude` | Claude Code print mode (`stream-json`) | Aliases `fable`, `opus`, `sonnet`, `haiku`; explicit IDs pass through |
|
|
92
|
+
| Google | `agy` (Antigravity CLI) or `gemini` | Antigravity print mode (`stream-json`) or Gemini ACP | Live from `agy models` or the ACP session |
|
|
93
|
+
|
|
94
|
+
`/models` refreshes the catalog supplied by each CLI. An explicit model ID that is
|
|
95
|
+
not in the catalog is passed to the CLI for validation instead of being rejected by
|
|
96
|
+
a bundled list. A model available in a desktop app may still be absent from the
|
|
97
|
+
installed CLI's catalog or account access.
|
|
98
|
+
|
|
99
|
+
Native connections are read-only by default. Codex starts with a read-only sandbox
|
|
100
|
+
and asks before operations outside it; approval requests appear in Thwip with the
|
|
101
|
+
command or file list and default to denial. Claude Code and the Antigravity CLI run
|
|
102
|
+
in their non-interactive print modes, where tools that would need an approval are
|
|
103
|
+
declined by the CLI itself. Each turn sends the portable text conversation to a
|
|
104
|
+
fresh native session, so native reasoning and tool state do not carry between turns.
|
|
105
|
+
Ctrl+C interrupts the current turn and stops the child process; the unanswered
|
|
106
|
+
message is removed so it can be re-sent or handed to another provider.
|
|
107
|
+
|
|
108
|
+
Native billing and usage limits are managed by each CLI account. Thwip records the
|
|
109
|
+
token counts the CLIs report but does not estimate cost for them. Codex and Claude
|
|
110
|
+
Code also report their account usage windows (5 hour and 7 day) after each response;
|
|
111
|
+
`/limits` and `/status` show the used percentage and reset time. When a CLI reports
|
|
112
|
+
an exhausted usage limit, the standard failover prompt offers the other connected
|
|
113
|
+
providers.
|
|
114
|
+
|
|
115
|
+
`/native codex` remains available as a launcher: it saves the Thwip session and
|
|
116
|
+
replaces Thwip with the Codex CLI itself in the selected project. Conversation
|
|
117
|
+
history is not transferred by the launcher; use `/session load` after restarting
|
|
118
|
+
Thwip to resume.
|
|
78
119
|
|
|
79
120
|
## Auditable handoffs
|
|
80
121
|
|
|
@@ -113,11 +154,11 @@ See [research and prior art](docs/handoff-research.md) for the differentiation r
|
|
|
113
154
|
|
|
114
155
|
| Company | Agent | Capabilities |
|
|
115
156
|
|:---|:---|:---|
|
|
116
|
-
| Anthropic | Claude API (Fable 5, Opus 5, Sonnet 5, Haiku 4.5) | Chat, File Edit, Code Run, Terminal, Git |
|
|
117
|
-
| Google | Gemini API (3.1 Pro Preview, 3.7 Flash, 3.5 Flash-Lite) | Chat, File Edit, Code Run, Terminal, Git |
|
|
118
|
-
| OpenAI | OpenAI API (GPT-5.6 Sol, Terra, Luna) | Chat, File Edit, Code Run, Terminal, Git |
|
|
157
|
+
| Anthropic | Claude Code sign-in, or Claude API (Fable 5, Opus 5, Sonnet 5, Haiku 4.5) | Chat, File Edit, Code Run, Terminal, Git |
|
|
158
|
+
| Google | Antigravity or Gemini CLI sign-in, or Gemini API (3.1 Pro Preview, 3.7 Flash, 3.5 Flash-Lite) | Chat, File Edit, Code Run, Terminal, Git |
|
|
159
|
+
| OpenAI | Codex CLI sign-in, or OpenAI API (GPT-5.6 Sol, Terra, Luna) | Chat, File Edit, Code Run, Terminal, Git |
|
|
119
160
|
| DeepSeek | DeepSeek V3 / R1 Reasoner | Chat, File Edit, Code Run, Reasoning |
|
|
120
|
-
| Groq | Llama 3.3
|
|
161
|
+
| Groq | GPT-OSS 120B (default); Llama 3.3 for eligible enterprise accounts only | Chat, File Edit, Code Run |
|
|
121
162
|
| Ollama | Local Models (Llama 3.3, Qwen Coder, DeepSeek R1) | Chat, File Edit, Code Run (Local, Offline) |
|
|
122
163
|
| OpenRouter | Multi-Company Models | Gateway Routing |
|
|
123
164
|
|
|
@@ -127,7 +168,7 @@ See [research and prior art](docs/handoff-research.md) for the differentiation r
|
|
|
127
168
|
|
|
128
169
|
thwip auto-detects existing API keys from environment variables and existing agent configs (`~/.claude.json`, `~/.gemini/config.json`). Configuration can also be set manually:
|
|
129
170
|
|
|
130
|
-
Installed
|
|
171
|
+
Installed CLI sign-ins and API access are separate paths. A signed-in Codex, Claude Code, or Antigravity CLI is used directly through its own protocol (see above). The direct SDK adapters for Anthropic, Google, OpenAI, DeepSeek, Groq, and OpenRouter require a provider API key. Ollama needs no key when its local server is running.
|
|
131
172
|
|
|
132
173
|
```toml
|
|
133
174
|
[defaults]
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
# Verification status
|
|
2
|
+
|
|
3
|
+
Verified locally on 2026-09-24 for the unreleased v1.4.0 source. The Python suite
|
|
4
|
+
passes 216 offline tests. Exhaustive behavior across every provider and
|
|
5
|
+
configuration has not been established.
|
|
6
|
+
|
|
7
|
+
## Native CLI connections (2026-09-24)
|
|
8
|
+
|
|
9
|
+
Live checks were run on macOS through a pseudo-terminal driving the real REPL with
|
|
10
|
+
the installed Codex CLI 0.152.1, Claude Code 2.1.281, and Antigravity CLI 1.2.8,
|
|
11
|
+
each using its existing sign-in. No API keys were configured.
|
|
12
|
+
|
|
13
|
+
| Flow | Result |
|
|
14
|
+
| --- | --- |
|
|
15
|
+
| Startup discovery | All three CLIs connected; live model lists shown (Codex 4 models, Antigravity 14, Claude aliases) |
|
|
16
|
+
| Chat turn per provider | Claude Code, Codex, and Antigravity each answered; text streamed into the Live view |
|
|
17
|
+
| Context across `/switch` | Codex and Antigravity both recalled the answer given by the previous provider |
|
|
18
|
+
| Codex approval request | A write command outside the read-only sandbox produced a permission prompt; denial left the workspace unchanged |
|
|
19
|
+
| Ctrl+C during a response | Turn cancelled, child process terminated, REPL continued, unanswered message removed |
|
|
20
|
+
| `/session save` and `/session load` | Session with a native provider saved and reloaded in a fresh run |
|
|
21
|
+
| `/models`, `/models <provider>`, `/models <tier>` | Live catalogs listed with `CLI account` in place of API pricing |
|
|
22
|
+
| `thwip --version`, `--help`, `--project` | Handled without starting the REPL; invalid project exits with code 2 |
|
|
23
|
+
| `/limits` and `/status` usage windows | Codex and Claude Code account windows (5h, 7d) displayed with reset times after live turns |
|
|
24
|
+
| Leftover processes | None after each run |
|
|
25
|
+
|
|
26
|
+
The defect that blocked the previous attempt was the Codex sandbox value: the
|
|
27
|
+
adapter sent `readOnly` and Codex rejected `thread/start` with an invalid-request
|
|
28
|
+
error. The protocol enum is `read-only`. A regression test now checks the request.
|
|
29
|
+
|
|
30
|
+
Remaining limitations: the Antigravity CLI showed intermittent network resets to
|
|
31
|
+
Google's backend during testing, which surface as turn errors; Claude Code's model
|
|
32
|
+
aliases are a curated list because the CLI exposes no model listing; native usage
|
|
33
|
+
limit failover is unit-tested from error text, not observed live; the real Gemini
|
|
34
|
+
CLI ACP path is covered by mocked tests only because it is not installed here.
|
|
35
|
+
|
|
36
|
+
| Area | Evidence | Remaining limitation |
|
|
37
|
+
| --- | --- | --- |
|
|
38
|
+
| Commands and aliases | Offline command smoke tests, invalid input cases | Most smoke tests check exceptions, not all rendered content |
|
|
39
|
+
| Sessions | Save/load, separate fresh conversations, malformed metadata, permissions, project rebinding | Concurrent writes to the same explicitly named session are not coordinated |
|
|
40
|
+
| Provider switching and handoff | All seven provider catalogs, portable history, bounded failover | Live account quotas and model availability unverified |
|
|
41
|
+
| Tool execution | Real temporary-file operations, path containment, process timeout/cancellation, invalid arguments | Shell and code tools retain local user privileges; output capture memory is unbounded |
|
|
42
|
+
| Native Codex launcher | Save-before-launch, consent, flags, missing binary, terminal checks, failures | Mocked process replacement; no conversation transfer |
|
|
43
|
+
| Native CLI connections | Live REPL runs above; mocked protocol tests for approvals, failures, limits, prompt building, discovery parsing | Gemini ACP path mocked only; live limit failover not observed |
|
|
44
|
+
| Provider responses | Mocked native tool continuations; DeepSeek/Groq/OpenRouter streaming, usage-only chunks, 429/500 errors and serialized tool arguments | Other streaming and error branches still have coverage gaps |
|
|
45
|
+
| Display/config/auth | Configuration validation, display settings, credential boundaries, short-key masking | No full terminal/platform matrix |
|
|
46
|
+
| Usage | Atomic writes, malformed records, valid totals | Unknown catalog pricing may appear as zero estimated cost |
|
|
47
|
+
| Website | Production build and deterministic demo completion/replay test | Real browser, layout, clipboard, keyboard and accessibility checks remain incomplete |
|
|
48
|
+
| Dependencies | npm audit: zero advisories; Python installed-dependency audit: none found | Python audit skipped legacy local `thwip 1.0.0` metadata |
|
|
49
|
+
| Packaging | Wheel/source build and metadata checks | Publication is separately verified through the release workflow |
|
|
50
|
+
|
|
51
|
+
The Python suite measured 69% statement coverage with 160 passing tests before
|
|
52
|
+
the final credential-masking tests were added. Coverage is diagnostic evidence,
|
|
53
|
+
not proof that every feature works. The website test uses a minimal DOM stand-in
|
|
54
|
+
and controlled timers, not a browser.
|
|
55
|
+
|
|
56
|
+
This pass fixed default-session save collisions, malformed session/message
|
|
57
|
+
metadata acceptance, tool-argument display crashes, Rich markup interpretation
|
|
58
|
+
in action output, and short-key masking leakage.
|
|
@@ -0,0 +1,205 @@
|
|
|
1
|
+
import json
|
|
2
|
+
from io import StringIO
|
|
3
|
+
from types import SimpleNamespace
|
|
4
|
+
|
|
5
|
+
import pytest
|
|
6
|
+
from rich.console import Console
|
|
7
|
+
|
|
8
|
+
from thwip.agents import AgentRegistry
|
|
9
|
+
from thwip.agents.base import AgentDone, Capability, LimitHit, LimitStatus, TextDelta
|
|
10
|
+
from thwip.cli import ThwipCLI
|
|
11
|
+
from thwip.config import ThwipConfig, get_usage_path
|
|
12
|
+
from thwip.limits import UsageTracker
|
|
13
|
+
from thwip.session import Session
|
|
14
|
+
from thwip.tools import ToolManager
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
@pytest.fixture
|
|
18
|
+
def cli(tmp_path, monkeypatch):
|
|
19
|
+
monkeypatch.setenv('THWIP_CONFIG_DIR', str(tmp_path / 'config'))
|
|
20
|
+
cli = ThwipCLI.__new__(ThwipCLI)
|
|
21
|
+
cli.config = ThwipConfig(project=str(tmp_path), auto_save=False)
|
|
22
|
+
cli.registry = AgentRegistry(cli.config)
|
|
23
|
+
async def no_native_discovery(project):
|
|
24
|
+
return None
|
|
25
|
+
monkeypatch.setattr(cli.registry, 'connect_native_agents', no_native_discovery)
|
|
26
|
+
for a in cli.registry.list_agents():
|
|
27
|
+
monkeypatch.setattr(a, 'is_installed', lambda: True)
|
|
28
|
+
monkeypatch.setattr(a, 'is_configured', lambda: False)
|
|
29
|
+
if a.name == 'ollama':
|
|
30
|
+
a._cached_models = a.get_handoff_models()
|
|
31
|
+
cli.current_agent = cli.registry.get_agent('claude')
|
|
32
|
+
cli.session = Session(project_path=str(tmp_path), current_agent='claude', current_model=cli.current_agent.get_default_model())
|
|
33
|
+
cli.usage_tracker = UsageTracker()
|
|
34
|
+
cli.tool_manager = ToolManager(str(tmp_path))
|
|
35
|
+
cli.detector = SimpleNamespace(scan_all=list)
|
|
36
|
+
monkeypatch.setattr('builtins.input', lambda *args: '')
|
|
37
|
+
monkeypatch.setattr('getpass.getpass', lambda *args: '')
|
|
38
|
+
console = Console(file=StringIO(), width=100, color_system=None)
|
|
39
|
+
monkeypatch.setattr('thwip.cli.console', console)
|
|
40
|
+
monkeypatch.setattr('thwip.theme.console', console)
|
|
41
|
+
return cli
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
@pytest.mark.parametrize('command', [
|
|
45
|
+
'/help', '/h', '/about', '/guide', '/g', '/info', '/agents', '/a', '/list',
|
|
46
|
+
'/models', '/m', '/models flagship', '/models balanced', '/models fast', '/models google',
|
|
47
|
+
'/tools', '/t', '/status', '/limits', '/detect', '/history', '/clear', '/reset', '/cost',
|
|
48
|
+
'/project', '/session save audit', '/session load missing', '/session list', '/session clear',
|
|
49
|
+
'/key', '/key google', '/key invalid', '/key openai REJECTED_TEST_KEY',
|
|
50
|
+
'/handoff', '/handoff invalid', '/handoff google invalid', '/switch', '/switch invalid',
|
|
51
|
+
'/switch google invalid', '/unknown', '/q', '/exit', '/quit',
|
|
52
|
+
])
|
|
53
|
+
@pytest.mark.asyncio
|
|
54
|
+
async def test_command_no_exception(cli, command):
|
|
55
|
+
await cli.handle_command(command)
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
@pytest.mark.parametrize('provider', ['claude','google','openai','deepseek','groq','ollama','openrouter'])
|
|
59
|
+
@pytest.mark.asyncio
|
|
60
|
+
async def test_all_catalogued_handoffs(cli, provider):
|
|
61
|
+
before = list(cli.session.messages)
|
|
62
|
+
for model in cli.registry.get_agent(provider).get_handoff_models():
|
|
63
|
+
await cli.handle_command(f'/handoff {provider} {model.id}')
|
|
64
|
+
assert cli.session.messages == before
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
@pytest.mark.asyncio
|
|
68
|
+
async def test_project_path_with_spaces(cli, tmp_path):
|
|
69
|
+
project = tmp_path / 'project with spaces'
|
|
70
|
+
project.mkdir()
|
|
71
|
+
await cli.handle_command(f'/project {project}')
|
|
72
|
+
assert cli.session.project_path == str(project)
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def test_usage_file_private(cli):
|
|
76
|
+
cli.usage_tracker.record_usage('openai', 'gpt-5.6-terra', 10, 5)
|
|
77
|
+
assert get_usage_path().stat().st_mode & 0o777 == 0o600
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
@pytest.mark.parametrize('tool,args', [
|
|
81
|
+
('read_file', {'file_path': 12}),
|
|
82
|
+
('list_files', {'sub_dir': ['bad']}),
|
|
83
|
+
('run_python', {'code': None}),
|
|
84
|
+
])
|
|
85
|
+
def test_invalid_tool_args_return_error_not_exception(cli, tool, args):
|
|
86
|
+
result = cli.tool_manager.execute_tool(tool, args)
|
|
87
|
+
assert isinstance(result, str)
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
@pytest.mark.asyncio
|
|
91
|
+
async def test_history_displays_literal_markup(cli):
|
|
92
|
+
cli.session.add_user_message('Show [/not-a-tag] literally')
|
|
93
|
+
await cli.handle_command('/history')
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
@pytest.mark.parametrize('value', ['wrong', None, -1, True])
|
|
97
|
+
def test_corrupted_handoff_counter_rejected_or_sanitized(cli, value):
|
|
98
|
+
path = cli.session.save('damaged')
|
|
99
|
+
data = json.loads(path.read_text())
|
|
100
|
+
data['observed_tool_results'] = value
|
|
101
|
+
path.write_text(json.dumps(data))
|
|
102
|
+
loaded = Session.load('damaged')
|
|
103
|
+
assert loaded is None or (type(loaded.observed_tool_results) is int and loaded.observed_tool_results >= 0)
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def test_invalid_config_type_rejected_or_defaulted(cli):
|
|
107
|
+
cli.config._apply_toml({'defaults': {'project': 123, 'confirm_tools': 'false'}})
|
|
108
|
+
assert isinstance(cli.config.project, str)
|
|
109
|
+
assert type(cli.config.confirm_tools) is bool
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
@pytest.mark.asyncio
|
|
113
|
+
async def test_session_roundtrip_rebinds_tools(cli, tmp_path):
|
|
114
|
+
target = tmp_path / 'other'
|
|
115
|
+
target.mkdir()
|
|
116
|
+
saved = Session(project_path=str(target), current_agent='google', current_model='missing')
|
|
117
|
+
saved.save('loadable')
|
|
118
|
+
await cli.handle_command('/session load loadable')
|
|
119
|
+
assert cli.tool_manager.file_editor.project_path == target
|
|
120
|
+
assert cli.current_agent.name == 'google'
|
|
121
|
+
assert cli.session.current_model == cli.current_agent.get_default_model()
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def test_all_tools_and_safety(cli, tmp_path):
|
|
125
|
+
tm = cli.tool_manager
|
|
126
|
+
assert 'Successfully' in tm.execute_tool('write_file', {'file_path':'a.txt','content':'one'})
|
|
127
|
+
assert tm.execute_tool('read_file', {'file_path':'a.txt'}) == 'one'
|
|
128
|
+
assert 'Successfully' in tm.execute_tool('edit_file', {'file_path':'a.txt','old_str':'one','new_str':'two'})
|
|
129
|
+
assert 'a.txt' in tm.execute_tool('list_files', {})
|
|
130
|
+
assert '42' in tm.execute_tool('run_python', {'code':'print(6 * 7)'})
|
|
131
|
+
assert 'Exit code: 0' in tm.execute_tool('run_command', {'command':'printf audit-ok'})
|
|
132
|
+
assert isinstance(tm.execute_tool('git_status', {}), str)
|
|
133
|
+
assert isinstance(tm.execute_tool('git_diff', {}), str)
|
|
134
|
+
assert 'outside' in tm.execute_tool('read_file', {'file_path':'../outside'})
|
|
135
|
+
outside = tmp_path.parent / 'audit-outside-file'
|
|
136
|
+
outside.write_text('not accessible')
|
|
137
|
+
(tmp_path / 'link').symlink_to(outside)
|
|
138
|
+
assert 'outside' in tm.execute_tool('read_file', {'file_path':'link'})
|
|
139
|
+
assert 'Error' in tm.execute_tool('unknown', {})
|
|
140
|
+
assert {t['name'] for t in tm.get_anthropic_tools()} == {t['function']['name'] for t in tm.get_openai_tools()}
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
@pytest.mark.asyncio
|
|
144
|
+
async def test_failover_does_not_retry_already_failed_provider(cli, monkeypatch):
|
|
145
|
+
agents = [cli.registry.get_agent('claude'), cli.registry.get_agent('google')]
|
|
146
|
+
attempts = []
|
|
147
|
+
for agent in agents:
|
|
148
|
+
monkeypatch.setattr(agent, 'is_configured', lambda: True)
|
|
149
|
+
monkeypatch.setattr(agent, 'get_capabilities_for_model', lambda model: {Capability.CHAT})
|
|
150
|
+
|
|
151
|
+
def make_chat(name):
|
|
152
|
+
async def chat(**kwargs):
|
|
153
|
+
attempts.append(name)
|
|
154
|
+
# Stop safely after four simulated rate limits; no actual network calls.
|
|
155
|
+
if len(attempts) <= 4:
|
|
156
|
+
yield LimitHit(error_type=LimitStatus.RATE_LIMITED, message='test quota')
|
|
157
|
+
else:
|
|
158
|
+
yield TextDelta(content='stop test')
|
|
159
|
+
yield AgentDone()
|
|
160
|
+
return chat
|
|
161
|
+
|
|
162
|
+
monkeypatch.setattr(agent, 'chat', make_chat(agent.name))
|
|
163
|
+
monkeypatch.setattr(cli.registry, 'get_ready_agents', lambda: agents)
|
|
164
|
+
cli.config.fallback.chain = ['claude', 'google']
|
|
165
|
+
cli.config.limits.auto_switch = True
|
|
166
|
+
await cli.process_user_message('hello')
|
|
167
|
+
assert len(attempts) <= len(agents), f'Repeated failed providers: {attempts}'
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def test_main_handles_version_and_help_without_starting_the_repl(capsys):
|
|
171
|
+
from thwip import __version__
|
|
172
|
+
from thwip import cli as cli_module
|
|
173
|
+
|
|
174
|
+
with pytest.raises(SystemExit) as exit_info:
|
|
175
|
+
cli_module.main(['--version'])
|
|
176
|
+
assert exit_info.value.code == 0 and f'thwip {__version__}' in capsys.readouterr().out
|
|
177
|
+
with pytest.raises(SystemExit) as exit_info:
|
|
178
|
+
cli_module.main(['--help'])
|
|
179
|
+
assert exit_info.value.code == 0 and '--project' in capsys.readouterr().out
|
|
180
|
+
with pytest.raises(SystemExit) as exit_info:
|
|
181
|
+
cli_module.main(['--project', '/definitely/missing/dir'])
|
|
182
|
+
assert exit_info.value.code == 2
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
@pytest.mark.asyncio
|
|
186
|
+
async def test_failed_turn_removes_unanswered_user_message(cli, monkeypatch):
|
|
187
|
+
class Broken:
|
|
188
|
+
name = 'openai'
|
|
189
|
+
display_name = 'Broken'
|
|
190
|
+
company = 'OpenAI'
|
|
191
|
+
native_tools = True
|
|
192
|
+
project = '.'
|
|
193
|
+
def is_configured(self):
|
|
194
|
+
return True
|
|
195
|
+
def is_installed(self):
|
|
196
|
+
return True
|
|
197
|
+
def get_capabilities_for_model(self, model):
|
|
198
|
+
return set()
|
|
199
|
+
async def chat(self, **kwargs):
|
|
200
|
+
raise RuntimeError('Native CLI rejected thread/start')
|
|
201
|
+
yield
|
|
202
|
+
cli.current_agent = Broken()
|
|
203
|
+
cli.session.current_agent = 'openai'
|
|
204
|
+
await cli.process_user_message('hello')
|
|
205
|
+
assert cli.session.messages == []
|
|
@@ -75,6 +75,30 @@ class UnconfiguredAgent(ToolCallingAgent):
|
|
|
75
75
|
return False
|
|
76
76
|
|
|
77
77
|
|
|
78
|
+
@pytest.mark.asyncio
|
|
79
|
+
@pytest.mark.parametrize("arguments", [None, [], "bad", {"file_path": "[broken]"}])
|
|
80
|
+
async def test_invalid_tool_arguments_return_errors_without_crashing(tmp_path, arguments):
|
|
81
|
+
class InvalidToolAgent(ToolCallingAgent):
|
|
82
|
+
async def chat(self, messages, **kwargs):
|
|
83
|
+
self.calls.append(messages)
|
|
84
|
+
if len(self.calls) == 1:
|
|
85
|
+
yield ToolUseStart(tool_id="bad", tool_name="read_file", args=arguments)
|
|
86
|
+
yield AgentDone()
|
|
87
|
+
else:
|
|
88
|
+
assert "Error" in messages[-1]["content"]
|
|
89
|
+
yield TextDelta(content="Recovered from invalid tool arguments.")
|
|
90
|
+
yield AgentDone()
|
|
91
|
+
|
|
92
|
+
cli = ThwipCLI.__new__(ThwipCLI)
|
|
93
|
+
cli.config = SimpleNamespace(stream=True, confirm_tools=True)
|
|
94
|
+
cli.session = Session(current_agent="fake", current_model="fake-model")
|
|
95
|
+
cli.current_agent = InvalidToolAgent()
|
|
96
|
+
cli.tool_manager = ToolManager(tmp_path)
|
|
97
|
+
cli.usage_tracker = FakeUsageTracker()
|
|
98
|
+
await cli.process_user_message("Read")
|
|
99
|
+
assert cli.session.messages[-1].content == "Recovered from invalid tool arguments."
|
|
100
|
+
|
|
101
|
+
|
|
78
102
|
@pytest.mark.asyncio
|
|
79
103
|
async def test_unconfigured_agent_shows_setup_guidance(tmp_path):
|
|
80
104
|
cli = ThwipCLI.__new__(ThwipCLI)
|
|
@@ -91,11 +115,12 @@ async def test_unconfigured_agent_shows_setup_guidance(tmp_path):
|
|
|
91
115
|
assert len(cli.session.messages) == 0
|
|
92
116
|
|
|
93
117
|
|
|
94
|
-
|
|
118
|
+
@pytest.mark.asyncio
|
|
119
|
+
async def test_inline_api_key_is_rejected():
|
|
95
120
|
cli = ThwipCLI.__new__(ThwipCLI)
|
|
96
121
|
cli.config = SimpleNamespace(keys={}, key_sources={}, save=lambda: None)
|
|
97
122
|
|
|
98
|
-
cli.cmd_auth_config("openai", "secret-value")
|
|
123
|
+
await cli.cmd_auth_config("openai", "secret-value")
|
|
99
124
|
|
|
100
125
|
assert cli.config.keys == {}
|
|
101
126
|
assert cli.config.key_sources == {}
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
"""Exercise streaming completion and provider error boundaries without network calls."""
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
from types import SimpleNamespace
|
|
5
|
+
|
|
6
|
+
import httpx
|
|
7
|
+
import openai
|
|
8
|
+
import pytest
|
|
9
|
+
|
|
10
|
+
from thwip.agents.base import AgentDone, LimitHit, LimitStatus, TextDelta
|
|
11
|
+
from thwip.agents.deepseek_agent import DeepSeekAgent
|
|
12
|
+
from thwip.agents.groq_agent import GroqAgent
|
|
13
|
+
from thwip.agents.openrouter_agent import OpenRouterAgent
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
@pytest.mark.parametrize("agent_type", [DeepSeekAgent, GroqAgent, OpenRouterAgent])
|
|
17
|
+
@pytest.mark.asyncio
|
|
18
|
+
async def test_tool_continuation_serializes_arguments_without_mutating_history(agent_type):
|
|
19
|
+
original = {"role": "assistant", "content": "", "tool_calls": [{
|
|
20
|
+
"id": "call1", "type": "function", "function": {
|
|
21
|
+
"name": "read_file", "arguments": {"file_path": "note.txt"},
|
|
22
|
+
},
|
|
23
|
+
}], "_native_state": {"other": "private"}}
|
|
24
|
+
|
|
25
|
+
async def create(**kwargs):
|
|
26
|
+
sent = kwargs["messages"][0]
|
|
27
|
+
assert "_native_state" not in sent
|
|
28
|
+
assert json.loads(sent["tool_calls"][0]["function"]["arguments"]) == {"file_path": "note.txt"}
|
|
29
|
+
return SimpleNamespace(choices=[SimpleNamespace(message=SimpleNamespace(content="done", tool_calls=[]))], usage=None)
|
|
30
|
+
|
|
31
|
+
agent = agent_type(api_key="test")
|
|
32
|
+
agent._client = SimpleNamespace(chat=SimpleNamespace(completions=SimpleNamespace(create=create)))
|
|
33
|
+
events = [event async for event in agent.chat([original], stream=False)]
|
|
34
|
+
assert any(isinstance(event, AgentDone) for event in events)
|
|
35
|
+
assert isinstance(original["tool_calls"][0]["function"]["arguments"], dict)
|
|
36
|
+
assert "_native_state" in original
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
@pytest.mark.parametrize("agent_type", [DeepSeekAgent, GroqAgent, OpenRouterAgent])
|
|
40
|
+
@pytest.mark.asyncio
|
|
41
|
+
async def test_stream_text_and_usage_only_final_chunk(agent_type):
|
|
42
|
+
async def chunks():
|
|
43
|
+
yield SimpleNamespace(choices=[SimpleNamespace(delta=SimpleNamespace(
|
|
44
|
+
content="hello", tool_calls=None))], usage=None)
|
|
45
|
+
yield SimpleNamespace(choices=[], usage=SimpleNamespace(prompt_tokens=9, completion_tokens=3))
|
|
46
|
+
|
|
47
|
+
async def create(**kwargs):
|
|
48
|
+
assert kwargs["stream"] is True
|
|
49
|
+
return chunks()
|
|
50
|
+
|
|
51
|
+
agent = agent_type(api_key="test")
|
|
52
|
+
agent._client = SimpleNamespace(chat=SimpleNamespace(completions=SimpleNamespace(create=create)))
|
|
53
|
+
events = [event async for event in agent.chat([{"role": "user", "content": "hi"}])]
|
|
54
|
+
assert isinstance(events[0], TextDelta) and events[0].content == "hello"
|
|
55
|
+
assert isinstance(events[-1], AgentDone)
|
|
56
|
+
assert events[-1].usage.input_tokens == 9
|
|
57
|
+
assert events[-1].usage.output_tokens == 3
|
|
58
|
+
assert agent.check_limits() == LimitStatus.OK
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
@pytest.mark.parametrize("agent_type", [DeepSeekAgent, GroqAgent, OpenRouterAgent])
|
|
62
|
+
@pytest.mark.parametrize("status", [429, 500])
|
|
63
|
+
@pytest.mark.asyncio
|
|
64
|
+
async def test_provider_errors_emit_failure_not_success(agent_type, status):
|
|
65
|
+
response = httpx.Response(status, request=httpx.Request("POST", "https://example.invalid"))
|
|
66
|
+
error_type = openai.RateLimitError if status == 429 else openai.APIStatusError
|
|
67
|
+
|
|
68
|
+
async def create(**kwargs):
|
|
69
|
+
raise error_type("simulated failure", response=response, body=None)
|
|
70
|
+
|
|
71
|
+
agent = agent_type(api_key="test")
|
|
72
|
+
agent._client = SimpleNamespace(chat=SimpleNamespace(completions=SimpleNamespace(create=create)))
|
|
73
|
+
events = [event async for event in agent.chat([])]
|
|
74
|
+
assert len(events) == 1 and isinstance(events[0], LimitHit)
|
|
75
|
+
assert events[0].error_type == (LimitStatus.RATE_LIMITED if status == 429 else LimitStatus.UNKNOWN)
|