thwip-cli 1.3.0__tar.gz → 1.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {thwip_cli-1.3.0 → thwip_cli-1.4.0}/PKG-INFO +50 -24
- {thwip_cli-1.3.0 → thwip_cli-1.4.0}/README.md +49 -23
- thwip_cli-1.4.0/docs/verification.md +58 -0
- {thwip_cli-1.3.0 → thwip_cli-1.4.0}/pyproject.toml +1 -1
- {thwip_cli-1.3.0 → thwip_cli-1.4.0}/tests/test_audit_regressions.py +41 -0
- {thwip_cli-1.3.0 → thwip_cli-1.4.0}/tests/test_cli.py +3 -2
- thwip_cli-1.4.0/tests/test_native_agents.py +198 -0
- thwip_cli-1.4.0/tests/test_native_print.py +234 -0
- thwip_cli-1.4.0/tests/test_native_rpc.py +51 -0
- {thwip_cli-1.3.0 → thwip_cli-1.4.0}/thwip/__init__.py +1 -1
- {thwip_cli-1.3.0 → thwip_cli-1.4.0}/thwip/agents/__init__.py +42 -0
- {thwip_cli-1.3.0 → thwip_cli-1.4.0}/thwip/agents/base.py +14 -1
- thwip_cli-1.4.0/thwip/agents/native_agent.py +281 -0
- thwip_cli-1.4.0/thwip/agents/native_common.py +102 -0
- thwip_cli-1.4.0/thwip/agents/native_print.py +327 -0
- thwip_cli-1.4.0/thwip/agents/native_rpc.py +86 -0
- {thwip_cli-1.3.0 → thwip_cli-1.4.0}/thwip/cli.py +149 -19
- {thwip_cli-1.3.0 → thwip_cli-1.4.0}/thwip/handoff.py +3 -1
- {thwip_cli-1.3.0 → thwip_cli-1.4.0}/uv.lock +1 -1
- thwip_cli-1.3.0/docs/verification.md +0 -28
- {thwip_cli-1.3.0 → thwip_cli-1.4.0}/.github/workflows/publish.yml +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.4.0}/.gitignore +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.4.0}/LICENSE +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.4.0}/docs/handoff-research.md +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.4.0}/install.sh +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.4.0}/tests/test_agents.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.4.0}/tests/test_compatible_streaming.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.4.0}/tests/test_config.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.4.0}/tests/test_detector.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.4.0}/tests/test_handoff.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.4.0}/tests/test_native_launcher.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.4.0}/tests/test_repair_verification.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.4.0}/tests/test_session.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.4.0}/tests/test_tools.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.4.0}/tests/test_utils.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.4.0}/thwip/__main__.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.4.0}/thwip/agents/chat_messages.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.4.0}/thwip/agents/claude_agent.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.4.0}/thwip/agents/deepseek_agent.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.4.0}/thwip/agents/google_agent.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.4.0}/thwip/agents/groq_agent.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.4.0}/thwip/agents/ollama_agent.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.4.0}/thwip/agents/openai_agent.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.4.0}/thwip/agents/openrouter_agent.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.4.0}/thwip/config.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.4.0}/thwip/detector.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.4.0}/thwip/limits.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.4.0}/thwip/session.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.4.0}/thwip/shortcuts.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.4.0}/thwip/theme.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.4.0}/thwip/tools/__init__.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.4.0}/thwip/tools/code_runner.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.4.0}/thwip/tools/file_editor.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.4.0}/thwip/tools/git_ops.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.4.0}/thwip/tools/terminal.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.4.0}/thwip/utils.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: thwip-cli
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.4.0
|
|
4
4
|
Summary: Universal coding agent multiplexer: detect, switch, and route between AI coding agents seamlessly
|
|
5
5
|
Project-URL: Homepage, https://github.com/tanmayhutt/thwip-cli
|
|
6
6
|
Project-URL: Repository, https://github.com/tanmayhutt/thwip-cli
|
|
@@ -46,6 +46,7 @@ Description-Content-Type: text/markdown
|
|
|
46
46
|
## Features
|
|
47
47
|
|
|
48
48
|
- **Auto-Detection**: Discovers installed AI coding agents (Claude Code, Antigravity, Gemini CLI, OpenAI/Codex, Aider, Copilot, Cursor, Windsurf, Cline, Ollama) and configured credentials.
|
|
49
|
+
- **Existing Sign-ins**: Chats through installed Codex, Claude Code, and Antigravity CLIs using their own logins and live model lists, with no API key required.
|
|
49
50
|
- **Context Portability**: Switch providers with stored conversational text. Working files remain in the selected local project; they are not automatically uploaded.
|
|
50
51
|
- **Handoff Preview**: Inspect text continuity, omitted state, capability changes, approximate context pressure, and a text fingerprint before switching. Runs locally without model calls.
|
|
51
52
|
- **Dynamic UI**: Terminal interface adapts its status bar, capabilities, and theme based on the active provider.
|
|
@@ -81,6 +82,8 @@ Start the interactive terminal in your current project directory:
|
|
|
81
82
|
|
|
82
83
|
```bash
|
|
83
84
|
thwip
|
|
85
|
+
thwip --project ~/code/my-app # open a specific project
|
|
86
|
+
thwip --version
|
|
84
87
|
```
|
|
85
88
|
|
|
86
89
|
---
|
|
@@ -91,13 +94,13 @@ thwip
|
|
|
91
94
|
|:---|:---|
|
|
92
95
|
| `/switch [agent] [model]` | Switch active agent or model mid-conversation |
|
|
93
96
|
| `/handoff [agent] [model]` | Preview a target locally without switching or sending data |
|
|
94
|
-
| `/native codex` | Save and leave Thwip for the installed Codex CLI
|
|
97
|
+
| `/native codex` | Save and leave Thwip for the installed Codex CLI itself (launcher, no context transfer) |
|
|
95
98
|
| `/agents` | Show all detected coding agents, company status, and capabilities |
|
|
96
99
|
| `/models [agent]` | List available models for current or target agent |
|
|
97
100
|
| `/key [provider]` | Enter an API key securely without placing it in prompt history |
|
|
98
101
|
| `/status` | Display current session, project, and token stats |
|
|
99
102
|
| `/limits` | View token usage, quota, and spend metrics |
|
|
100
|
-
| `/detect` | Re-scan
|
|
103
|
+
| `/detect` | Re-scan installed agents and reconnect CLI sign-ins |
|
|
101
104
|
| `/session save [name]` | Save current chat session |
|
|
102
105
|
| `/session load <name>` | Load a previously saved session |
|
|
103
106
|
| `/session list` | List all saved sessions |
|
|
@@ -110,23 +113,46 @@ thwip
|
|
|
110
113
|
| `Ctrl + H` | View history |
|
|
111
114
|
| `/quit` | Exit thwip |
|
|
112
115
|
|
|
113
|
-
Short aliases are available for frequent commands: `/a`, `/m`, `/s`, `/k`, `/g`, and `/t`.
|
|
114
|
-
|
|
115
|
-
### Existing
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
116
|
+
Short aliases are available for frequent commands: `/a`, `/m`, `/s`, `/sw`, `/k`, `/g`, and `/t`.
|
|
117
|
+
|
|
118
|
+
### Existing CLI sign-ins (no API key needed)
|
|
119
|
+
|
|
120
|
+
If Codex, Claude Code, or the Antigravity CLI is installed and signed in, Thwip
|
|
121
|
+
connects to it at startup and uses that sign-in for chat. No API key is copied or
|
|
122
|
+
required, and Thwip never reads the CLI's stored credentials. A configured direct
|
|
123
|
+
API key for the same provider always takes precedence over the native connection.
|
|
124
|
+
|
|
125
|
+
| Provider | Installed CLI | Transport | Model list |
|
|
126
|
+
|:---|:---|:---|:---|
|
|
127
|
+
| OpenAI | `codex` | Codex App Server (JSON-RPC over stdio) | Live from `model/list` |
|
|
128
|
+
| Anthropic | `claude` | Claude Code print mode (`stream-json`) | Aliases `fable`, `opus`, `sonnet`, `haiku`; explicit IDs pass through |
|
|
129
|
+
| Google | `agy` (Antigravity CLI) or `gemini` | Antigravity print mode (`stream-json`) or Gemini ACP | Live from `agy models` or the ACP session |
|
|
130
|
+
|
|
131
|
+
`/models` refreshes the catalog supplied by each CLI. An explicit model ID that is
|
|
132
|
+
not in the catalog is passed to the CLI for validation instead of being rejected by
|
|
133
|
+
a bundled list. A model available in a desktop app may still be absent from the
|
|
134
|
+
installed CLI's catalog or account access.
|
|
135
|
+
|
|
136
|
+
Native connections are read-only by default. Codex starts with a read-only sandbox
|
|
137
|
+
and asks before operations outside it; approval requests appear in Thwip with the
|
|
138
|
+
command or file list and default to denial. Claude Code and the Antigravity CLI run
|
|
139
|
+
in their non-interactive print modes, where tools that would need an approval are
|
|
140
|
+
declined by the CLI itself. Each turn sends the portable text conversation to a
|
|
141
|
+
fresh native session, so native reasoning and tool state do not carry between turns.
|
|
142
|
+
Ctrl+C interrupts the current turn and stops the child process; the unanswered
|
|
143
|
+
message is removed so it can be re-sent or handed to another provider.
|
|
144
|
+
|
|
145
|
+
Native billing and usage limits are managed by each CLI account. Thwip records the
|
|
146
|
+
token counts the CLIs report but does not estimate cost for them. Codex and Claude
|
|
147
|
+
Code also report their account usage windows (5 hour and 7 day) after each response;
|
|
148
|
+
`/limits` and `/status` show the used percentage and reset time. When a CLI reports
|
|
149
|
+
an exhausted usage limit, the standard failover prompt offers the other connected
|
|
150
|
+
providers.
|
|
151
|
+
|
|
152
|
+
`/native codex` remains available as a launcher: it saves the Thwip session and
|
|
153
|
+
replaces Thwip with the Codex CLI itself in the selected project. Conversation
|
|
154
|
+
history is not transferred by the launcher; use `/session load` after restarting
|
|
155
|
+
Thwip to resume.
|
|
130
156
|
|
|
131
157
|
## Auditable handoffs
|
|
132
158
|
|
|
@@ -165,9 +191,9 @@ See [research and prior art](docs/handoff-research.md) for the differentiation r
|
|
|
165
191
|
|
|
166
192
|
| Company | Agent | Capabilities |
|
|
167
193
|
|:---|:---|:---|
|
|
168
|
-
| Anthropic | Claude API (Fable 5, Opus 5, Sonnet 5, Haiku 4.5) | Chat, File Edit, Code Run, Terminal, Git |
|
|
169
|
-
| Google | Gemini API (3.1 Pro Preview, 3.7 Flash, 3.5 Flash-Lite) | Chat, File Edit, Code Run, Terminal, Git |
|
|
170
|
-
| OpenAI | OpenAI API (GPT-5.6 Sol, Terra, Luna) | Chat, File Edit, Code Run, Terminal, Git |
|
|
194
|
+
| Anthropic | Claude Code sign-in, or Claude API (Fable 5, Opus 5, Sonnet 5, Haiku 4.5) | Chat, File Edit, Code Run, Terminal, Git |
|
|
195
|
+
| Google | Antigravity or Gemini CLI sign-in, or Gemini API (3.1 Pro Preview, 3.7 Flash, 3.5 Flash-Lite) | Chat, File Edit, Code Run, Terminal, Git |
|
|
196
|
+
| OpenAI | Codex CLI sign-in, or OpenAI API (GPT-5.6 Sol, Terra, Luna) | Chat, File Edit, Code Run, Terminal, Git |
|
|
171
197
|
| DeepSeek | DeepSeek V3 / R1 Reasoner | Chat, File Edit, Code Run, Reasoning |
|
|
172
198
|
| Groq | GPT-OSS 120B (default); Llama 3.3 for eligible enterprise accounts only | Chat, File Edit, Code Run |
|
|
173
199
|
| Ollama | Local Models (Llama 3.3, Qwen Coder, DeepSeek R1) | Chat, File Edit, Code Run (Local, Offline) |
|
|
@@ -179,7 +205,7 @@ See [research and prior art](docs/handoff-research.md) for the differentiation r
|
|
|
179
205
|
|
|
180
206
|
thwip auto-detects existing API keys from environment variables and existing agent configs (`~/.claude.json`, `~/.gemini/config.json`). Configuration can also be set manually:
|
|
181
207
|
|
|
182
|
-
Installed
|
|
208
|
+
Installed CLI sign-ins and API access are separate paths. A signed-in Codex, Claude Code, or Antigravity CLI is used directly through its own protocol (see above). The direct SDK adapters for Anthropic, Google, OpenAI, DeepSeek, Groq, and OpenRouter require a provider API key. Ollama needs no key when its local server is running.
|
|
183
209
|
|
|
184
210
|
```toml
|
|
185
211
|
[defaults]
|
|
@@ -9,6 +9,7 @@
|
|
|
9
9
|
## Features
|
|
10
10
|
|
|
11
11
|
- **Auto-Detection**: Discovers installed AI coding agents (Claude Code, Antigravity, Gemini CLI, OpenAI/Codex, Aider, Copilot, Cursor, Windsurf, Cline, Ollama) and configured credentials.
|
|
12
|
+
- **Existing Sign-ins**: Chats through installed Codex, Claude Code, and Antigravity CLIs using their own logins and live model lists, with no API key required.
|
|
12
13
|
- **Context Portability**: Switch providers with stored conversational text. Working files remain in the selected local project; they are not automatically uploaded.
|
|
13
14
|
- **Handoff Preview**: Inspect text continuity, omitted state, capability changes, approximate context pressure, and a text fingerprint before switching. Runs locally without model calls.
|
|
14
15
|
- **Dynamic UI**: Terminal interface adapts its status bar, capabilities, and theme based on the active provider.
|
|
@@ -44,6 +45,8 @@ Start the interactive terminal in your current project directory:
|
|
|
44
45
|
|
|
45
46
|
```bash
|
|
46
47
|
thwip
|
|
48
|
+
thwip --project ~/code/my-app # open a specific project
|
|
49
|
+
thwip --version
|
|
47
50
|
```
|
|
48
51
|
|
|
49
52
|
---
|
|
@@ -54,13 +57,13 @@ thwip
|
|
|
54
57
|
|:---|:---|
|
|
55
58
|
| `/switch [agent] [model]` | Switch active agent or model mid-conversation |
|
|
56
59
|
| `/handoff [agent] [model]` | Preview a target locally without switching or sending data |
|
|
57
|
-
| `/native codex` | Save and leave Thwip for the installed Codex CLI
|
|
60
|
+
| `/native codex` | Save and leave Thwip for the installed Codex CLI itself (launcher, no context transfer) |
|
|
58
61
|
| `/agents` | Show all detected coding agents, company status, and capabilities |
|
|
59
62
|
| `/models [agent]` | List available models for current or target agent |
|
|
60
63
|
| `/key [provider]` | Enter an API key securely without placing it in prompt history |
|
|
61
64
|
| `/status` | Display current session, project, and token stats |
|
|
62
65
|
| `/limits` | View token usage, quota, and spend metrics |
|
|
63
|
-
| `/detect` | Re-scan
|
|
66
|
+
| `/detect` | Re-scan installed agents and reconnect CLI sign-ins |
|
|
64
67
|
| `/session save [name]` | Save current chat session |
|
|
65
68
|
| `/session load <name>` | Load a previously saved session |
|
|
66
69
|
| `/session list` | List all saved sessions |
|
|
@@ -73,23 +76,46 @@ thwip
|
|
|
73
76
|
| `Ctrl + H` | View history |
|
|
74
77
|
| `/quit` | Exit thwip |
|
|
75
78
|
|
|
76
|
-
Short aliases are available for frequent commands: `/a`, `/m`, `/s`, `/k`, `/g`, and `/t`.
|
|
77
|
-
|
|
78
|
-
### Existing
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
79
|
+
Short aliases are available for frequent commands: `/a`, `/m`, `/s`, `/sw`, `/k`, `/g`, and `/t`.
|
|
80
|
+
|
|
81
|
+
### Existing CLI sign-ins (no API key needed)
|
|
82
|
+
|
|
83
|
+
If Codex, Claude Code, or the Antigravity CLI is installed and signed in, Thwip
|
|
84
|
+
connects to it at startup and uses that sign-in for chat. No API key is copied or
|
|
85
|
+
required, and Thwip never reads the CLI's stored credentials. A configured direct
|
|
86
|
+
API key for the same provider always takes precedence over the native connection.
|
|
87
|
+
|
|
88
|
+
| Provider | Installed CLI | Transport | Model list |
|
|
89
|
+
|:---|:---|:---|:---|
|
|
90
|
+
| OpenAI | `codex` | Codex App Server (JSON-RPC over stdio) | Live from `model/list` |
|
|
91
|
+
| Anthropic | `claude` | Claude Code print mode (`stream-json`) | Aliases `fable`, `opus`, `sonnet`, `haiku`; explicit IDs pass through |
|
|
92
|
+
| Google | `agy` (Antigravity CLI) or `gemini` | Antigravity print mode (`stream-json`) or Gemini ACP | Live from `agy models` or the ACP session |
|
|
93
|
+
|
|
94
|
+
`/models` refreshes the catalog supplied by each CLI. An explicit model ID that is
|
|
95
|
+
not in the catalog is passed to the CLI for validation instead of being rejected by
|
|
96
|
+
a bundled list. A model available in a desktop app may still be absent from the
|
|
97
|
+
installed CLI's catalog or account access.
|
|
98
|
+
|
|
99
|
+
Native connections are read-only by default. Codex starts with a read-only sandbox
|
|
100
|
+
and asks before operations outside it; approval requests appear in Thwip with the
|
|
101
|
+
command or file list and default to denial. Claude Code and the Antigravity CLI run
|
|
102
|
+
in their non-interactive print modes, where tools that would need an approval are
|
|
103
|
+
declined by the CLI itself. Each turn sends the portable text conversation to a
|
|
104
|
+
fresh native session, so native reasoning and tool state do not carry between turns.
|
|
105
|
+
Ctrl+C interrupts the current turn and stops the child process; the unanswered
|
|
106
|
+
message is removed so it can be re-sent or handed to another provider.
|
|
107
|
+
|
|
108
|
+
Native billing and usage limits are managed by each CLI account. Thwip records the
|
|
109
|
+
token counts the CLIs report but does not estimate cost for them. Codex and Claude
|
|
110
|
+
Code also report their account usage windows (5 hour and 7 day) after each response;
|
|
111
|
+
`/limits` and `/status` show the used percentage and reset time. When a CLI reports
|
|
112
|
+
an exhausted usage limit, the standard failover prompt offers the other connected
|
|
113
|
+
providers.
|
|
114
|
+
|
|
115
|
+
`/native codex` remains available as a launcher: it saves the Thwip session and
|
|
116
|
+
replaces Thwip with the Codex CLI itself in the selected project. Conversation
|
|
117
|
+
history is not transferred by the launcher; use `/session load` after restarting
|
|
118
|
+
Thwip to resume.
|
|
93
119
|
|
|
94
120
|
## Auditable handoffs
|
|
95
121
|
|
|
@@ -128,9 +154,9 @@ See [research and prior art](docs/handoff-research.md) for the differentiation r
|
|
|
128
154
|
|
|
129
155
|
| Company | Agent | Capabilities |
|
|
130
156
|
|:---|:---|:---|
|
|
131
|
-
| Anthropic | Claude API (Fable 5, Opus 5, Sonnet 5, Haiku 4.5) | Chat, File Edit, Code Run, Terminal, Git |
|
|
132
|
-
| Google | Gemini API (3.1 Pro Preview, 3.7 Flash, 3.5 Flash-Lite) | Chat, File Edit, Code Run, Terminal, Git |
|
|
133
|
-
| OpenAI | OpenAI API (GPT-5.6 Sol, Terra, Luna) | Chat, File Edit, Code Run, Terminal, Git |
|
|
157
|
+
| Anthropic | Claude Code sign-in, or Claude API (Fable 5, Opus 5, Sonnet 5, Haiku 4.5) | Chat, File Edit, Code Run, Terminal, Git |
|
|
158
|
+
| Google | Antigravity or Gemini CLI sign-in, or Gemini API (3.1 Pro Preview, 3.7 Flash, 3.5 Flash-Lite) | Chat, File Edit, Code Run, Terminal, Git |
|
|
159
|
+
| OpenAI | Codex CLI sign-in, or OpenAI API (GPT-5.6 Sol, Terra, Luna) | Chat, File Edit, Code Run, Terminal, Git |
|
|
134
160
|
| DeepSeek | DeepSeek V3 / R1 Reasoner | Chat, File Edit, Code Run, Reasoning |
|
|
135
161
|
| Groq | GPT-OSS 120B (default); Llama 3.3 for eligible enterprise accounts only | Chat, File Edit, Code Run |
|
|
136
162
|
| Ollama | Local Models (Llama 3.3, Qwen Coder, DeepSeek R1) | Chat, File Edit, Code Run (Local, Offline) |
|
|
@@ -142,7 +168,7 @@ See [research and prior art](docs/handoff-research.md) for the differentiation r
|
|
|
142
168
|
|
|
143
169
|
thwip auto-detects existing API keys from environment variables and existing agent configs (`~/.claude.json`, `~/.gemini/config.json`). Configuration can also be set manually:
|
|
144
170
|
|
|
145
|
-
Installed
|
|
171
|
+
Installed CLI sign-ins and API access are separate paths. A signed-in Codex, Claude Code, or Antigravity CLI is used directly through its own protocol (see above). The direct SDK adapters for Anthropic, Google, OpenAI, DeepSeek, Groq, and OpenRouter require a provider API key. Ollama needs no key when its local server is running.
|
|
146
172
|
|
|
147
173
|
```toml
|
|
148
174
|
[defaults]
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
# Verification status
|
|
2
|
+
|
|
3
|
+
Verified locally on 2026-09-24 for the unreleased v1.4.0 source. The Python suite
|
|
4
|
+
passes 216 offline tests. Exhaustive behavior across every provider and
|
|
5
|
+
configuration has not been established.
|
|
6
|
+
|
|
7
|
+
## Native CLI connections (2026-09-24)
|
|
8
|
+
|
|
9
|
+
Live checks were run on macOS through a pseudo-terminal driving the real REPL with
|
|
10
|
+
the installed Codex CLI 0.152.1, Claude Code 2.1.281, and Antigravity CLI 1.2.8,
|
|
11
|
+
each using its existing sign-in. No API keys were configured.
|
|
12
|
+
|
|
13
|
+
| Flow | Result |
|
|
14
|
+
| --- | --- |
|
|
15
|
+
| Startup discovery | All three CLIs connected; live model lists shown (Codex 4 models, Antigravity 14, Claude aliases) |
|
|
16
|
+
| Chat turn per provider | Claude Code, Codex, and Antigravity each answered; text streamed into the Live view |
|
|
17
|
+
| Context across `/switch` | Codex and Antigravity both recalled the answer given by the previous provider |
|
|
18
|
+
| Codex approval request | A write command outside the read-only sandbox produced a permission prompt; denial left the workspace unchanged |
|
|
19
|
+
| Ctrl+C during a response | Turn cancelled, child process terminated, REPL continued, unanswered message removed |
|
|
20
|
+
| `/session save` and `/session load` | Session with a native provider saved and reloaded in a fresh run |
|
|
21
|
+
| `/models`, `/models <provider>`, `/models <tier>` | Live catalogs listed with `CLI account` in place of API pricing |
|
|
22
|
+
| `thwip --version`, `--help`, `--project` | Handled without starting the REPL; invalid project exits with code 2 |
|
|
23
|
+
| `/limits` and `/status` usage windows | Codex and Claude Code account windows (5h, 7d) displayed with reset times after live turns |
|
|
24
|
+
| Leftover processes | None after each run |
|
|
25
|
+
|
|
26
|
+
The defect that blocked the previous attempt was the Codex sandbox value: the
|
|
27
|
+
adapter sent `readOnly` and Codex rejected `thread/start` with an invalid-request
|
|
28
|
+
error. The protocol enum is `read-only`. A regression test now checks the request.
|
|
29
|
+
|
|
30
|
+
Remaining limitations: the Antigravity CLI showed intermittent network resets to
|
|
31
|
+
Google's backend during testing, which surface as turn errors; Claude Code's model
|
|
32
|
+
aliases are a curated list because the CLI exposes no model listing; native usage
|
|
33
|
+
limit failover is unit-tested from error text, not observed live; the real Gemini
|
|
34
|
+
CLI ACP path is covered by mocked tests only because it is not installed here.
|
|
35
|
+
|
|
36
|
+
| Area | Evidence | Remaining limitation |
|
|
37
|
+
| --- | --- | --- |
|
|
38
|
+
| Commands and aliases | Offline command smoke tests, invalid input cases | Most smoke tests check exceptions, not all rendered content |
|
|
39
|
+
| Sessions | Save/load, separate fresh conversations, malformed metadata, permissions, project rebinding | Concurrent writes to the same explicitly named session are not coordinated |
|
|
40
|
+
| Provider switching and handoff | All seven provider catalogs, portable history, bounded failover | Live account quotas and model availability unverified |
|
|
41
|
+
| Tool execution | Real temporary-file operations, path containment, process timeout/cancellation, invalid arguments | Shell and code tools retain local user privileges; output capture memory is unbounded |
|
|
42
|
+
| Native Codex launcher | Save-before-launch, consent, flags, missing binary, terminal checks, failures | Mocked process replacement; no conversation transfer |
|
|
43
|
+
| Native CLI connections | Live REPL runs above; mocked protocol tests for approvals, failures, limits, prompt building, discovery parsing | Gemini ACP path mocked only; live limit failover not observed |
|
|
44
|
+
| Provider responses | Mocked native tool continuations; DeepSeek/Groq/OpenRouter streaming, usage-only chunks, 429/500 errors and serialized tool arguments | Other streaming and error branches still have coverage gaps |
|
|
45
|
+
| Display/config/auth | Configuration validation, display settings, credential boundaries, short-key masking | No full terminal/platform matrix |
|
|
46
|
+
| Usage | Atomic writes, malformed records, valid totals | Unknown catalog pricing may appear as zero estimated cost |
|
|
47
|
+
| Website | Production build and deterministic demo completion/replay test | Real browser, layout, clipboard, keyboard and accessibility checks remain incomplete |
|
|
48
|
+
| Dependencies | npm audit: zero advisories; Python installed-dependency audit: none found | Python audit skipped legacy local `thwip 1.0.0` metadata |
|
|
49
|
+
| Packaging | Wheel/source build and metadata checks | Publication is separately verified through the release workflow |
|
|
50
|
+
|
|
51
|
+
The Python suite measured 69% statement coverage with 160 passing tests before
|
|
52
|
+
the final credential-masking tests were added. Coverage is diagnostic evidence,
|
|
53
|
+
not proof that every feature works. The website test uses a minimal DOM stand-in
|
|
54
|
+
and controlled timers, not a browser.
|
|
55
|
+
|
|
56
|
+
This pass fixed default-session save collisions, malformed session/message
|
|
57
|
+
metadata acceptance, tool-argument display crashes, Rich markup interpretation
|
|
58
|
+
in action output, and short-key masking leakage.
|
|
@@ -20,6 +20,9 @@ def cli(tmp_path, monkeypatch):
|
|
|
20
20
|
cli = ThwipCLI.__new__(ThwipCLI)
|
|
21
21
|
cli.config = ThwipConfig(project=str(tmp_path), auto_save=False)
|
|
22
22
|
cli.registry = AgentRegistry(cli.config)
|
|
23
|
+
async def no_native_discovery(project):
|
|
24
|
+
return None
|
|
25
|
+
monkeypatch.setattr(cli.registry, 'connect_native_agents', no_native_discovery)
|
|
23
26
|
for a in cli.registry.list_agents():
|
|
24
27
|
monkeypatch.setattr(a, 'is_installed', lambda: True)
|
|
25
28
|
monkeypatch.setattr(a, 'is_configured', lambda: False)
|
|
@@ -162,3 +165,41 @@ async def test_failover_does_not_retry_already_failed_provider(cli, monkeypatch)
|
|
|
162
165
|
cli.config.limits.auto_switch = True
|
|
163
166
|
await cli.process_user_message('hello')
|
|
164
167
|
assert len(attempts) <= len(agents), f'Repeated failed providers: {attempts}'
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def test_main_handles_version_and_help_without_starting_the_repl(capsys):
|
|
171
|
+
from thwip import __version__
|
|
172
|
+
from thwip import cli as cli_module
|
|
173
|
+
|
|
174
|
+
with pytest.raises(SystemExit) as exit_info:
|
|
175
|
+
cli_module.main(['--version'])
|
|
176
|
+
assert exit_info.value.code == 0 and f'thwip {__version__}' in capsys.readouterr().out
|
|
177
|
+
with pytest.raises(SystemExit) as exit_info:
|
|
178
|
+
cli_module.main(['--help'])
|
|
179
|
+
assert exit_info.value.code == 0 and '--project' in capsys.readouterr().out
|
|
180
|
+
with pytest.raises(SystemExit) as exit_info:
|
|
181
|
+
cli_module.main(['--project', '/definitely/missing/dir'])
|
|
182
|
+
assert exit_info.value.code == 2
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
@pytest.mark.asyncio
|
|
186
|
+
async def test_failed_turn_removes_unanswered_user_message(cli, monkeypatch):
|
|
187
|
+
class Broken:
|
|
188
|
+
name = 'openai'
|
|
189
|
+
display_name = 'Broken'
|
|
190
|
+
company = 'OpenAI'
|
|
191
|
+
native_tools = True
|
|
192
|
+
project = '.'
|
|
193
|
+
def is_configured(self):
|
|
194
|
+
return True
|
|
195
|
+
def is_installed(self):
|
|
196
|
+
return True
|
|
197
|
+
def get_capabilities_for_model(self, model):
|
|
198
|
+
return set()
|
|
199
|
+
async def chat(self, **kwargs):
|
|
200
|
+
raise RuntimeError('Native CLI rejected thread/start')
|
|
201
|
+
yield
|
|
202
|
+
cli.current_agent = Broken()
|
|
203
|
+
cli.session.current_agent = 'openai'
|
|
204
|
+
await cli.process_user_message('hello')
|
|
205
|
+
assert cli.session.messages == []
|
|
@@ -115,11 +115,12 @@ async def test_unconfigured_agent_shows_setup_guidance(tmp_path):
|
|
|
115
115
|
assert len(cli.session.messages) == 0
|
|
116
116
|
|
|
117
117
|
|
|
118
|
-
|
|
118
|
+
@pytest.mark.asyncio
|
|
119
|
+
async def test_inline_api_key_is_rejected():
|
|
119
120
|
cli = ThwipCLI.__new__(ThwipCLI)
|
|
120
121
|
cli.config = SimpleNamespace(keys={}, key_sources={}, save=lambda: None)
|
|
121
122
|
|
|
122
|
-
cli.cmd_auth_config("openai", "secret-value")
|
|
123
|
+
await cli.cmd_auth_config("openai", "secret-value")
|
|
123
124
|
|
|
124
125
|
assert cli.config.keys == {}
|
|
125
126
|
assert cli.config.key_sources == {}
|
|
@@ -0,0 +1,198 @@
|
|
|
1
|
+
"""Native protocol behavior without credentials or model requests."""
|
|
2
|
+
|
|
3
|
+
import asyncio
|
|
4
|
+
|
|
5
|
+
import pytest
|
|
6
|
+
|
|
7
|
+
from thwip.agents.base import AgentDone, LimitHit, LimitStatus, ModelInfo, NativePermission, TextDelta
|
|
8
|
+
from thwip.agents.native_agent import NativeAgent
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class FakeRPC:
|
|
12
|
+
def __init__(self, provider, failure=False):
|
|
13
|
+
self.events = asyncio.Queue()
|
|
14
|
+
self.requests = []
|
|
15
|
+
self.sent = []
|
|
16
|
+
self.closed = False
|
|
17
|
+
self.provider = provider
|
|
18
|
+
self.failure = failure
|
|
19
|
+
|
|
20
|
+
async def request(self, method, params, **kwargs):
|
|
21
|
+
self.requests.append((method, params))
|
|
22
|
+
if method == 'account/read':
|
|
23
|
+
return {'account': {'type': 'chatgpt'}}
|
|
24
|
+
if method == 'model/list':
|
|
25
|
+
return {'data': [{'model': 'future-model', 'displayName': 'Future', 'isDefault': True}]}
|
|
26
|
+
if method == 'session/new':
|
|
27
|
+
return {'sessionId': 's', 'models': {'currentModelId': 'future-model',
|
|
28
|
+
'availableModels': [{'modelId': 'future-model', 'name': 'Future'}]}}
|
|
29
|
+
if method == 'thread/start':
|
|
30
|
+
return {'thread': {'id': 't'}}
|
|
31
|
+
if method == 'turn/start':
|
|
32
|
+
await self.events.put({'id': 99, 'method': 'item/commandExecution/requestApproval',
|
|
33
|
+
'params': {'command': 'touch example'}})
|
|
34
|
+
return {}
|
|
35
|
+
if method == 'session/prompt':
|
|
36
|
+
await self.events.put({'id': 99, 'method': 'session/request_permission', 'params': {
|
|
37
|
+
'toolCall': {'title': 'Edit'}, 'options': [
|
|
38
|
+
{'kind': 'allow_once', 'optionId': 'yes'}, {'kind': 'reject_once', 'optionId': 'no'}]}})
|
|
39
|
+
while not self.sent:
|
|
40
|
+
await asyncio.sleep(0)
|
|
41
|
+
await self.events.put({'method': 'session/update', 'params': {'update': {
|
|
42
|
+
'sessionUpdate': 'agent_message_chunk', 'content': {'type': 'text', 'text': 'done'}}}})
|
|
43
|
+
return {'stopReason': 'end_turn'}
|
|
44
|
+
return {}
|
|
45
|
+
|
|
46
|
+
async def send(self, message):
|
|
47
|
+
self.sent.append(message)
|
|
48
|
+
if self.provider == 'openai':
|
|
49
|
+
await self.events.put({'method': 'item/agentMessage/delta', 'params': {'itemId': 'a', 'delta': 'done'}})
|
|
50
|
+
await self.events.put({'method': 'turn/completed', 'params': {
|
|
51
|
+
'turn': {'status': 'failed' if self.failure else 'completed'}}})
|
|
52
|
+
|
|
53
|
+
async def close(self):
|
|
54
|
+
self.closed = True
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
@pytest.mark.parametrize('provider', ['openai', 'google'])
|
|
58
|
+
@pytest.mark.asyncio
|
|
59
|
+
async def test_discovered_models_replace_bundled_catalog(provider, monkeypatch):
|
|
60
|
+
agent = NativeAgent(provider, '.')
|
|
61
|
+
rpc = FakeRPC(provider)
|
|
62
|
+
async def connect():
|
|
63
|
+
return rpc
|
|
64
|
+
monkeypatch.setattr(agent, '_connect', connect)
|
|
65
|
+
await agent.refresh_models()
|
|
66
|
+
assert agent.ready and agent.get_default_model() == 'future-model'
|
|
67
|
+
assert agent.get_model_info('explicit-new-model').id == 'explicit-new-model'
|
|
68
|
+
assert rpc.closed
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
@pytest.mark.parametrize('provider', ['openai', 'google'])
|
|
72
|
+
@pytest.mark.parametrize('approve', [False, True])
|
|
73
|
+
@pytest.mark.asyncio
|
|
74
|
+
async def test_native_permission_response_and_completion(provider, approve, monkeypatch):
|
|
75
|
+
agent = NativeAgent(provider, '.')
|
|
76
|
+
rpc = FakeRPC(provider)
|
|
77
|
+
async def connect():
|
|
78
|
+
return rpc
|
|
79
|
+
monkeypatch.setattr(agent, '_connect', connect)
|
|
80
|
+
events = []
|
|
81
|
+
async for event in agent.chat([{'role': 'user', 'content': 'hello'}], model='future-model'):
|
|
82
|
+
events.append(event)
|
|
83
|
+
if isinstance(event, NativePermission):
|
|
84
|
+
event.approved = approve
|
|
85
|
+
assert any(isinstance(event, TextDelta) and event.content == 'done' for event in events)
|
|
86
|
+
assert isinstance(events[-1], AgentDone)
|
|
87
|
+
response = rpc.sent[0]['result']
|
|
88
|
+
if provider == 'openai':
|
|
89
|
+
assert response['decision'] == ('accept' if approve else 'decline')
|
|
90
|
+
else:
|
|
91
|
+
assert response['outcome']['optionId'] == ('yes' if approve else 'no')
|
|
92
|
+
assert rpc.closed
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
@pytest.mark.asyncio
|
|
96
|
+
async def test_failed_turn_never_emits_success(monkeypatch):
|
|
97
|
+
agent = NativeAgent('openai', '.')
|
|
98
|
+
rpc = FakeRPC('openai', failure=True)
|
|
99
|
+
async def connect():
|
|
100
|
+
return rpc
|
|
101
|
+
monkeypatch.setattr(agent, '_connect', connect)
|
|
102
|
+
events = []
|
|
103
|
+
with pytest.raises(RuntimeError, match='could not complete'):
|
|
104
|
+
async for event in agent.chat([], model='future-model'):
|
|
105
|
+
events.append(event)
|
|
106
|
+
assert not any(isinstance(event, AgentDone) for event in events)
|
|
107
|
+
assert rpc.closed
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
@pytest.mark.asyncio
|
|
111
|
+
async def test_discovery_failure_marks_connection_unready(monkeypatch):
|
|
112
|
+
agent = NativeAgent('google', '.')
|
|
113
|
+
async def connect():
|
|
114
|
+
raise TimeoutError()
|
|
115
|
+
monkeypatch.setattr(agent, '_connect', connect)
|
|
116
|
+
await agent.refresh_models()
|
|
117
|
+
assert not agent.ready and 'timed out' in agent.discovery_error
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
@pytest.mark.asyncio
|
|
121
|
+
async def test_codex_thread_uses_protocol_sandbox_spelling(monkeypatch):
|
|
122
|
+
"""Codex App Server rejects camelCase sandbox modes with an invalid-request error."""
|
|
123
|
+
agent = NativeAgent('openai', '.')
|
|
124
|
+
rpc = FakeRPC('openai')
|
|
125
|
+
async def connect():
|
|
126
|
+
return rpc
|
|
127
|
+
monkeypatch.setattr(agent, '_connect', connect)
|
|
128
|
+
async for _ in agent.chat([{'role': 'user', 'content': 'hello'}], model='future-model', system_prompt='Be terse.'):
|
|
129
|
+
pass
|
|
130
|
+
_method, params = next(request for request in rpc.requests if request[0] == 'thread/start')
|
|
131
|
+
assert params['sandbox'] == 'read-only' and params['approvalPolicy'] == 'on-request'
|
|
132
|
+
assert params['developerInstructions'] == 'Be terse.'
|
|
133
|
+
turn = next(params for method, params in rpc.requests if method == 'turn/start')
|
|
134
|
+
assert turn['input'][0]['text'] == 'hello'
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
@pytest.mark.asyncio
|
|
138
|
+
async def test_codex_usage_limit_failure_becomes_limit_hit(monkeypatch):
|
|
139
|
+
agent = NativeAgent('openai', '.')
|
|
140
|
+
rpc = FakeRPC('openai')
|
|
141
|
+
async def send(message):
|
|
142
|
+
rpc.sent.append(message)
|
|
143
|
+
await rpc.events.put({'method': 'turn/completed', 'params': {'turn': {
|
|
144
|
+
'status': 'failed', 'error': {'message': 'You have hit your usage limit.'}}}})
|
|
145
|
+
rpc.send = send
|
|
146
|
+
async def connect():
|
|
147
|
+
return rpc
|
|
148
|
+
monkeypatch.setattr(agent, '_connect', connect)
|
|
149
|
+
events = [event async for event in agent.chat([{'role': 'user', 'content': 'hi'}], model='future-model')]
|
|
150
|
+
assert isinstance(events[-1], LimitHit) and events[-1].error_type == LimitStatus.QUOTA_EXHAUSTED
|
|
151
|
+
assert rpc.closed
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def test_permission_descriptions_are_readable_and_scrubbed():
|
|
155
|
+
from thwip.agents.native_agent import describe_permission
|
|
156
|
+
|
|
157
|
+
text = describe_permission({"type": "commandExecution", "command": "curl -H 'Authorization: Bearer " + "k" * 40 + "'",
|
|
158
|
+
"cwd": "/repo", "reason": "Fetch data"}, "Codex")
|
|
159
|
+
assert text.startswith("Codex wants to run a command.") and "Directory: /repo" in text and "Reason: Fetch data" in text
|
|
160
|
+
assert "kkkk" not in text and "[redacted]" in text
|
|
161
|
+
files = describe_permission({"type": "fileChange", "changes": [{"path": "a.py", "kind": "update"}]}, "Codex")
|
|
162
|
+
assert "wants to change files" in files and "a.py (update)" in files
|
|
163
|
+
assert "Gemini requests permission for Edit" in describe_permission({"title": "Edit"}, "Gemini")
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def test_handoff_accepts_explicit_native_model_ids():
|
|
167
|
+
from thwip.handoff import build_handoff_report, local_model
|
|
168
|
+
from thwip.session import Session
|
|
169
|
+
|
|
170
|
+
agent = NativeAgent('openai', '.')
|
|
171
|
+
agent.available_models = [ModelInfo(id='listed', name='Listed', is_default=True)]
|
|
172
|
+
assert local_model(agent, 'listed').name == 'Listed'
|
|
173
|
+
assert local_model(agent, 'brand-new').id == 'brand-new'
|
|
174
|
+
assert local_model(agent, 'has space') is None
|
|
175
|
+
session = Session(current_agent='openai', current_model='listed')
|
|
176
|
+
report = build_handoff_report(session, agent, agent, 'brand-new')
|
|
177
|
+
assert report.target == 'openai/brand-new' and report.context_pressure == 'unknown'
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
@pytest.mark.asyncio
|
|
181
|
+
async def test_codex_rate_limit_notification_is_recorded(monkeypatch):
|
|
182
|
+
agent = NativeAgent('openai', '.')
|
|
183
|
+
rpc = FakeRPC('openai')
|
|
184
|
+
async def send(message):
|
|
185
|
+
rpc.sent.append(message)
|
|
186
|
+
await rpc.events.put({'method': 'account/rateLimits/updated', 'params': {'rateLimits': {
|
|
187
|
+
'primary': {'usedPercent': 2, 'windowDurationMins': 300, 'resetsAt': 1790212672},
|
|
188
|
+
'secondary': {'usedPercent': 23, 'windowDurationMins': 10080}}}})
|
|
189
|
+
await rpc.events.put({'method': 'turn/completed', 'params': {'turn': {'status': 'completed'}}})
|
|
190
|
+
rpc.send = send
|
|
191
|
+
async def connect():
|
|
192
|
+
return rpc
|
|
193
|
+
monkeypatch.setattr(agent, '_connect', connect)
|
|
194
|
+
events = [event async for event in agent.chat([{'role': 'user', 'content': 'hi'}], model='future-model')]
|
|
195
|
+
assert isinstance(events[-1], AgentDone)
|
|
196
|
+
assert agent.limit_windows == [
|
|
197
|
+
{'label': '5h', 'used_percent': 2, 'resets_at': 1790212672},
|
|
198
|
+
{'label': '7d', 'used_percent': 23, 'resets_at': None}]
|