thwip-cli 1.3.0__tar.gz → 1.5.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {thwip_cli-1.3.0 → thwip_cli-1.5.0}/PKG-INFO +84 -26
- {thwip_cli-1.3.0 → thwip_cli-1.5.0}/README.md +83 -25
- thwip_cli-1.5.0/docs/verification.md +62 -0
- {thwip_cli-1.3.0 → thwip_cli-1.5.0}/pyproject.toml +1 -1
- {thwip_cli-1.3.0 → thwip_cli-1.5.0}/tests/test_audit_regressions.py +101 -1
- thwip_cli-1.5.0/tests/test_catalog.py +131 -0
- {thwip_cli-1.3.0 → thwip_cli-1.5.0}/tests/test_cli.py +3 -2
- thwip_cli-1.5.0/tests/test_native_agents.py +198 -0
- thwip_cli-1.5.0/tests/test_native_print.py +234 -0
- thwip_cli-1.5.0/tests/test_native_rpc.py +51 -0
- thwip_cli-1.5.0/tests/test_parity_commands.py +173 -0
- {thwip_cli-1.3.0 → thwip_cli-1.5.0}/thwip/__init__.py +1 -1
- {thwip_cli-1.3.0 → thwip_cli-1.5.0}/thwip/agents/__init__.py +47 -0
- {thwip_cli-1.3.0 → thwip_cli-1.5.0}/thwip/agents/base.py +42 -1
- thwip_cli-1.5.0/thwip/agents/catalog.py +141 -0
- thwip_cli-1.5.0/thwip/agents/native_agent.py +281 -0
- thwip_cli-1.5.0/thwip/agents/native_common.py +104 -0
- thwip_cli-1.5.0/thwip/agents/native_print.py +327 -0
- thwip_cli-1.5.0/thwip/agents/native_rpc.py +86 -0
- {thwip_cli-1.3.0 → thwip_cli-1.5.0}/thwip/cli.py +494 -76
- {thwip_cli-1.3.0 → thwip_cli-1.5.0}/thwip/handoff.py +3 -1
- {thwip_cli-1.3.0 → thwip_cli-1.5.0}/thwip/shortcuts.py +7 -0
- {thwip_cli-1.3.0 → thwip_cli-1.5.0}/thwip/theme.py +5 -0
- {thwip_cli-1.3.0 → thwip_cli-1.5.0}/uv.lock +1 -1
- thwip_cli-1.3.0/docs/verification.md +0 -28
- {thwip_cli-1.3.0 → thwip_cli-1.5.0}/.github/workflows/publish.yml +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.5.0}/.gitignore +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.5.0}/LICENSE +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.5.0}/docs/handoff-research.md +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.5.0}/install.sh +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.5.0}/tests/test_agents.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.5.0}/tests/test_compatible_streaming.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.5.0}/tests/test_config.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.5.0}/tests/test_detector.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.5.0}/tests/test_handoff.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.5.0}/tests/test_native_launcher.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.5.0}/tests/test_repair_verification.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.5.0}/tests/test_session.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.5.0}/tests/test_tools.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.5.0}/tests/test_utils.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.5.0}/thwip/__main__.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.5.0}/thwip/agents/chat_messages.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.5.0}/thwip/agents/claude_agent.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.5.0}/thwip/agents/deepseek_agent.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.5.0}/thwip/agents/google_agent.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.5.0}/thwip/agents/groq_agent.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.5.0}/thwip/agents/ollama_agent.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.5.0}/thwip/agents/openai_agent.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.5.0}/thwip/agents/openrouter_agent.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.5.0}/thwip/config.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.5.0}/thwip/detector.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.5.0}/thwip/limits.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.5.0}/thwip/session.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.5.0}/thwip/tools/__init__.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.5.0}/thwip/tools/code_runner.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.5.0}/thwip/tools/file_editor.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.5.0}/thwip/tools/git_ops.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.5.0}/thwip/tools/terminal.py +0 -0
- {thwip_cli-1.3.0 → thwip_cli-1.5.0}/thwip/utils.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: thwip-cli
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.5.0
|
|
4
4
|
Summary: Universal coding agent multiplexer: detect, switch, and route between AI coding agents seamlessly
|
|
5
5
|
Project-URL: Homepage, https://github.com/tanmayhutt/thwip-cli
|
|
6
6
|
Project-URL: Repository, https://github.com/tanmayhutt/thwip-cli
|
|
@@ -46,6 +46,7 @@ Description-Content-Type: text/markdown
|
|
|
46
46
|
## Features
|
|
47
47
|
|
|
48
48
|
- **Auto-Detection**: Discovers installed AI coding agents (Claude Code, Antigravity, Gemini CLI, OpenAI/Codex, Aider, Copilot, Cursor, Windsurf, Cline, Ollama) and configured credentials.
|
|
49
|
+
- **Existing Sign-ins**: Chats through installed Codex, Claude Code, and Antigravity CLIs using their own logins and live model lists, with no API key required.
|
|
49
50
|
- **Context Portability**: Switch providers with stored conversational text. Working files remain in the selected local project; they are not automatically uploaded.
|
|
50
51
|
- **Handoff Preview**: Inspect text continuity, omitted state, capability changes, approximate context pressure, and a text fingerprint before switching. Runs locally without model calls.
|
|
51
52
|
- **Dynamic UI**: Terminal interface adapts its status bar, capabilities, and theme based on the active provider.
|
|
@@ -53,7 +54,8 @@ Description-Content-Type: text/markdown
|
|
|
53
54
|
- **Rate Limit Failover**: Detects HTTP 429 errors or quota exhaustion and prompts instant switching to ready fallback models.
|
|
54
55
|
- **Tool Layer**: Shared file editing, code execution (Python, Node, Shell), and Git integration across connected models.
|
|
55
56
|
- **Session Persistence**: Save and resume sessions across projects with `/session save` and `/session load`.
|
|
56
|
-
- **
|
|
57
|
+
- **Familiar Flow**: The same everyday commands as Codex, Claude Code, and Antigravity (`/model`, `/new`, `/resume`, `/compact`, `/diff`, `/copy`, `!cmd`, `@file`), plus autocompletion, Ctrl+S / Ctrl+T shortcuts, and streaming Markdown.
|
|
58
|
+
- **Live Model Lists**: Model catalogs come from the connected CLI or the provider's list-models endpoint, never from a hardcoded list.
|
|
57
59
|
|
|
58
60
|
---
|
|
59
61
|
|
|
@@ -81,6 +83,8 @@ Start the interactive terminal in your current project directory:
|
|
|
81
83
|
|
|
82
84
|
```bash
|
|
83
85
|
thwip
|
|
86
|
+
thwip --project ~/code/my-app # open a specific project
|
|
87
|
+
thwip --version
|
|
84
88
|
```
|
|
85
89
|
|
|
86
90
|
---
|
|
@@ -91,13 +95,22 @@ thwip
|
|
|
91
95
|
|:---|:---|
|
|
92
96
|
| `/switch [agent] [model]` | Switch active agent or model mid-conversation |
|
|
93
97
|
| `/handoff [agent] [model]` | Preview a target locally without switching or sending data |
|
|
94
|
-
| `/native codex` | Save and leave Thwip for the installed Codex CLI
|
|
98
|
+
| `/native codex` | Save and leave Thwip for the installed Codex CLI itself (launcher, no context transfer) |
|
|
95
99
|
| `/agents` | Show all detected coding agents, company status, and capabilities |
|
|
96
100
|
| `/models [agent]` | List available models for current or target agent |
|
|
97
101
|
| `/key [provider]` | Enter an API key securely without placing it in prompt history |
|
|
98
102
|
| `/status` | Display current session, project, and token stats |
|
|
99
|
-
| `/limits` | View token usage,
|
|
100
|
-
| `/detect` | Re-scan
|
|
103
|
+
| `/limits`, `/usage` | View token usage, CLI account usage windows, and spend metrics |
|
|
104
|
+
| `/detect` | Re-scan installed agents and reconnect CLI sign-ins |
|
|
105
|
+
| `/model [id]` | Pick a model for the current agent from a numbered list or by ID |
|
|
106
|
+
| `/new` | Start a fresh conversation; the current one is saved first |
|
|
107
|
+
| `/resume [name]` | Resume a saved session from a numbered list |
|
|
108
|
+
| `/compact` | Summarize the conversation with the current model to free context; the summary is plain text and travels across providers |
|
|
109
|
+
| `/diff [staged]` | Show the project's git diff |
|
|
110
|
+
| `/copy` | Copy the last response to the clipboard |
|
|
111
|
+
| `/export [path]` | Write the conversation to Markdown with per-message model attribution |
|
|
112
|
+
| `!<command>` | Run a shell command in the project without involving a model |
|
|
113
|
+
| `@path` in a message | Attach a project file's content to the message (files outside the project are ignored) |
|
|
101
114
|
| `/session save [name]` | Save current chat session |
|
|
102
115
|
| `/session load <name>` | Load a previously saved session |
|
|
103
116
|
| `/session list` | List all saved sessions |
|
|
@@ -110,23 +123,68 @@ thwip
|
|
|
110
123
|
| `Ctrl + H` | View history |
|
|
111
124
|
| `/quit` | Exit thwip |
|
|
112
125
|
|
|
113
|
-
Short aliases are available for frequent commands: `/a`, `/m`, `/s`, `/k`, `/g`, and `/t`.
|
|
114
|
-
|
|
115
|
-
### Existing
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
126
|
+
Short aliases are available for frequent commands: `/a`, `/m`, `/s`, `/sw`, `/k`, `/g`, and `/t`.
|
|
127
|
+
|
|
128
|
+
### Existing CLI sign-ins (no API key needed)
|
|
129
|
+
|
|
130
|
+
If Codex, Claude Code, or the Antigravity CLI is installed and signed in, Thwip
|
|
131
|
+
connects to it at startup and uses that sign-in for chat. No API key is copied or
|
|
132
|
+
required, and Thwip never reads the CLI's stored credentials. A configured direct
|
|
133
|
+
API key for the same provider always takes precedence over the native connection.
|
|
134
|
+
|
|
135
|
+
| Provider | Installed CLI | Transport | Model list |
|
|
136
|
+
|:---|:---|:---|:---|
|
|
137
|
+
| OpenAI | `codex` | Codex App Server (JSON-RPC over stdio) | Live from `model/list` |
|
|
138
|
+
| Anthropic | `claude` | Claude Code print mode (`stream-json`) | Aliases `fable`, `opus`, `sonnet`, `haiku`; explicit IDs pass through |
|
|
139
|
+
| Google | `agy` (Antigravity CLI) or `gemini` | Antigravity print mode (`stream-json`) or Gemini ACP | Live from `agy models` or the ACP session |
|
|
140
|
+
|
|
141
|
+
`/models` refreshes the catalog supplied by each CLI. An explicit model ID that is
|
|
142
|
+
not in the catalog is passed to the CLI for validation instead of being rejected by
|
|
143
|
+
a bundled list. A model available in a desktop app may still be absent from the
|
|
144
|
+
installed CLI's catalog or account access.
|
|
145
|
+
|
|
146
|
+
Native connections are read-only by default. Codex starts with a read-only sandbox
|
|
147
|
+
and asks before operations outside it; approval requests appear in Thwip with the
|
|
148
|
+
command or file list and default to denial. Claude Code and the Antigravity CLI run
|
|
149
|
+
in their non-interactive print modes, where tools that would need an approval are
|
|
150
|
+
declined by the CLI itself. Each turn sends the portable text conversation to a
|
|
151
|
+
fresh native session, so native reasoning and tool state do not carry between turns.
|
|
152
|
+
Ctrl+C interrupts the current turn and stops the child process; the unanswered
|
|
153
|
+
message is removed so it can be re-sent or handed to another provider.
|
|
154
|
+
|
|
155
|
+
Native billing and usage limits are managed by each CLI account. Thwip records the
|
|
156
|
+
token counts the CLIs report but does not estimate cost for them. Codex and Claude
|
|
157
|
+
Code also report their account usage windows (5 hour and 7 day) after each response;
|
|
158
|
+
`/limits` and `/status` show the used percentage and reset time. When a CLI reports
|
|
159
|
+
an exhausted usage limit, the standard failover prompt offers the other connected
|
|
160
|
+
providers.
|
|
161
|
+
|
|
162
|
+
`/native codex` remains available as a launcher: it saves the Thwip session and
|
|
163
|
+
replaces Thwip with the Codex CLI itself in the selected project. Conversation
|
|
164
|
+
history is not transferred by the launcher; use `/session load` after restarting
|
|
165
|
+
Thwip to resume.
|
|
166
|
+
|
|
167
|
+
## Live model lists
|
|
168
|
+
|
|
169
|
+
Model lists are not hardcoded. Each connected source supplies its own list:
|
|
170
|
+
|
|
171
|
+
- Installed CLIs report their models (Codex `model/list`, `agy models`, Claude Code aliases).
|
|
172
|
+
- Direct providers with a key are queried through their list-models endpoint (OpenAI,
|
|
173
|
+
Anthropic, Google, DeepSeek, Groq, OpenRouter) at startup, on `/models`, and right
|
|
174
|
+
after `/key`. Non-chat models (embeddings, speech, image, moderation) are filtered
|
|
175
|
+
out. Context sizes and prices are taken from the provider when it publishes them.
|
|
176
|
+
- A small bundled list per provider remains only as an offline fallback and is labelled
|
|
177
|
+
"bundled fallback" in `/models` until a key or connection is available.
|
|
178
|
+
|
|
179
|
+
## Usage-limit failover
|
|
180
|
+
|
|
181
|
+
When the active provider reports an exhausted limit (Codex "You've hit your usage
|
|
182
|
+
limit", Claude Code "usage limit reached" or a rejected rate-limit event, Google
|
|
183
|
+
`RESOURCE_EXHAUSTED`, or HTTP 429 from a direct API), Thwip stops, shows the ready
|
|
184
|
+
alternatives with their default models, and offers to switch. Accepting switches the
|
|
185
|
+
provider, keeps the text conversation, and re-sends the unanswered message. Set
|
|
186
|
+
`[limits] auto_switch = true` to skip the question. The `[fallback] chain` order is
|
|
187
|
+
honored; a chain model the provider does not list falls back to that provider's default.
|
|
130
188
|
|
|
131
189
|
## Auditable handoffs
|
|
132
190
|
|
|
@@ -165,9 +223,9 @@ See [research and prior art](docs/handoff-research.md) for the differentiation r
|
|
|
165
223
|
|
|
166
224
|
| Company | Agent | Capabilities |
|
|
167
225
|
|:---|:---|:---|
|
|
168
|
-
| Anthropic | Claude API (Fable 5, Opus 5, Sonnet 5, Haiku 4.5) | Chat, File Edit, Code Run, Terminal, Git |
|
|
169
|
-
| Google | Gemini API (3.1 Pro Preview, 3.7 Flash, 3.5 Flash-Lite) | Chat, File Edit, Code Run, Terminal, Git |
|
|
170
|
-
| OpenAI | OpenAI API (GPT-5.6 Sol, Terra, Luna) | Chat, File Edit, Code Run, Terminal, Git |
|
|
226
|
+
| Anthropic | Claude Code sign-in, or Claude API (Fable 5, Opus 5, Sonnet 5, Haiku 4.5) | Chat, File Edit, Code Run, Terminal, Git |
|
|
227
|
+
| Google | Antigravity or Gemini CLI sign-in, or Gemini API (3.1 Pro Preview, 3.7 Flash, 3.5 Flash-Lite) | Chat, File Edit, Code Run, Terminal, Git |
|
|
228
|
+
| OpenAI | Codex CLI sign-in, or OpenAI API (GPT-5.6 Sol, Terra, Luna) | Chat, File Edit, Code Run, Terminal, Git |
|
|
171
229
|
| DeepSeek | DeepSeek V3 / R1 Reasoner | Chat, File Edit, Code Run, Reasoning |
|
|
172
230
|
| Groq | GPT-OSS 120B (default); Llama 3.3 for eligible enterprise accounts only | Chat, File Edit, Code Run |
|
|
173
231
|
| Ollama | Local Models (Llama 3.3, Qwen Coder, DeepSeek R1) | Chat, File Edit, Code Run (Local, Offline) |
|
|
@@ -179,7 +237,7 @@ See [research and prior art](docs/handoff-research.md) for the differentiation r
|
|
|
179
237
|
|
|
180
238
|
thwip auto-detects existing API keys from environment variables and existing agent configs (`~/.claude.json`, `~/.gemini/config.json`). Configuration can also be set manually:
|
|
181
239
|
|
|
182
|
-
Installed
|
|
240
|
+
Installed CLI sign-ins and API access are separate paths. A signed-in Codex, Claude Code, or Antigravity CLI is used directly through its own protocol (see above). The direct SDK adapters for Anthropic, Google, OpenAI, DeepSeek, Groq, and OpenRouter require a provider API key. Ollama needs no key when its local server is running.
|
|
183
241
|
|
|
184
242
|
```toml
|
|
185
243
|
[defaults]
|
|
@@ -9,6 +9,7 @@
|
|
|
9
9
|
## Features
|
|
10
10
|
|
|
11
11
|
- **Auto-Detection**: Discovers installed AI coding agents (Claude Code, Antigravity, Gemini CLI, OpenAI/Codex, Aider, Copilot, Cursor, Windsurf, Cline, Ollama) and configured credentials.
|
|
12
|
+
- **Existing Sign-ins**: Chats through installed Codex, Claude Code, and Antigravity CLIs using their own logins and live model lists, with no API key required.
|
|
12
13
|
- **Context Portability**: Switch providers with stored conversational text. Working files remain in the selected local project; they are not automatically uploaded.
|
|
13
14
|
- **Handoff Preview**: Inspect text continuity, omitted state, capability changes, approximate context pressure, and a text fingerprint before switching. Runs locally without model calls.
|
|
14
15
|
- **Dynamic UI**: Terminal interface adapts its status bar, capabilities, and theme based on the active provider.
|
|
@@ -16,7 +17,8 @@
|
|
|
16
17
|
- **Rate Limit Failover**: Detects HTTP 429 errors or quota exhaustion and prompts instant switching to ready fallback models.
|
|
17
18
|
- **Tool Layer**: Shared file editing, code execution (Python, Node, Shell), and Git integration across connected models.
|
|
18
19
|
- **Session Persistence**: Save and resume sessions across projects with `/session save` and `/session load`.
|
|
19
|
-
- **
|
|
20
|
+
- **Familiar Flow**: The same everyday commands as Codex, Claude Code, and Antigravity (`/model`, `/new`, `/resume`, `/compact`, `/diff`, `/copy`, `!cmd`, `@file`), plus autocompletion, Ctrl+S / Ctrl+T shortcuts, and streaming Markdown.
|
|
21
|
+
- **Live Model Lists**: Model catalogs come from the connected CLI or the provider's list-models endpoint, never from a hardcoded list.
|
|
20
22
|
|
|
21
23
|
---
|
|
22
24
|
|
|
@@ -44,6 +46,8 @@ Start the interactive terminal in your current project directory:
|
|
|
44
46
|
|
|
45
47
|
```bash
|
|
46
48
|
thwip
|
|
49
|
+
thwip --project ~/code/my-app # open a specific project
|
|
50
|
+
thwip --version
|
|
47
51
|
```
|
|
48
52
|
|
|
49
53
|
---
|
|
@@ -54,13 +58,22 @@ thwip
|
|
|
54
58
|
|:---|:---|
|
|
55
59
|
| `/switch [agent] [model]` | Switch active agent or model mid-conversation |
|
|
56
60
|
| `/handoff [agent] [model]` | Preview a target locally without switching or sending data |
|
|
57
|
-
| `/native codex` | Save and leave Thwip for the installed Codex CLI
|
|
61
|
+
| `/native codex` | Save and leave Thwip for the installed Codex CLI itself (launcher, no context transfer) |
|
|
58
62
|
| `/agents` | Show all detected coding agents, company status, and capabilities |
|
|
59
63
|
| `/models [agent]` | List available models for current or target agent |
|
|
60
64
|
| `/key [provider]` | Enter an API key securely without placing it in prompt history |
|
|
61
65
|
| `/status` | Display current session, project, and token stats |
|
|
62
|
-
| `/limits` | View token usage,
|
|
63
|
-
| `/detect` | Re-scan
|
|
66
|
+
| `/limits`, `/usage` | View token usage, CLI account usage windows, and spend metrics |
|
|
67
|
+
| `/detect` | Re-scan installed agents and reconnect CLI sign-ins |
|
|
68
|
+
| `/model [id]` | Pick a model for the current agent from a numbered list or by ID |
|
|
69
|
+
| `/new` | Start a fresh conversation; the current one is saved first |
|
|
70
|
+
| `/resume [name]` | Resume a saved session from a numbered list |
|
|
71
|
+
| `/compact` | Summarize the conversation with the current model to free context; the summary is plain text and travels across providers |
|
|
72
|
+
| `/diff [staged]` | Show the project's git diff |
|
|
73
|
+
| `/copy` | Copy the last response to the clipboard |
|
|
74
|
+
| `/export [path]` | Write the conversation to Markdown with per-message model attribution |
|
|
75
|
+
| `!<command>` | Run a shell command in the project without involving a model |
|
|
76
|
+
| `@path` in a message | Attach a project file's content to the message (files outside the project are ignored) |
|
|
64
77
|
| `/session save [name]` | Save current chat session |
|
|
65
78
|
| `/session load <name>` | Load a previously saved session |
|
|
66
79
|
| `/session list` | List all saved sessions |
|
|
@@ -73,23 +86,68 @@ thwip
|
|
|
73
86
|
| `Ctrl + H` | View history |
|
|
74
87
|
| `/quit` | Exit thwip |
|
|
75
88
|
|
|
76
|
-
Short aliases are available for frequent commands: `/a`, `/m`, `/s`, `/k`, `/g`, and `/t`.
|
|
77
|
-
|
|
78
|
-
### Existing
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
89
|
+
Short aliases are available for frequent commands: `/a`, `/m`, `/s`, `/sw`, `/k`, `/g`, and `/t`.
|
|
90
|
+
|
|
91
|
+
### Existing CLI sign-ins (no API key needed)
|
|
92
|
+
|
|
93
|
+
If Codex, Claude Code, or the Antigravity CLI is installed and signed in, Thwip
|
|
94
|
+
connects to it at startup and uses that sign-in for chat. No API key is copied or
|
|
95
|
+
required, and Thwip never reads the CLI's stored credentials. A configured direct
|
|
96
|
+
API key for the same provider always takes precedence over the native connection.
|
|
97
|
+
|
|
98
|
+
| Provider | Installed CLI | Transport | Model list |
|
|
99
|
+
|:---|:---|:---|:---|
|
|
100
|
+
| OpenAI | `codex` | Codex App Server (JSON-RPC over stdio) | Live from `model/list` |
|
|
101
|
+
| Anthropic | `claude` | Claude Code print mode (`stream-json`) | Aliases `fable`, `opus`, `sonnet`, `haiku`; explicit IDs pass through |
|
|
102
|
+
| Google | `agy` (Antigravity CLI) or `gemini` | Antigravity print mode (`stream-json`) or Gemini ACP | Live from `agy models` or the ACP session |
|
|
103
|
+
|
|
104
|
+
`/models` refreshes the catalog supplied by each CLI. An explicit model ID that is
|
|
105
|
+
not in the catalog is passed to the CLI for validation instead of being rejected by
|
|
106
|
+
a bundled list. A model available in a desktop app may still be absent from the
|
|
107
|
+
installed CLI's catalog or account access.
|
|
108
|
+
|
|
109
|
+
Native connections are read-only by default. Codex starts with a read-only sandbox
|
|
110
|
+
and asks before operations outside it; approval requests appear in Thwip with the
|
|
111
|
+
command or file list and default to denial. Claude Code and the Antigravity CLI run
|
|
112
|
+
in their non-interactive print modes, where tools that would need an approval are
|
|
113
|
+
declined by the CLI itself. Each turn sends the portable text conversation to a
|
|
114
|
+
fresh native session, so native reasoning and tool state do not carry between turns.
|
|
115
|
+
Ctrl+C interrupts the current turn and stops the child process; the unanswered
|
|
116
|
+
message is removed so it can be re-sent or handed to another provider.
|
|
117
|
+
|
|
118
|
+
Native billing and usage limits are managed by each CLI account. Thwip records the
|
|
119
|
+
token counts the CLIs report but does not estimate cost for them. Codex and Claude
|
|
120
|
+
Code also report their account usage windows (5 hour and 7 day) after each response;
|
|
121
|
+
`/limits` and `/status` show the used percentage and reset time. When a CLI reports
|
|
122
|
+
an exhausted usage limit, the standard failover prompt offers the other connected
|
|
123
|
+
providers.
|
|
124
|
+
|
|
125
|
+
`/native codex` remains available as a launcher: it saves the Thwip session and
|
|
126
|
+
replaces Thwip with the Codex CLI itself in the selected project. Conversation
|
|
127
|
+
history is not transferred by the launcher; use `/session load` after restarting
|
|
128
|
+
Thwip to resume.
|
|
129
|
+
|
|
130
|
+
## Live model lists
|
|
131
|
+
|
|
132
|
+
Model lists are not hardcoded. Each connected source supplies its own list:
|
|
133
|
+
|
|
134
|
+
- Installed CLIs report their models (Codex `model/list`, `agy models`, Claude Code aliases).
|
|
135
|
+
- Direct providers with a key are queried through their list-models endpoint (OpenAI,
|
|
136
|
+
Anthropic, Google, DeepSeek, Groq, OpenRouter) at startup, on `/models`, and right
|
|
137
|
+
after `/key`. Non-chat models (embeddings, speech, image, moderation) are filtered
|
|
138
|
+
out. Context sizes and prices are taken from the provider when it publishes them.
|
|
139
|
+
- A small bundled list per provider remains only as an offline fallback and is labelled
|
|
140
|
+
"bundled fallback" in `/models` until a key or connection is available.
|
|
141
|
+
|
|
142
|
+
## Usage-limit failover
|
|
143
|
+
|
|
144
|
+
When the active provider reports an exhausted limit (Codex "You've hit your usage
|
|
145
|
+
limit", Claude Code "usage limit reached" or a rejected rate-limit event, Google
|
|
146
|
+
`RESOURCE_EXHAUSTED`, or HTTP 429 from a direct API), Thwip stops, shows the ready
|
|
147
|
+
alternatives with their default models, and offers to switch. Accepting switches the
|
|
148
|
+
provider, keeps the text conversation, and re-sends the unanswered message. Set
|
|
149
|
+
`[limits] auto_switch = true` to skip the question. The `[fallback] chain` order is
|
|
150
|
+
honored; a chain model the provider does not list falls back to that provider's default.
|
|
93
151
|
|
|
94
152
|
## Auditable handoffs
|
|
95
153
|
|
|
@@ -128,9 +186,9 @@ See [research and prior art](docs/handoff-research.md) for the differentiation r
|
|
|
128
186
|
|
|
129
187
|
| Company | Agent | Capabilities |
|
|
130
188
|
|:---|:---|:---|
|
|
131
|
-
| Anthropic | Claude API (Fable 5, Opus 5, Sonnet 5, Haiku 4.5) | Chat, File Edit, Code Run, Terminal, Git |
|
|
132
|
-
| Google | Gemini API (3.1 Pro Preview, 3.7 Flash, 3.5 Flash-Lite) | Chat, File Edit, Code Run, Terminal, Git |
|
|
133
|
-
| OpenAI | OpenAI API (GPT-5.6 Sol, Terra, Luna) | Chat, File Edit, Code Run, Terminal, Git |
|
|
189
|
+
| Anthropic | Claude Code sign-in, or Claude API (Fable 5, Opus 5, Sonnet 5, Haiku 4.5) | Chat, File Edit, Code Run, Terminal, Git |
|
|
190
|
+
| Google | Antigravity or Gemini CLI sign-in, or Gemini API (3.1 Pro Preview, 3.7 Flash, 3.5 Flash-Lite) | Chat, File Edit, Code Run, Terminal, Git |
|
|
191
|
+
| OpenAI | Codex CLI sign-in, or OpenAI API (GPT-5.6 Sol, Terra, Luna) | Chat, File Edit, Code Run, Terminal, Git |
|
|
134
192
|
| DeepSeek | DeepSeek V3 / R1 Reasoner | Chat, File Edit, Code Run, Reasoning |
|
|
135
193
|
| Groq | GPT-OSS 120B (default); Llama 3.3 for eligible enterprise accounts only | Chat, File Edit, Code Run |
|
|
136
194
|
| Ollama | Local Models (Llama 3.3, Qwen Coder, DeepSeek R1) | Chat, File Edit, Code Run (Local, Offline) |
|
|
@@ -142,7 +200,7 @@ See [research and prior art](docs/handoff-research.md) for the differentiation r
|
|
|
142
200
|
|
|
143
201
|
thwip auto-detects existing API keys from environment variables and existing agent configs (`~/.claude.json`, `~/.gemini/config.json`). Configuration can also be set manually:
|
|
144
202
|
|
|
145
|
-
Installed
|
|
203
|
+
Installed CLI sign-ins and API access are separate paths. A signed-in Codex, Claude Code, or Antigravity CLI is used directly through its own protocol (see above). The direct SDK adapters for Anthropic, Google, OpenAI, DeepSeek, Groq, and OpenRouter require a provider API key. Ollama needs no key when its local server is running.
|
|
146
204
|
|
|
147
205
|
```toml
|
|
148
206
|
[defaults]
|
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
# Verification status
|
|
2
|
+
|
|
3
|
+
Verified locally on 2026-09-24 for v1.5.0. The Python suite
|
|
4
|
+
passes 240 offline tests. Exhaustive behavior across every provider and
|
|
5
|
+
configuration has not been established.
|
|
6
|
+
|
|
7
|
+
## Native CLI connections (2026-09-24)
|
|
8
|
+
|
|
9
|
+
Live checks were run on macOS through a pseudo-terminal driving the real REPL with
|
|
10
|
+
the installed Codex CLI 0.152.1, Claude Code 2.1.281, and Antigravity CLI 1.2.8,
|
|
11
|
+
each using its existing sign-in. No API keys were configured.
|
|
12
|
+
|
|
13
|
+
| Flow | Result |
|
|
14
|
+
| --- | --- |
|
|
15
|
+
| Startup discovery | All three CLIs connected; live model lists shown (Codex 4 models, Antigravity 14, Claude aliases) |
|
|
16
|
+
| Chat turn per provider | Claude Code, Codex, and Antigravity each answered; text streamed into the Live view |
|
|
17
|
+
| Context across `/switch` | Codex and Antigravity both recalled the answer given by the previous provider |
|
|
18
|
+
| Codex approval request | A write command outside the read-only sandbox produced a permission prompt; denial left the workspace unchanged |
|
|
19
|
+
| Ctrl+C during a response | Turn cancelled, child process terminated, REPL continued, unanswered message removed |
|
|
20
|
+
| Ctrl+C at a Codex permission prompt | Turn cancelled, no file created, no leftover process, REPL continued |
|
|
21
|
+
| Usage-limit failover | With a test-only shim making Codex report "You've hit your usage limit", the real REPL showed the alternatives, switched to Claude Code on `1`, retried the message, and answered; history held one clean pair |
|
|
22
|
+
| Parity commands | Live REPL run: `!git log`, `/model` picker, `@file` mention answered by Codex, `/compact` summary, `/export`, `/copy`, `/diff`, `/new`, `/resume`, `/usage` |
|
|
23
|
+
| Live catalogs | OpenRouter public list fetched live (459 models with context and pricing); other providers covered by recorded-payload tests because no keys are present here |
|
|
24
|
+
| `/session save` and `/session load` | Session with a native provider saved and reloaded in a fresh run |
|
|
25
|
+
| `/models`, `/models <provider>`, `/models <tier>` | Live catalogs listed with `CLI account` in place of API pricing |
|
|
26
|
+
| `thwip --version`, `--help`, `--project` | Handled without starting the REPL; invalid project exits with code 2 |
|
|
27
|
+
| `/limits` and `/status` usage windows | Codex and Claude Code account windows (5h, 7d) displayed with reset times after live turns |
|
|
28
|
+
| Leftover processes | None after each run |
|
|
29
|
+
|
|
30
|
+
The defect that blocked the previous attempt was the Codex sandbox value: the
|
|
31
|
+
adapter sent `readOnly` and Codex rejected `thread/start` with an invalid-request
|
|
32
|
+
error. The protocol enum is `read-only`. A regression test now checks the request.
|
|
33
|
+
|
|
34
|
+
Remaining limitations: the Antigravity CLI showed intermittent network resets to
|
|
35
|
+
Google's backend during testing, which surface as turn errors; Claude Code's model
|
|
36
|
+
aliases are a curated list because the CLI exposes no model listing; native usage
|
|
37
|
+
limit failover was observed live only through an injected limit, not a real provider cap; the real Gemini
|
|
38
|
+
CLI ACP path is covered by mocked tests only because it is not installed here.
|
|
39
|
+
|
|
40
|
+
| Area | Evidence | Remaining limitation |
|
|
41
|
+
| --- | --- | --- |
|
|
42
|
+
| Commands and aliases | Offline command smoke tests, invalid input cases | Most smoke tests check exceptions, not all rendered content |
|
|
43
|
+
| Sessions | Save/load, separate fresh conversations, malformed metadata, permissions, project rebinding | Concurrent writes to the same explicitly named session are not coordinated |
|
|
44
|
+
| Provider switching and handoff | All seven provider catalogs, portable history, bounded failover | Live account quotas and model availability unverified |
|
|
45
|
+
| Tool execution | Real temporary-file operations, path containment, process timeout/cancellation, invalid arguments | Shell and code tools retain local user privileges; output capture memory is unbounded |
|
|
46
|
+
| Native Codex launcher | Save-before-launch, consent, flags, missing binary, terminal checks, failures | Mocked process replacement; no conversation transfer |
|
|
47
|
+
| Native CLI connections | Live REPL runs above; mocked protocol tests for approvals, failures, limits, prompt building, discovery parsing | Gemini ACP path mocked only; live limit failover not observed |
|
|
48
|
+
| Provider responses | Mocked native tool continuations; DeepSeek/Groq/OpenRouter streaming, usage-only chunks, 429/500 errors and serialized tool arguments | Other streaming and error branches still have coverage gaps |
|
|
49
|
+
| Display/config/auth | Configuration validation, display settings, credential boundaries, short-key masking | No full terminal/platform matrix |
|
|
50
|
+
| Usage | Atomic writes, malformed records, valid totals | Unknown catalog pricing may appear as zero estimated cost |
|
|
51
|
+
| Website | Production build and deterministic demo completion/replay test | Real browser, layout, clipboard, keyboard and accessibility checks remain incomplete |
|
|
52
|
+
| Dependencies | npm audit: zero advisories; Python installed-dependency audit: none found | Python audit skipped legacy local `thwip 1.0.0` metadata |
|
|
53
|
+
| Packaging | Wheel/source build and metadata checks | Publication is separately verified through the release workflow |
|
|
54
|
+
|
|
55
|
+
The Python suite measured 69% statement coverage with 160 passing tests before
|
|
56
|
+
the final credential-masking tests were added. Coverage is diagnostic evidence,
|
|
57
|
+
not proof that every feature works. The website test uses a minimal DOM stand-in
|
|
58
|
+
and controlled timers, not a browser.
|
|
59
|
+
|
|
60
|
+
This pass fixed default-session save collisions, malformed session/message
|
|
61
|
+
metadata acceptance, tool-argument display crashes, Rich markup interpretation
|
|
62
|
+
in action output, and short-key masking leakage.
|
|
@@ -6,7 +6,7 @@ import pytest
|
|
|
6
6
|
from rich.console import Console
|
|
7
7
|
|
|
8
8
|
from thwip.agents import AgentRegistry
|
|
9
|
-
from thwip.agents.base import AgentDone, Capability, LimitHit, LimitStatus, TextDelta
|
|
9
|
+
from thwip.agents.base import AgentDone, Capability, LimitHit, LimitStatus, ModelInfo, TextDelta
|
|
10
10
|
from thwip.cli import ThwipCLI
|
|
11
11
|
from thwip.config import ThwipConfig, get_usage_path
|
|
12
12
|
from thwip.limits import UsageTracker
|
|
@@ -20,6 +20,9 @@ def cli(tmp_path, monkeypatch):
|
|
|
20
20
|
cli = ThwipCLI.__new__(ThwipCLI)
|
|
21
21
|
cli.config = ThwipConfig(project=str(tmp_path), auto_save=False)
|
|
22
22
|
cli.registry = AgentRegistry(cli.config)
|
|
23
|
+
async def no_native_discovery(project):
|
|
24
|
+
return None
|
|
25
|
+
monkeypatch.setattr(cli.registry, 'connect_native_agents', no_native_discovery)
|
|
23
26
|
for a in cli.registry.list_agents():
|
|
24
27
|
monkeypatch.setattr(a, 'is_installed', lambda: True)
|
|
25
28
|
monkeypatch.setattr(a, 'is_configured', lambda: False)
|
|
@@ -162,3 +165,100 @@ async def test_failover_does_not_retry_already_failed_provider(cli, monkeypatch)
|
|
|
162
165
|
cli.config.limits.auto_switch = True
|
|
163
166
|
await cli.process_user_message('hello')
|
|
164
167
|
assert len(attempts) <= len(agents), f'Repeated failed providers: {attempts}'
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def test_main_handles_version_and_help_without_starting_the_repl(capsys):
|
|
171
|
+
from thwip import __version__
|
|
172
|
+
from thwip import cli as cli_module
|
|
173
|
+
|
|
174
|
+
with pytest.raises(SystemExit) as exit_info:
|
|
175
|
+
cli_module.main(['--version'])
|
|
176
|
+
assert exit_info.value.code == 0 and f'thwip {__version__}' in capsys.readouterr().out
|
|
177
|
+
with pytest.raises(SystemExit) as exit_info:
|
|
178
|
+
cli_module.main(['--help'])
|
|
179
|
+
assert exit_info.value.code == 0 and '--project' in capsys.readouterr().out
|
|
180
|
+
with pytest.raises(SystemExit) as exit_info:
|
|
181
|
+
cli_module.main(['--project', '/definitely/missing/dir'])
|
|
182
|
+
assert exit_info.value.code == 2
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
@pytest.mark.asyncio
|
|
186
|
+
async def test_failed_turn_removes_unanswered_user_message(cli, monkeypatch):
|
|
187
|
+
class Broken:
|
|
188
|
+
name = 'openai'
|
|
189
|
+
display_name = 'Broken'
|
|
190
|
+
company = 'OpenAI'
|
|
191
|
+
native_tools = True
|
|
192
|
+
project = '.'
|
|
193
|
+
def is_configured(self):
|
|
194
|
+
return True
|
|
195
|
+
def is_installed(self):
|
|
196
|
+
return True
|
|
197
|
+
def get_capabilities_for_model(self, model):
|
|
198
|
+
return set()
|
|
199
|
+
async def chat(self, **kwargs):
|
|
200
|
+
raise RuntimeError('Native CLI rejected thread/start')
|
|
201
|
+
yield
|
|
202
|
+
cli.current_agent = Broken()
|
|
203
|
+
cli.session.current_agent = 'openai'
|
|
204
|
+
await cli.process_user_message('hello')
|
|
205
|
+
assert cli.session.messages == []
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
@pytest.mark.asyncio
|
|
209
|
+
async def test_ctrl_c_at_native_permission_prompt_cancels_turn_and_closes_adapter(cli, monkeypatch):
|
|
210
|
+
from thwip.agents.base import NativePermission
|
|
211
|
+
from thwip.cli import TurnInterrupted
|
|
212
|
+
|
|
213
|
+
closed = []
|
|
214
|
+
|
|
215
|
+
class NativeAsking:
|
|
216
|
+
name = 'openai'
|
|
217
|
+
display_name = 'Native'
|
|
218
|
+
company = 'OpenAI'
|
|
219
|
+
native_tools = True
|
|
220
|
+
project = '.'
|
|
221
|
+
def is_configured(self):
|
|
222
|
+
return True
|
|
223
|
+
def is_installed(self):
|
|
224
|
+
return True
|
|
225
|
+
def get_capabilities_for_model(self, model):
|
|
226
|
+
return set()
|
|
227
|
+
async def chat(self, **kwargs):
|
|
228
|
+
try:
|
|
229
|
+
yield NativePermission('run something')
|
|
230
|
+
yield TextDelta(content='never reached')
|
|
231
|
+
finally:
|
|
232
|
+
closed.append(True)
|
|
233
|
+
|
|
234
|
+
async def interrupted(self, question):
|
|
235
|
+
raise TurnInterrupted
|
|
236
|
+
|
|
237
|
+
monkeypatch.setattr(ThwipCLI, '_ask_yes_no', interrupted)
|
|
238
|
+
cli.current_agent = NativeAsking()
|
|
239
|
+
cli.session.current_agent = 'openai'
|
|
240
|
+
await cli._run_interruptible(cli.process_user_message('hello'))
|
|
241
|
+
assert cli.session.messages == []
|
|
242
|
+
assert closed == [True]
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
@pytest.mark.asyncio
|
|
246
|
+
async def test_failover_chain_model_must_be_listed_by_provider(cli, monkeypatch):
|
|
247
|
+
"""A chain entry like claude/claude-opus-5 must not force an unlisted model onto a native adapter."""
|
|
248
|
+
from thwip.agents.native_print import PrintAgent
|
|
249
|
+
|
|
250
|
+
native = PrintAgent('claude', '.')
|
|
251
|
+
native.ready = True
|
|
252
|
+
native.available_models = [ModelInfo(id='fable', name='Fable', is_default=True)]
|
|
253
|
+
monkeypatch.setattr(native, 'is_installed', lambda: True)
|
|
254
|
+
cli.registry._agents['claude'] = native
|
|
255
|
+
cli.config.fallback.chain = ['claude/claude-opus-5']
|
|
256
|
+
switched = []
|
|
257
|
+
|
|
258
|
+
async def fake_switch(name, model=''):
|
|
259
|
+
switched.append((name, model))
|
|
260
|
+
monkeypatch.setattr(cli, 'cmd_switch', fake_switch)
|
|
261
|
+
monkeypatch.setattr('builtins.input', lambda *args: '1')
|
|
262
|
+
cli.current_agent = cli.registry.get_agent('openai')
|
|
263
|
+
await cli.handle_limit_failover(LimitHit(error_type=LimitStatus.QUOTA_EXHAUSTED, message='usage limit'), {'openai'})
|
|
264
|
+
assert switched == [('claude', 'fable')]
|