thwip-cli 1.3.0__tar.gz → 1.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (56) hide show
  1. {thwip_cli-1.3.0 → thwip_cli-1.4.0}/PKG-INFO +50 -24
  2. {thwip_cli-1.3.0 → thwip_cli-1.4.0}/README.md +49 -23
  3. thwip_cli-1.4.0/docs/verification.md +58 -0
  4. {thwip_cli-1.3.0 → thwip_cli-1.4.0}/pyproject.toml +1 -1
  5. {thwip_cli-1.3.0 → thwip_cli-1.4.0}/tests/test_audit_regressions.py +41 -0
  6. {thwip_cli-1.3.0 → thwip_cli-1.4.0}/tests/test_cli.py +3 -2
  7. thwip_cli-1.4.0/tests/test_native_agents.py +198 -0
  8. thwip_cli-1.4.0/tests/test_native_print.py +234 -0
  9. thwip_cli-1.4.0/tests/test_native_rpc.py +51 -0
  10. {thwip_cli-1.3.0 → thwip_cli-1.4.0}/thwip/__init__.py +1 -1
  11. {thwip_cli-1.3.0 → thwip_cli-1.4.0}/thwip/agents/__init__.py +42 -0
  12. {thwip_cli-1.3.0 → thwip_cli-1.4.0}/thwip/agents/base.py +14 -1
  13. thwip_cli-1.4.0/thwip/agents/native_agent.py +281 -0
  14. thwip_cli-1.4.0/thwip/agents/native_common.py +102 -0
  15. thwip_cli-1.4.0/thwip/agents/native_print.py +327 -0
  16. thwip_cli-1.4.0/thwip/agents/native_rpc.py +86 -0
  17. {thwip_cli-1.3.0 → thwip_cli-1.4.0}/thwip/cli.py +149 -19
  18. {thwip_cli-1.3.0 → thwip_cli-1.4.0}/thwip/handoff.py +3 -1
  19. {thwip_cli-1.3.0 → thwip_cli-1.4.0}/uv.lock +1 -1
  20. thwip_cli-1.3.0/docs/verification.md +0 -28
  21. {thwip_cli-1.3.0 → thwip_cli-1.4.0}/.github/workflows/publish.yml +0 -0
  22. {thwip_cli-1.3.0 → thwip_cli-1.4.0}/.gitignore +0 -0
  23. {thwip_cli-1.3.0 → thwip_cli-1.4.0}/LICENSE +0 -0
  24. {thwip_cli-1.3.0 → thwip_cli-1.4.0}/docs/handoff-research.md +0 -0
  25. {thwip_cli-1.3.0 → thwip_cli-1.4.0}/install.sh +0 -0
  26. {thwip_cli-1.3.0 → thwip_cli-1.4.0}/tests/test_agents.py +0 -0
  27. {thwip_cli-1.3.0 → thwip_cli-1.4.0}/tests/test_compatible_streaming.py +0 -0
  28. {thwip_cli-1.3.0 → thwip_cli-1.4.0}/tests/test_config.py +0 -0
  29. {thwip_cli-1.3.0 → thwip_cli-1.4.0}/tests/test_detector.py +0 -0
  30. {thwip_cli-1.3.0 → thwip_cli-1.4.0}/tests/test_handoff.py +0 -0
  31. {thwip_cli-1.3.0 → thwip_cli-1.4.0}/tests/test_native_launcher.py +0 -0
  32. {thwip_cli-1.3.0 → thwip_cli-1.4.0}/tests/test_repair_verification.py +0 -0
  33. {thwip_cli-1.3.0 → thwip_cli-1.4.0}/tests/test_session.py +0 -0
  34. {thwip_cli-1.3.0 → thwip_cli-1.4.0}/tests/test_tools.py +0 -0
  35. {thwip_cli-1.3.0 → thwip_cli-1.4.0}/tests/test_utils.py +0 -0
  36. {thwip_cli-1.3.0 → thwip_cli-1.4.0}/thwip/__main__.py +0 -0
  37. {thwip_cli-1.3.0 → thwip_cli-1.4.0}/thwip/agents/chat_messages.py +0 -0
  38. {thwip_cli-1.3.0 → thwip_cli-1.4.0}/thwip/agents/claude_agent.py +0 -0
  39. {thwip_cli-1.3.0 → thwip_cli-1.4.0}/thwip/agents/deepseek_agent.py +0 -0
  40. {thwip_cli-1.3.0 → thwip_cli-1.4.0}/thwip/agents/google_agent.py +0 -0
  41. {thwip_cli-1.3.0 → thwip_cli-1.4.0}/thwip/agents/groq_agent.py +0 -0
  42. {thwip_cli-1.3.0 → thwip_cli-1.4.0}/thwip/agents/ollama_agent.py +0 -0
  43. {thwip_cli-1.3.0 → thwip_cli-1.4.0}/thwip/agents/openai_agent.py +0 -0
  44. {thwip_cli-1.3.0 → thwip_cli-1.4.0}/thwip/agents/openrouter_agent.py +0 -0
  45. {thwip_cli-1.3.0 → thwip_cli-1.4.0}/thwip/config.py +0 -0
  46. {thwip_cli-1.3.0 → thwip_cli-1.4.0}/thwip/detector.py +0 -0
  47. {thwip_cli-1.3.0 → thwip_cli-1.4.0}/thwip/limits.py +0 -0
  48. {thwip_cli-1.3.0 → thwip_cli-1.4.0}/thwip/session.py +0 -0
  49. {thwip_cli-1.3.0 → thwip_cli-1.4.0}/thwip/shortcuts.py +0 -0
  50. {thwip_cli-1.3.0 → thwip_cli-1.4.0}/thwip/theme.py +0 -0
  51. {thwip_cli-1.3.0 → thwip_cli-1.4.0}/thwip/tools/__init__.py +0 -0
  52. {thwip_cli-1.3.0 → thwip_cli-1.4.0}/thwip/tools/code_runner.py +0 -0
  53. {thwip_cli-1.3.0 → thwip_cli-1.4.0}/thwip/tools/file_editor.py +0 -0
  54. {thwip_cli-1.3.0 → thwip_cli-1.4.0}/thwip/tools/git_ops.py +0 -0
  55. {thwip_cli-1.3.0 → thwip_cli-1.4.0}/thwip/tools/terminal.py +0 -0
  56. {thwip_cli-1.3.0 → thwip_cli-1.4.0}/thwip/utils.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: thwip-cli
3
- Version: 1.3.0
3
+ Version: 1.4.0
4
4
  Summary: Universal coding agent multiplexer: detect, switch, and route between AI coding agents seamlessly
5
5
  Project-URL: Homepage, https://github.com/tanmayhutt/thwip-cli
6
6
  Project-URL: Repository, https://github.com/tanmayhutt/thwip-cli
@@ -46,6 +46,7 @@ Description-Content-Type: text/markdown
46
46
  ## Features
47
47
 
48
48
  - **Auto-Detection**: Discovers installed AI coding agents (Claude Code, Antigravity, Gemini CLI, OpenAI/Codex, Aider, Copilot, Cursor, Windsurf, Cline, Ollama) and configured credentials.
49
+ - **Existing Sign-ins**: Chats through installed Codex, Claude Code, and Antigravity CLIs using their own logins and live model lists, with no API key required.
49
50
  - **Context Portability**: Switch providers with stored conversational text. Working files remain in the selected local project; they are not automatically uploaded.
50
51
  - **Handoff Preview**: Inspect text continuity, omitted state, capability changes, approximate context pressure, and a text fingerprint before switching. Runs locally without model calls.
51
52
  - **Dynamic UI**: Terminal interface adapts its status bar, capabilities, and theme based on the active provider.
@@ -81,6 +82,8 @@ Start the interactive terminal in your current project directory:
81
82
 
82
83
  ```bash
83
84
  thwip
85
+ thwip --project ~/code/my-app # open a specific project
86
+ thwip --version
84
87
  ```
85
88
 
86
89
  ---
@@ -91,13 +94,13 @@ thwip
91
94
  |:---|:---|
92
95
  | `/switch [agent] [model]` | Switch active agent or model mid-conversation |
93
96
  | `/handoff [agent] [model]` | Preview a target locally without switching or sending data |
94
- | `/native codex` | Save and leave Thwip for the installed Codex CLI using its own authentication |
97
+ | `/native codex` | Save and leave Thwip for the installed Codex CLI itself (launcher, no context transfer) |
95
98
  | `/agents` | Show all detected coding agents, company status, and capabilities |
96
99
  | `/models [agent]` | List available models for current or target agent |
97
100
  | `/key [provider]` | Enter an API key securely without placing it in prompt history |
98
101
  | `/status` | Display current session, project, and token stats |
99
102
  | `/limits` | View token usage, quota, and spend metrics |
100
- | `/detect` | Re-scan system for newly installed coding agents |
103
+ | `/detect` | Re-scan installed agents and reconnect CLI sign-ins |
101
104
  | `/session save [name]` | Save current chat session |
102
105
  | `/session load <name>` | Load a previously saved session |
103
106
  | `/session list` | List all saved sessions |
@@ -110,23 +113,46 @@ thwip
110
113
  | `Ctrl + H` | View history |
111
114
  | `/quit` | Exit thwip |
112
115
 
113
- Short aliases are available for frequent commands: `/a`, `/m`, `/s`, `/k`, `/g`, and `/t`.
114
-
115
- ### Existing Codex login
116
-
117
- Use `/native codex` to open the installed native CLI without configuring a Thwip API
118
- key. After confirmation, Thwip saves its session and replaces itself with Codex in
119
- the selected project, using a read-only sandbox and on-request approvals. Codex
120
- owns authentication, model selection, permissions, and usage limits. An expired
121
- or missing login must be resolved in Codex itself.
122
-
123
- This is a launcher, not an in-REPL provider adapter. Conversation history is not
124
- transferred, native activity is not included in Thwip usage totals, and exiting
125
- Codex does not automatically restart Thwip. Restart `thwip` and use `/session load`
126
- to resume the saved Thwip conversation. Claude and Gemini native launchers are not
127
- implemented. Direct API mode remains unchanged.
128
-
129
- ---
116
+ Short aliases are available for frequent commands: `/a`, `/m`, `/s`, `/sw`, `/k`, `/g`, and `/t`.
117
+
118
+ ### Existing CLI sign-ins (no API key needed)
119
+
120
+ If Codex, Claude Code, or the Antigravity CLI is installed and signed in, Thwip
121
+ connects to it at startup and uses that sign-in for chat. No API key is copied or
122
+ required, and Thwip never reads the CLI's stored credentials. A configured direct
123
+ API key for the same provider always takes precedence over the native connection.
124
+
125
+ | Provider | Installed CLI | Transport | Model list |
126
+ |:---|:---|:---|:---|
127
+ | OpenAI | `codex` | Codex App Server (JSON-RPC over stdio) | Live from `model/list` |
128
+ | Anthropic | `claude` | Claude Code print mode (`stream-json`) | Aliases `fable`, `opus`, `sonnet`, `haiku`; explicit IDs pass through |
129
+ | Google | `agy` (Antigravity CLI) or `gemini` | Antigravity print mode (`stream-json`) or Gemini ACP | Live from `agy models` or the ACP session |
130
+
131
+ `/models` refreshes the catalog supplied by each CLI. An explicit model ID that is
132
+ not in the catalog is passed to the CLI for validation instead of being rejected by
133
+ a bundled list. A model available in a desktop app may still be absent from the
134
+ installed CLI's catalog or account access.
135
+
136
+ Native connections are read-only by default. Codex starts with a read-only sandbox
137
+ and asks before operations outside it; approval requests appear in Thwip with the
138
+ command or file list and default to denial. Claude Code and the Antigravity CLI run
139
+ in their non-interactive print modes, where tools that would need an approval are
140
+ declined by the CLI itself. Each turn sends the portable text conversation to a
141
+ fresh native session, so native reasoning and tool state do not carry between turns.
142
+ Ctrl+C interrupts the current turn and stops the child process; the unanswered
143
+ message is removed so it can be re-sent or handed to another provider.
144
+
145
+ Native billing and usage limits are managed by each CLI account. Thwip records the
146
+ token counts the CLIs report but does not estimate cost for them. Codex and Claude
147
+ Code also report their account usage windows (5 hour and 7 day) after each response;
148
+ `/limits` and `/status` show the used percentage and reset time. When a CLI reports
149
+ an exhausted usage limit, the standard failover prompt offers the other connected
150
+ providers.
151
+
152
+ `/native codex` remains available as a launcher: it saves the Thwip session and
153
+ replaces Thwip with the Codex CLI itself in the selected project. Conversation
154
+ history is not transferred by the launcher; use `/session load` after restarting
155
+ Thwip to resume.
130
156
 
131
157
  ## Auditable handoffs
132
158
 
@@ -165,9 +191,9 @@ See [research and prior art](docs/handoff-research.md) for the differentiation r
165
191
 
166
192
  | Company | Agent | Capabilities |
167
193
  |:---|:---|:---|
168
- | Anthropic | Claude API (Fable 5, Opus 5, Sonnet 5, Haiku 4.5) | Chat, File Edit, Code Run, Terminal, Git |
169
- | Google | Gemini API (3.1 Pro Preview, 3.7 Flash, 3.5 Flash-Lite) | Chat, File Edit, Code Run, Terminal, Git |
170
- | OpenAI | OpenAI API (GPT-5.6 Sol, Terra, Luna) | Chat, File Edit, Code Run, Terminal, Git |
194
+ | Anthropic | Claude Code sign-in, or Claude API (Fable 5, Opus 5, Sonnet 5, Haiku 4.5) | Chat, File Edit, Code Run, Terminal, Git |
195
+ | Google | Antigravity or Gemini CLI sign-in, or Gemini API (3.1 Pro Preview, 3.7 Flash, 3.5 Flash-Lite) | Chat, File Edit, Code Run, Terminal, Git |
196
+ | OpenAI | Codex CLI sign-in, or OpenAI API (GPT-5.6 Sol, Terra, Luna) | Chat, File Edit, Code Run, Terminal, Git |
171
197
  | DeepSeek | DeepSeek V3 / R1 Reasoner | Chat, File Edit, Code Run, Reasoning |
172
198
  | Groq | GPT-OSS 120B (default); Llama 3.3 for eligible enterprise accounts only | Chat, File Edit, Code Run |
173
199
  | Ollama | Local Models (Llama 3.3, Qwen Coder, DeepSeek R1) | Chat, File Edit, Code Run (Local, Offline) |
@@ -179,7 +205,7 @@ See [research and prior art](docs/handoff-research.md) for the differentiation r
179
205
 
180
206
  thwip auto-detects existing API keys from environment variables and existing agent configs (`~/.claude.json`, `~/.gemini/config.json`). Configuration can also be set manually:
181
207
 
182
- Installed apps, CLI sign-ins, and API access are separate. thwip can report a detected Claude, Gemini, or Codex CLI login, but the current SDK adapters require a provider API key. A ChatGPT, Claude, or Google subscription does not automatically provide a reusable third-party API key. Ollama needs no key when its local server is running.
208
+ Installed CLI sign-ins and API access are separate paths. A signed-in Codex, Claude Code, or Antigravity CLI is used directly through its own protocol (see above). The direct SDK adapters for Anthropic, Google, OpenAI, DeepSeek, Groq, and OpenRouter require a provider API key. Ollama needs no key when its local server is running.
183
209
 
184
210
  ```toml
185
211
  [defaults]
@@ -9,6 +9,7 @@
9
9
  ## Features
10
10
 
11
11
  - **Auto-Detection**: Discovers installed AI coding agents (Claude Code, Antigravity, Gemini CLI, OpenAI/Codex, Aider, Copilot, Cursor, Windsurf, Cline, Ollama) and configured credentials.
12
+ - **Existing Sign-ins**: Chats through installed Codex, Claude Code, and Antigravity CLIs using their own logins and live model lists, with no API key required.
12
13
  - **Context Portability**: Switch providers with stored conversational text. Working files remain in the selected local project; they are not automatically uploaded.
13
14
  - **Handoff Preview**: Inspect text continuity, omitted state, capability changes, approximate context pressure, and a text fingerprint before switching. Runs locally without model calls.
14
15
  - **Dynamic UI**: Terminal interface adapts its status bar, capabilities, and theme based on the active provider.
@@ -44,6 +45,8 @@ Start the interactive terminal in your current project directory:
44
45
 
45
46
  ```bash
46
47
  thwip
48
+ thwip --project ~/code/my-app # open a specific project
49
+ thwip --version
47
50
  ```
48
51
 
49
52
  ---
@@ -54,13 +57,13 @@ thwip
54
57
  |:---|:---|
55
58
  | `/switch [agent] [model]` | Switch active agent or model mid-conversation |
56
59
  | `/handoff [agent] [model]` | Preview a target locally without switching or sending data |
57
- | `/native codex` | Save and leave Thwip for the installed Codex CLI using its own authentication |
60
+ | `/native codex` | Save and leave Thwip for the installed Codex CLI itself (launcher, no context transfer) |
58
61
  | `/agents` | Show all detected coding agents, company status, and capabilities |
59
62
  | `/models [agent]` | List available models for current or target agent |
60
63
  | `/key [provider]` | Enter an API key securely without placing it in prompt history |
61
64
  | `/status` | Display current session, project, and token stats |
62
65
  | `/limits` | View token usage, quota, and spend metrics |
63
- | `/detect` | Re-scan system for newly installed coding agents |
66
+ | `/detect` | Re-scan installed agents and reconnect CLI sign-ins |
64
67
  | `/session save [name]` | Save current chat session |
65
68
  | `/session load <name>` | Load a previously saved session |
66
69
  | `/session list` | List all saved sessions |
@@ -73,23 +76,46 @@ thwip
73
76
  | `Ctrl + H` | View history |
74
77
  | `/quit` | Exit thwip |
75
78
 
76
- Short aliases are available for frequent commands: `/a`, `/m`, `/s`, `/k`, `/g`, and `/t`.
77
-
78
- ### Existing Codex login
79
-
80
- Use `/native codex` to open the installed native CLI without configuring a Thwip API
81
- key. After confirmation, Thwip saves its session and replaces itself with Codex in
82
- the selected project, using a read-only sandbox and on-request approvals. Codex
83
- owns authentication, model selection, permissions, and usage limits. An expired
84
- or missing login must be resolved in Codex itself.
85
-
86
- This is a launcher, not an in-REPL provider adapter. Conversation history is not
87
- transferred, native activity is not included in Thwip usage totals, and exiting
88
- Codex does not automatically restart Thwip. Restart `thwip` and use `/session load`
89
- to resume the saved Thwip conversation. Claude and Gemini native launchers are not
90
- implemented. Direct API mode remains unchanged.
91
-
92
- ---
79
+ Short aliases are available for frequent commands: `/a`, `/m`, `/s`, `/sw`, `/k`, `/g`, and `/t`.
80
+
81
+ ### Existing CLI sign-ins (no API key needed)
82
+
83
+ If Codex, Claude Code, or the Antigravity CLI is installed and signed in, Thwip
84
+ connects to it at startup and uses that sign-in for chat. No API key is copied or
85
+ required, and Thwip never reads the CLI's stored credentials. A configured direct
86
+ API key for the same provider always takes precedence over the native connection.
87
+
88
+ | Provider | Installed CLI | Transport | Model list |
89
+ |:---|:---|:---|:---|
90
+ | OpenAI | `codex` | Codex App Server (JSON-RPC over stdio) | Live from `model/list` |
91
+ | Anthropic | `claude` | Claude Code print mode (`stream-json`) | Aliases `fable`, `opus`, `sonnet`, `haiku`; explicit IDs pass through |
92
+ | Google | `agy` (Antigravity CLI) or `gemini` | Antigravity print mode (`stream-json`) or Gemini ACP | Live from `agy models` or the ACP session |
93
+
94
+ `/models` refreshes the catalog supplied by each CLI. An explicit model ID that is
95
+ not in the catalog is passed to the CLI for validation instead of being rejected by
96
+ a bundled list. A model available in a desktop app may still be absent from the
97
+ installed CLI's catalog or account access.
98
+
99
+ Native connections are read-only by default. Codex starts with a read-only sandbox
100
+ and asks before operations outside it; approval requests appear in Thwip with the
101
+ command or file list and default to denial. Claude Code and the Antigravity CLI run
102
+ in their non-interactive print modes, where tools that would need an approval are
103
+ declined by the CLI itself. Each turn sends the portable text conversation to a
104
+ fresh native session, so native reasoning and tool state do not carry between turns.
105
+ Ctrl+C interrupts the current turn and stops the child process; the unanswered
106
+ message is removed so it can be re-sent or handed to another provider.
107
+
108
+ Native billing and usage limits are managed by each CLI account. Thwip records the
109
+ token counts the CLIs report but does not estimate cost for them. Codex and Claude
110
+ Code also report their account usage windows (5 hour and 7 day) after each response;
111
+ `/limits` and `/status` show the used percentage and reset time. When a CLI reports
112
+ an exhausted usage limit, the standard failover prompt offers the other connected
113
+ providers.
114
+
115
+ `/native codex` remains available as a launcher: it saves the Thwip session and
116
+ replaces Thwip with the Codex CLI itself in the selected project. Conversation
117
+ history is not transferred by the launcher; use `/session load` after restarting
118
+ Thwip to resume.
93
119
 
94
120
  ## Auditable handoffs
95
121
 
@@ -128,9 +154,9 @@ See [research and prior art](docs/handoff-research.md) for the differentiation r
128
154
 
129
155
  | Company | Agent | Capabilities |
130
156
  |:---|:---|:---|
131
- | Anthropic | Claude API (Fable 5, Opus 5, Sonnet 5, Haiku 4.5) | Chat, File Edit, Code Run, Terminal, Git |
132
- | Google | Gemini API (3.1 Pro Preview, 3.7 Flash, 3.5 Flash-Lite) | Chat, File Edit, Code Run, Terminal, Git |
133
- | OpenAI | OpenAI API (GPT-5.6 Sol, Terra, Luna) | Chat, File Edit, Code Run, Terminal, Git |
157
+ | Anthropic | Claude Code sign-in, or Claude API (Fable 5, Opus 5, Sonnet 5, Haiku 4.5) | Chat, File Edit, Code Run, Terminal, Git |
158
+ | Google | Antigravity or Gemini CLI sign-in, or Gemini API (3.1 Pro Preview, 3.7 Flash, 3.5 Flash-Lite) | Chat, File Edit, Code Run, Terminal, Git |
159
+ | OpenAI | Codex CLI sign-in, or OpenAI API (GPT-5.6 Sol, Terra, Luna) | Chat, File Edit, Code Run, Terminal, Git |
134
160
  | DeepSeek | DeepSeek V3 / R1 Reasoner | Chat, File Edit, Code Run, Reasoning |
135
161
  | Groq | GPT-OSS 120B (default); Llama 3.3 for eligible enterprise accounts only | Chat, File Edit, Code Run |
136
162
  | Ollama | Local Models (Llama 3.3, Qwen Coder, DeepSeek R1) | Chat, File Edit, Code Run (Local, Offline) |
@@ -142,7 +168,7 @@ See [research and prior art](docs/handoff-research.md) for the differentiation r
142
168
 
143
169
  thwip auto-detects existing API keys from environment variables and existing agent configs (`~/.claude.json`, `~/.gemini/config.json`). Configuration can also be set manually:
144
170
 
145
- Installed apps, CLI sign-ins, and API access are separate. thwip can report a detected Claude, Gemini, or Codex CLI login, but the current SDK adapters require a provider API key. A ChatGPT, Claude, or Google subscription does not automatically provide a reusable third-party API key. Ollama needs no key when its local server is running.
171
+ Installed CLI sign-ins and API access are separate paths. A signed-in Codex, Claude Code, or Antigravity CLI is used directly through its own protocol (see above). The direct SDK adapters for Anthropic, Google, OpenAI, DeepSeek, Groq, and OpenRouter require a provider API key. Ollama needs no key when its local server is running.
146
172
 
147
173
  ```toml
148
174
  [defaults]
@@ -0,0 +1,58 @@
1
+ # Verification status
2
+
3
+ Verified locally on 2026-09-24 for the unreleased v1.4.0 source. The Python suite
4
+ passes 216 offline tests. Exhaustive behavior across every provider and
5
+ configuration has not been established.
6
+
7
+ ## Native CLI connections (2026-09-24)
8
+
9
+ Live checks were run on macOS through a pseudo-terminal driving the real REPL with
10
+ the installed Codex CLI 0.152.1, Claude Code 2.1.281, and Antigravity CLI 1.2.8,
11
+ each using its existing sign-in. No API keys were configured.
12
+
13
+ | Flow | Result |
14
+ | --- | --- |
15
+ | Startup discovery | All three CLIs connected; live model lists shown (Codex 4 models, Antigravity 14, Claude aliases) |
16
+ | Chat turn per provider | Claude Code, Codex, and Antigravity each answered; text streamed into the Live view |
17
+ | Context across `/switch` | Codex and Antigravity both recalled the answer given by the previous provider |
18
+ | Codex approval request | A write command outside the read-only sandbox produced a permission prompt; denial left the workspace unchanged |
19
+ | Ctrl+C during a response | Turn cancelled, child process terminated, REPL continued, unanswered message removed |
20
+ | `/session save` and `/session load` | Session with a native provider saved and reloaded in a fresh run |
21
+ | `/models`, `/models <provider>`, `/models <tier>` | Live catalogs listed with `CLI account` in place of API pricing |
22
+ | `thwip --version`, `--help`, `--project` | Handled without starting the REPL; invalid project exits with code 2 |
23
+ | `/limits` and `/status` usage windows | Codex and Claude Code account windows (5h, 7d) displayed with reset times after live turns |
24
+ | Leftover processes | None after each run |
25
+
26
+ The defect that blocked the previous attempt was the Codex sandbox value: the
27
+ adapter sent `readOnly` and Codex rejected `thread/start` with an invalid-request
28
+ error. The protocol enum is `read-only`. A regression test now checks the request.
29
+
30
+ Remaining limitations: the Antigravity CLI showed intermittent network resets to
31
+ Google's backend during testing, which surface as turn errors; Claude Code's model
32
+ aliases are a curated list because the CLI exposes no model listing; native usage
33
+ limit failover is unit-tested from error text, not observed live; the real Gemini
34
+ CLI ACP path is covered by mocked tests only because it is not installed here.
35
+
36
+ | Area | Evidence | Remaining limitation |
37
+ | --- | --- | --- |
38
+ | Commands and aliases | Offline command smoke tests, invalid input cases | Most smoke tests check exceptions, not all rendered content |
39
+ | Sessions | Save/load, separate fresh conversations, malformed metadata, permissions, project rebinding | Concurrent writes to the same explicitly named session are not coordinated |
40
+ | Provider switching and handoff | All seven provider catalogs, portable history, bounded failover | Live account quotas and model availability unverified |
41
+ | Tool execution | Real temporary-file operations, path containment, process timeout/cancellation, invalid arguments | Shell and code tools retain local user privileges; output capture memory is unbounded |
42
+ | Native Codex launcher | Save-before-launch, consent, flags, missing binary, terminal checks, failures | Mocked process replacement; no conversation transfer |
43
+ | Native CLI connections | Live REPL runs above; mocked protocol tests for approvals, failures, limits, prompt building, discovery parsing | Gemini ACP path mocked only; live limit failover not observed |
44
+ | Provider responses | Mocked native tool continuations; DeepSeek/Groq/OpenRouter streaming, usage-only chunks, 429/500 errors and serialized tool arguments | Other streaming and error branches still have coverage gaps |
45
+ | Display/config/auth | Configuration validation, display settings, credential boundaries, short-key masking | No full terminal/platform matrix |
46
+ | Usage | Atomic writes, malformed records, valid totals | Unknown catalog pricing may appear as zero estimated cost |
47
+ | Website | Production build and deterministic demo completion/replay test | Real browser, layout, clipboard, keyboard and accessibility checks remain incomplete |
48
+ | Dependencies | npm audit: zero advisories; Python installed-dependency audit: none found | Python audit skipped legacy local `thwip 1.0.0` metadata |
49
+ | Packaging | Wheel/source build and metadata checks | Publication is separately verified through the release workflow |
50
+
51
+ The Python suite measured 69% statement coverage with 160 passing tests before
52
+ the final credential-masking tests were added. Coverage is diagnostic evidence,
53
+ not proof that every feature works. The website test uses a minimal DOM stand-in
54
+ and controlled timers, not a browser.
55
+
56
+ This pass fixed default-session save collisions, malformed session/message
57
+ metadata acceptance, tool-argument display crashes, Rich markup interpretation
58
+ in action output, and short-key masking leakage.
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "thwip-cli"
7
- version = "1.3.0"
7
+ version = "1.4.0"
8
8
  description = "Universal coding agent multiplexer: detect, switch, and route between AI coding agents seamlessly"
9
9
  readme = "README.md"
10
10
  license = "MIT"
@@ -20,6 +20,9 @@ def cli(tmp_path, monkeypatch):
20
20
  cli = ThwipCLI.__new__(ThwipCLI)
21
21
  cli.config = ThwipConfig(project=str(tmp_path), auto_save=False)
22
22
  cli.registry = AgentRegistry(cli.config)
23
+ async def no_native_discovery(project):
24
+ return None
25
+ monkeypatch.setattr(cli.registry, 'connect_native_agents', no_native_discovery)
23
26
  for a in cli.registry.list_agents():
24
27
  monkeypatch.setattr(a, 'is_installed', lambda: True)
25
28
  monkeypatch.setattr(a, 'is_configured', lambda: False)
@@ -162,3 +165,41 @@ async def test_failover_does_not_retry_already_failed_provider(cli, monkeypatch)
162
165
  cli.config.limits.auto_switch = True
163
166
  await cli.process_user_message('hello')
164
167
  assert len(attempts) <= len(agents), f'Repeated failed providers: {attempts}'
168
+
169
+
170
+ def test_main_handles_version_and_help_without_starting_the_repl(capsys):
171
+ from thwip import __version__
172
+ from thwip import cli as cli_module
173
+
174
+ with pytest.raises(SystemExit) as exit_info:
175
+ cli_module.main(['--version'])
176
+ assert exit_info.value.code == 0 and f'thwip {__version__}' in capsys.readouterr().out
177
+ with pytest.raises(SystemExit) as exit_info:
178
+ cli_module.main(['--help'])
179
+ assert exit_info.value.code == 0 and '--project' in capsys.readouterr().out
180
+ with pytest.raises(SystemExit) as exit_info:
181
+ cli_module.main(['--project', '/definitely/missing/dir'])
182
+ assert exit_info.value.code == 2
183
+
184
+
185
+ @pytest.mark.asyncio
186
+ async def test_failed_turn_removes_unanswered_user_message(cli, monkeypatch):
187
+ class Broken:
188
+ name = 'openai'
189
+ display_name = 'Broken'
190
+ company = 'OpenAI'
191
+ native_tools = True
192
+ project = '.'
193
+ def is_configured(self):
194
+ return True
195
+ def is_installed(self):
196
+ return True
197
+ def get_capabilities_for_model(self, model):
198
+ return set()
199
+ async def chat(self, **kwargs):
200
+ raise RuntimeError('Native CLI rejected thread/start')
201
+ yield
202
+ cli.current_agent = Broken()
203
+ cli.session.current_agent = 'openai'
204
+ await cli.process_user_message('hello')
205
+ assert cli.session.messages == []
@@ -115,11 +115,12 @@ async def test_unconfigured_agent_shows_setup_guidance(tmp_path):
115
115
  assert len(cli.session.messages) == 0
116
116
 
117
117
 
118
- def test_inline_api_key_is_rejected():
118
+ @pytest.mark.asyncio
119
+ async def test_inline_api_key_is_rejected():
119
120
  cli = ThwipCLI.__new__(ThwipCLI)
120
121
  cli.config = SimpleNamespace(keys={}, key_sources={}, save=lambda: None)
121
122
 
122
- cli.cmd_auth_config("openai", "secret-value")
123
+ await cli.cmd_auth_config("openai", "secret-value")
123
124
 
124
125
  assert cli.config.keys == {}
125
126
  assert cli.config.key_sources == {}
@@ -0,0 +1,198 @@
1
+ """Native protocol behavior without credentials or model requests."""
2
+
3
+ import asyncio
4
+
5
+ import pytest
6
+
7
+ from thwip.agents.base import AgentDone, LimitHit, LimitStatus, ModelInfo, NativePermission, TextDelta
8
+ from thwip.agents.native_agent import NativeAgent
9
+
10
+
11
+ class FakeRPC:
12
+ def __init__(self, provider, failure=False):
13
+ self.events = asyncio.Queue()
14
+ self.requests = []
15
+ self.sent = []
16
+ self.closed = False
17
+ self.provider = provider
18
+ self.failure = failure
19
+
20
+ async def request(self, method, params, **kwargs):
21
+ self.requests.append((method, params))
22
+ if method == 'account/read':
23
+ return {'account': {'type': 'chatgpt'}}
24
+ if method == 'model/list':
25
+ return {'data': [{'model': 'future-model', 'displayName': 'Future', 'isDefault': True}]}
26
+ if method == 'session/new':
27
+ return {'sessionId': 's', 'models': {'currentModelId': 'future-model',
28
+ 'availableModels': [{'modelId': 'future-model', 'name': 'Future'}]}}
29
+ if method == 'thread/start':
30
+ return {'thread': {'id': 't'}}
31
+ if method == 'turn/start':
32
+ await self.events.put({'id': 99, 'method': 'item/commandExecution/requestApproval',
33
+ 'params': {'command': 'touch example'}})
34
+ return {}
35
+ if method == 'session/prompt':
36
+ await self.events.put({'id': 99, 'method': 'session/request_permission', 'params': {
37
+ 'toolCall': {'title': 'Edit'}, 'options': [
38
+ {'kind': 'allow_once', 'optionId': 'yes'}, {'kind': 'reject_once', 'optionId': 'no'}]}})
39
+ while not self.sent:
40
+ await asyncio.sleep(0)
41
+ await self.events.put({'method': 'session/update', 'params': {'update': {
42
+ 'sessionUpdate': 'agent_message_chunk', 'content': {'type': 'text', 'text': 'done'}}}})
43
+ return {'stopReason': 'end_turn'}
44
+ return {}
45
+
46
+ async def send(self, message):
47
+ self.sent.append(message)
48
+ if self.provider == 'openai':
49
+ await self.events.put({'method': 'item/agentMessage/delta', 'params': {'itemId': 'a', 'delta': 'done'}})
50
+ await self.events.put({'method': 'turn/completed', 'params': {
51
+ 'turn': {'status': 'failed' if self.failure else 'completed'}}})
52
+
53
+ async def close(self):
54
+ self.closed = True
55
+
56
+
57
+ @pytest.mark.parametrize('provider', ['openai', 'google'])
58
+ @pytest.mark.asyncio
59
+ async def test_discovered_models_replace_bundled_catalog(provider, monkeypatch):
60
+ agent = NativeAgent(provider, '.')
61
+ rpc = FakeRPC(provider)
62
+ async def connect():
63
+ return rpc
64
+ monkeypatch.setattr(agent, '_connect', connect)
65
+ await agent.refresh_models()
66
+ assert agent.ready and agent.get_default_model() == 'future-model'
67
+ assert agent.get_model_info('explicit-new-model').id == 'explicit-new-model'
68
+ assert rpc.closed
69
+
70
+
71
+ @pytest.mark.parametrize('provider', ['openai', 'google'])
72
+ @pytest.mark.parametrize('approve', [False, True])
73
+ @pytest.mark.asyncio
74
+ async def test_native_permission_response_and_completion(provider, approve, monkeypatch):
75
+ agent = NativeAgent(provider, '.')
76
+ rpc = FakeRPC(provider)
77
+ async def connect():
78
+ return rpc
79
+ monkeypatch.setattr(agent, '_connect', connect)
80
+ events = []
81
+ async for event in agent.chat([{'role': 'user', 'content': 'hello'}], model='future-model'):
82
+ events.append(event)
83
+ if isinstance(event, NativePermission):
84
+ event.approved = approve
85
+ assert any(isinstance(event, TextDelta) and event.content == 'done' for event in events)
86
+ assert isinstance(events[-1], AgentDone)
87
+ response = rpc.sent[0]['result']
88
+ if provider == 'openai':
89
+ assert response['decision'] == ('accept' if approve else 'decline')
90
+ else:
91
+ assert response['outcome']['optionId'] == ('yes' if approve else 'no')
92
+ assert rpc.closed
93
+
94
+
95
+ @pytest.mark.asyncio
96
+ async def test_failed_turn_never_emits_success(monkeypatch):
97
+ agent = NativeAgent('openai', '.')
98
+ rpc = FakeRPC('openai', failure=True)
99
+ async def connect():
100
+ return rpc
101
+ monkeypatch.setattr(agent, '_connect', connect)
102
+ events = []
103
+ with pytest.raises(RuntimeError, match='could not complete'):
104
+ async for event in agent.chat([], model='future-model'):
105
+ events.append(event)
106
+ assert not any(isinstance(event, AgentDone) for event in events)
107
+ assert rpc.closed
108
+
109
+
110
+ @pytest.mark.asyncio
111
+ async def test_discovery_failure_marks_connection_unready(monkeypatch):
112
+ agent = NativeAgent('google', '.')
113
+ async def connect():
114
+ raise TimeoutError()
115
+ monkeypatch.setattr(agent, '_connect', connect)
116
+ await agent.refresh_models()
117
+ assert not agent.ready and 'timed out' in agent.discovery_error
118
+
119
+
120
+ @pytest.mark.asyncio
121
+ async def test_codex_thread_uses_protocol_sandbox_spelling(monkeypatch):
122
+ """Codex App Server rejects camelCase sandbox modes with an invalid-request error."""
123
+ agent = NativeAgent('openai', '.')
124
+ rpc = FakeRPC('openai')
125
+ async def connect():
126
+ return rpc
127
+ monkeypatch.setattr(agent, '_connect', connect)
128
+ async for _ in agent.chat([{'role': 'user', 'content': 'hello'}], model='future-model', system_prompt='Be terse.'):
129
+ pass
130
+ _method, params = next(request for request in rpc.requests if request[0] == 'thread/start')
131
+ assert params['sandbox'] == 'read-only' and params['approvalPolicy'] == 'on-request'
132
+ assert params['developerInstructions'] == 'Be terse.'
133
+ turn = next(params for method, params in rpc.requests if method == 'turn/start')
134
+ assert turn['input'][0]['text'] == 'hello'
135
+
136
+
137
+ @pytest.mark.asyncio
138
+ async def test_codex_usage_limit_failure_becomes_limit_hit(monkeypatch):
139
+ agent = NativeAgent('openai', '.')
140
+ rpc = FakeRPC('openai')
141
+ async def send(message):
142
+ rpc.sent.append(message)
143
+ await rpc.events.put({'method': 'turn/completed', 'params': {'turn': {
144
+ 'status': 'failed', 'error': {'message': 'You have hit your usage limit.'}}}})
145
+ rpc.send = send
146
+ async def connect():
147
+ return rpc
148
+ monkeypatch.setattr(agent, '_connect', connect)
149
+ events = [event async for event in agent.chat([{'role': 'user', 'content': 'hi'}], model='future-model')]
150
+ assert isinstance(events[-1], LimitHit) and events[-1].error_type == LimitStatus.QUOTA_EXHAUSTED
151
+ assert rpc.closed
152
+
153
+
154
+ def test_permission_descriptions_are_readable_and_scrubbed():
155
+ from thwip.agents.native_agent import describe_permission
156
+
157
+ text = describe_permission({"type": "commandExecution", "command": "curl -H 'Authorization: Bearer " + "k" * 40 + "'",
158
+ "cwd": "/repo", "reason": "Fetch data"}, "Codex")
159
+ assert text.startswith("Codex wants to run a command.") and "Directory: /repo" in text and "Reason: Fetch data" in text
160
+ assert "kkkk" not in text and "[redacted]" in text
161
+ files = describe_permission({"type": "fileChange", "changes": [{"path": "a.py", "kind": "update"}]}, "Codex")
162
+ assert "wants to change files" in files and "a.py (update)" in files
163
+ assert "Gemini requests permission for Edit" in describe_permission({"title": "Edit"}, "Gemini")
164
+
165
+
166
+ def test_handoff_accepts_explicit_native_model_ids():
167
+ from thwip.handoff import build_handoff_report, local_model
168
+ from thwip.session import Session
169
+
170
+ agent = NativeAgent('openai', '.')
171
+ agent.available_models = [ModelInfo(id='listed', name='Listed', is_default=True)]
172
+ assert local_model(agent, 'listed').name == 'Listed'
173
+ assert local_model(agent, 'brand-new').id == 'brand-new'
174
+ assert local_model(agent, 'has space') is None
175
+ session = Session(current_agent='openai', current_model='listed')
176
+ report = build_handoff_report(session, agent, agent, 'brand-new')
177
+ assert report.target == 'openai/brand-new' and report.context_pressure == 'unknown'
178
+
179
+
180
+ @pytest.mark.asyncio
181
+ async def test_codex_rate_limit_notification_is_recorded(monkeypatch):
182
+ agent = NativeAgent('openai', '.')
183
+ rpc = FakeRPC('openai')
184
+ async def send(message):
185
+ rpc.sent.append(message)
186
+ await rpc.events.put({'method': 'account/rateLimits/updated', 'params': {'rateLimits': {
187
+ 'primary': {'usedPercent': 2, 'windowDurationMins': 300, 'resetsAt': 1790212672},
188
+ 'secondary': {'usedPercent': 23, 'windowDurationMins': 10080}}}})
189
+ await rpc.events.put({'method': 'turn/completed', 'params': {'turn': {'status': 'completed'}}})
190
+ rpc.send = send
191
+ async def connect():
192
+ return rpc
193
+ monkeypatch.setattr(agent, '_connect', connect)
194
+ events = [event async for event in agent.chat([{'role': 'user', 'content': 'hi'}], model='future-model')]
195
+ assert isinstance(events[-1], AgentDone)
196
+ assert agent.limit_windows == [
197
+ {'label': '5h', 'used_percent': 2, 'resets_at': 1790212672},
198
+ {'label': '7d', 'used_percent': 23, 'resets_at': None}]