thwip-cli 1.2.0__tar.gz → 1.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (55) hide show
  1. {thwip_cli-1.2.0 → thwip_cli-1.4.0}/PKG-INFO +51 -10
  2. {thwip_cli-1.2.0 → thwip_cli-1.4.0}/README.md +50 -9
  3. thwip_cli-1.4.0/docs/verification.md +58 -0
  4. {thwip_cli-1.2.0 → thwip_cli-1.4.0}/pyproject.toml +1 -1
  5. thwip_cli-1.4.0/tests/test_audit_regressions.py +205 -0
  6. {thwip_cli-1.2.0 → thwip_cli-1.4.0}/tests/test_cli.py +27 -2
  7. thwip_cli-1.4.0/tests/test_compatible_streaming.py +75 -0
  8. thwip_cli-1.4.0/tests/test_native_agents.py +198 -0
  9. thwip_cli-1.4.0/tests/test_native_launcher.py +69 -0
  10. thwip_cli-1.4.0/tests/test_native_print.py +234 -0
  11. thwip_cli-1.4.0/tests/test_native_rpc.py +51 -0
  12. thwip_cli-1.4.0/tests/test_repair_verification.py +177 -0
  13. {thwip_cli-1.2.0 → thwip_cli-1.4.0}/tests/test_session.py +41 -0
  14. thwip_cli-1.4.0/tests/test_utils.py +17 -0
  15. {thwip_cli-1.2.0 → thwip_cli-1.4.0}/thwip/__init__.py +1 -1
  16. {thwip_cli-1.2.0 → thwip_cli-1.4.0}/thwip/agents/__init__.py +42 -0
  17. {thwip_cli-1.2.0 → thwip_cli-1.4.0}/thwip/agents/base.py +15 -1
  18. thwip_cli-1.4.0/thwip/agents/chat_messages.py +20 -0
  19. {thwip_cli-1.2.0 → thwip_cli-1.4.0}/thwip/agents/claude_agent.py +6 -1
  20. {thwip_cli-1.2.0 → thwip_cli-1.4.0}/thwip/agents/deepseek_agent.py +2 -1
  21. {thwip_cli-1.2.0 → thwip_cli-1.4.0}/thwip/agents/google_agent.py +5 -1
  22. {thwip_cli-1.2.0 → thwip_cli-1.4.0}/thwip/agents/groq_agent.py +5 -3
  23. thwip_cli-1.4.0/thwip/agents/native_agent.py +281 -0
  24. thwip_cli-1.4.0/thwip/agents/native_common.py +102 -0
  25. thwip_cli-1.4.0/thwip/agents/native_print.py +327 -0
  26. thwip_cli-1.4.0/thwip/agents/native_rpc.py +86 -0
  27. {thwip_cli-1.2.0 → thwip_cli-1.4.0}/thwip/agents/openai_agent.py +8 -1
  28. {thwip_cli-1.2.0 → thwip_cli-1.4.0}/thwip/agents/openrouter_agent.py +2 -1
  29. {thwip_cli-1.2.0 → thwip_cli-1.4.0}/thwip/cli.py +284 -57
  30. {thwip_cli-1.2.0 → thwip_cli-1.4.0}/thwip/config.py +27 -0
  31. {thwip_cli-1.2.0 → thwip_cli-1.4.0}/thwip/detector.py +16 -10
  32. {thwip_cli-1.2.0 → thwip_cli-1.4.0}/thwip/handoff.py +3 -1
  33. {thwip_cli-1.2.0 → thwip_cli-1.4.0}/thwip/limits.py +33 -4
  34. {thwip_cli-1.2.0 → thwip_cli-1.4.0}/thwip/session.py +29 -1
  35. {thwip_cli-1.2.0 → thwip_cli-1.4.0}/thwip/shortcuts.py +1 -0
  36. {thwip_cli-1.2.0 → thwip_cli-1.4.0}/thwip/theme.py +17 -11
  37. {thwip_cli-1.2.0 → thwip_cli-1.4.0}/thwip/tools/__init__.py +22 -1
  38. {thwip_cli-1.2.0 → thwip_cli-1.4.0}/thwip/tools/code_runner.py +4 -6
  39. {thwip_cli-1.2.0 → thwip_cli-1.4.0}/thwip/tools/file_editor.py +1 -1
  40. {thwip_cli-1.2.0 → thwip_cli-1.4.0}/thwip/tools/terminal.py +35 -5
  41. {thwip_cli-1.2.0 → thwip_cli-1.4.0}/thwip/utils.py +1 -1
  42. {thwip_cli-1.2.0 → thwip_cli-1.4.0}/uv.lock +1 -1
  43. {thwip_cli-1.2.0 → thwip_cli-1.4.0}/.github/workflows/publish.yml +0 -0
  44. {thwip_cli-1.2.0 → thwip_cli-1.4.0}/.gitignore +0 -0
  45. {thwip_cli-1.2.0 → thwip_cli-1.4.0}/LICENSE +0 -0
  46. {thwip_cli-1.2.0 → thwip_cli-1.4.0}/docs/handoff-research.md +0 -0
  47. {thwip_cli-1.2.0 → thwip_cli-1.4.0}/install.sh +0 -0
  48. {thwip_cli-1.2.0 → thwip_cli-1.4.0}/tests/test_agents.py +0 -0
  49. {thwip_cli-1.2.0 → thwip_cli-1.4.0}/tests/test_config.py +0 -0
  50. {thwip_cli-1.2.0 → thwip_cli-1.4.0}/tests/test_detector.py +0 -0
  51. {thwip_cli-1.2.0 → thwip_cli-1.4.0}/tests/test_handoff.py +0 -0
  52. {thwip_cli-1.2.0 → thwip_cli-1.4.0}/tests/test_tools.py +0 -0
  53. {thwip_cli-1.2.0 → thwip_cli-1.4.0}/thwip/__main__.py +0 -0
  54. {thwip_cli-1.2.0 → thwip_cli-1.4.0}/thwip/agents/ollama_agent.py +0 -0
  55. {thwip_cli-1.2.0 → thwip_cli-1.4.0}/thwip/tools/git_ops.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: thwip-cli
3
- Version: 1.2.0
3
+ Version: 1.4.0
4
4
  Summary: Universal coding agent multiplexer: detect, switch, and route between AI coding agents seamlessly
5
5
  Project-URL: Homepage, https://github.com/tanmayhutt/thwip-cli
6
6
  Project-URL: Repository, https://github.com/tanmayhutt/thwip-cli
@@ -46,6 +46,7 @@ Description-Content-Type: text/markdown
46
46
  ## Features
47
47
 
48
48
  - **Auto-Detection**: Discovers installed AI coding agents (Claude Code, Antigravity, Gemini CLI, OpenAI/Codex, Aider, Copilot, Cursor, Windsurf, Cline, Ollama) and configured credentials.
49
+ - **Existing Sign-ins**: Chats through installed Codex, Claude Code, and Antigravity CLIs using their own logins and live model lists, with no API key required.
49
50
  - **Context Portability**: Switch providers with stored conversational text. Working files remain in the selected local project; they are not automatically uploaded.
50
51
  - **Handoff Preview**: Inspect text continuity, omitted state, capability changes, approximate context pressure, and a text fingerprint before switching. Runs locally without model calls.
51
52
  - **Dynamic UI**: Terminal interface adapts its status bar, capabilities, and theme based on the active provider.
@@ -81,6 +82,8 @@ Start the interactive terminal in your current project directory:
81
82
 
82
83
  ```bash
83
84
  thwip
85
+ thwip --project ~/code/my-app # open a specific project
86
+ thwip --version
84
87
  ```
85
88
 
86
89
  ---
@@ -91,12 +94,13 @@ thwip
91
94
  |:---|:---|
92
95
  | `/switch [agent] [model]` | Switch active agent or model mid-conversation |
93
96
  | `/handoff [agent] [model]` | Preview a target locally without switching or sending data |
97
+ | `/native codex` | Save and leave Thwip for the installed Codex CLI itself (launcher, no context transfer) |
94
98
  | `/agents` | Show all detected coding agents, company status, and capabilities |
95
99
  | `/models [agent]` | List available models for current or target agent |
96
100
  | `/key [provider]` | Enter an API key securely without placing it in prompt history |
97
101
  | `/status` | Display current session, project, and token stats |
98
102
  | `/limits` | View token usage, quota, and spend metrics |
99
- | `/detect` | Re-scan system for newly installed coding agents |
103
+ | `/detect` | Re-scan installed agents and reconnect CLI sign-ins |
100
104
  | `/session save [name]` | Save current chat session |
101
105
  | `/session load <name>` | Load a previously saved session |
102
106
  | `/session list` | List all saved sessions |
@@ -105,13 +109,50 @@ thwip
105
109
  | `/cost` | Show estimated session and cumulative cost |
106
110
  | `/project [path]` | View or change project working directory |
107
111
  | `Ctrl + S` | Quick switch agent prompt |
108
-
109
- Short aliases are available for frequent commands: `/a`, `/m`, `/s`, `/k`, `/g`, and `/t`.
110
112
  | `Ctrl + T` | Show agent status |
111
113
  | `Ctrl + H` | View history |
112
114
  | `/quit` | Exit thwip |
113
115
 
114
- ---
116
+ Short aliases are available for frequent commands: `/a`, `/m`, `/s`, `/sw`, `/k`, `/g`, and `/t`.
117
+
118
+ ### Existing CLI sign-ins (no API key needed)
119
+
120
+ If Codex, Claude Code, or the Antigravity CLI is installed and signed in, Thwip
121
+ connects to it at startup and uses that sign-in for chat. No API key is copied or
122
+ required, and Thwip never reads the CLI's stored credentials. A configured direct
123
+ API key for the same provider always takes precedence over the native connection.
124
+
125
+ | Provider | Installed CLI | Transport | Model list |
126
+ |:---|:---|:---|:---|
127
+ | OpenAI | `codex` | Codex App Server (JSON-RPC over stdio) | Live from `model/list` |
128
+ | Anthropic | `claude` | Claude Code print mode (`stream-json`) | Aliases `fable`, `opus`, `sonnet`, `haiku`; explicit IDs pass through |
129
+ | Google | `agy` (Antigravity CLI) or `gemini` | Antigravity print mode (`stream-json`) or Gemini ACP | Live from `agy models` or the ACP session |
130
+
131
+ `/models` refreshes the catalog supplied by each CLI. An explicit model ID that is
132
+ not in the catalog is passed to the CLI for validation instead of being rejected by
133
+ a bundled list. A model available in a desktop app may still be absent from the
134
+ installed CLI's catalog or account access.
135
+
136
+ Native connections are read-only by default. Codex starts with a read-only sandbox
137
+ and asks before operations outside it; approval requests appear in Thwip with the
138
+ command or file list and default to denial. Claude Code and the Antigravity CLI run
139
+ in their non-interactive print modes, where tools that would need an approval are
140
+ declined by the CLI itself. Each turn sends the portable text conversation to a
141
+ fresh native session, so native reasoning and tool state do not carry between turns.
142
+ Ctrl+C interrupts the current turn and stops the child process; the unanswered
143
+ message is removed so it can be re-sent or handed to another provider.
144
+
145
+ Native billing and usage limits are managed by each CLI account. Thwip records the
146
+ token counts the CLIs report but does not estimate cost for them. Codex and Claude
147
+ Code also report their account usage windows (5 hour and 7 day) after each response;
148
+ `/limits` and `/status` show the used percentage and reset time. When a CLI reports
149
+ an exhausted usage limit, the standard failover prompt offers the other connected
150
+ providers.
151
+
152
+ `/native codex` remains available as a launcher: it saves the Thwip session and
153
+ replaces Thwip with the Codex CLI itself in the selected project. Conversation
154
+ history is not transferred by the launcher; use `/session load` after restarting
155
+ Thwip to resume.
115
156
 
116
157
  ## Auditable handoffs
117
158
 
@@ -150,11 +191,11 @@ See [research and prior art](docs/handoff-research.md) for the differentiation r
150
191
 
151
192
  | Company | Agent | Capabilities |
152
193
  |:---|:---|:---|
153
- | Anthropic | Claude API (Fable 5, Opus 5, Sonnet 5, Haiku 4.5) | Chat, File Edit, Code Run, Terminal, Git |
154
- | Google | Gemini API (3.1 Pro Preview, 3.7 Flash, 3.5 Flash-Lite) | Chat, File Edit, Code Run, Terminal, Git |
155
- | OpenAI | OpenAI API (GPT-5.6 Sol, Terra, Luna) | Chat, File Edit, Code Run, Terminal, Git |
194
+ | Anthropic | Claude Code sign-in, or Claude API (Fable 5, Opus 5, Sonnet 5, Haiku 4.5) | Chat, File Edit, Code Run, Terminal, Git |
195
+ | Google | Antigravity or Gemini CLI sign-in, or Gemini API (3.1 Pro Preview, 3.7 Flash, 3.5 Flash-Lite) | Chat, File Edit, Code Run, Terminal, Git |
196
+ | OpenAI | Codex CLI sign-in, or OpenAI API (GPT-5.6 Sol, Terra, Luna) | Chat, File Edit, Code Run, Terminal, Git |
156
197
  | DeepSeek | DeepSeek V3 / R1 Reasoner | Chat, File Edit, Code Run, Reasoning |
157
- | Groq | Llama 3.3 70B, Mixtral | Chat, File Edit, Code Run |
198
+ | Groq | GPT-OSS 120B (default); Llama 3.3 for eligible enterprise accounts only | Chat, File Edit, Code Run |
158
199
  | Ollama | Local Models (Llama 3.3, Qwen Coder, DeepSeek R1) | Chat, File Edit, Code Run (Local, Offline) |
159
200
  | OpenRouter | Multi-Company Models | Gateway Routing |
160
201
 
@@ -164,7 +205,7 @@ See [research and prior art](docs/handoff-research.md) for the differentiation r
164
205
 
165
206
  thwip auto-detects existing API keys from environment variables and existing agent configs (`~/.claude.json`, `~/.gemini/config.json`). Configuration can also be set manually:
166
207
 
167
- Installed apps, CLI sign-ins, and API access are separate. thwip can report a detected Claude, Gemini, or Codex CLI login, but the current SDK adapters require a provider API key. A ChatGPT, Claude, or Google subscription does not automatically provide a reusable third-party API key. Ollama needs no key when its local server is running.
208
+ Installed CLI sign-ins and API access are separate paths. A signed-in Codex, Claude Code, or Antigravity CLI is used directly through its own protocol (see above). The direct SDK adapters for Anthropic, Google, OpenAI, DeepSeek, Groq, and OpenRouter require a provider API key. Ollama needs no key when its local server is running.
168
209
 
169
210
  ```toml
170
211
  [defaults]
@@ -9,6 +9,7 @@
9
9
  ## Features
10
10
 
11
11
  - **Auto-Detection**: Discovers installed AI coding agents (Claude Code, Antigravity, Gemini CLI, OpenAI/Codex, Aider, Copilot, Cursor, Windsurf, Cline, Ollama) and configured credentials.
12
+ - **Existing Sign-ins**: Chats through installed Codex, Claude Code, and Antigravity CLIs using their own logins and live model lists, with no API key required.
12
13
  - **Context Portability**: Switch providers with stored conversational text. Working files remain in the selected local project; they are not automatically uploaded.
13
14
  - **Handoff Preview**: Inspect text continuity, omitted state, capability changes, approximate context pressure, and a text fingerprint before switching. Runs locally without model calls.
14
15
  - **Dynamic UI**: Terminal interface adapts its status bar, capabilities, and theme based on the active provider.
@@ -44,6 +45,8 @@ Start the interactive terminal in your current project directory:
44
45
 
45
46
  ```bash
46
47
  thwip
48
+ thwip --project ~/code/my-app # open a specific project
49
+ thwip --version
47
50
  ```
48
51
 
49
52
  ---
@@ -54,12 +57,13 @@ thwip
54
57
  |:---|:---|
55
58
  | `/switch [agent] [model]` | Switch active agent or model mid-conversation |
56
59
  | `/handoff [agent] [model]` | Preview a target locally without switching or sending data |
60
+ | `/native codex` | Save and leave Thwip for the installed Codex CLI itself (launcher, no context transfer) |
57
61
  | `/agents` | Show all detected coding agents, company status, and capabilities |
58
62
  | `/models [agent]` | List available models for current or target agent |
59
63
  | `/key [provider]` | Enter an API key securely without placing it in prompt history |
60
64
  | `/status` | Display current session, project, and token stats |
61
65
  | `/limits` | View token usage, quota, and spend metrics |
62
- | `/detect` | Re-scan system for newly installed coding agents |
66
+ | `/detect` | Re-scan installed agents and reconnect CLI sign-ins |
63
67
  | `/session save [name]` | Save current chat session |
64
68
  | `/session load <name>` | Load a previously saved session |
65
69
  | `/session list` | List all saved sessions |
@@ -68,13 +72,50 @@ thwip
68
72
  | `/cost` | Show estimated session and cumulative cost |
69
73
  | `/project [path]` | View or change project working directory |
70
74
  | `Ctrl + S` | Quick switch agent prompt |
71
-
72
- Short aliases are available for frequent commands: `/a`, `/m`, `/s`, `/k`, `/g`, and `/t`.
73
75
  | `Ctrl + T` | Show agent status |
74
76
  | `Ctrl + H` | View history |
75
77
  | `/quit` | Exit thwip |
76
78
 
77
- ---
79
+ Short aliases are available for frequent commands: `/a`, `/m`, `/s`, `/sw`, `/k`, `/g`, and `/t`.
80
+
81
+ ### Existing CLI sign-ins (no API key needed)
82
+
83
+ If Codex, Claude Code, or the Antigravity CLI is installed and signed in, Thwip
84
+ connects to it at startup and uses that sign-in for chat. No API key is copied or
85
+ required, and Thwip never reads the CLI's stored credentials. A configured direct
86
+ API key for the same provider always takes precedence over the native connection.
87
+
88
+ | Provider | Installed CLI | Transport | Model list |
89
+ |:---|:---|:---|:---|
90
+ | OpenAI | `codex` | Codex App Server (JSON-RPC over stdio) | Live from `model/list` |
91
+ | Anthropic | `claude` | Claude Code print mode (`stream-json`) | Aliases `fable`, `opus`, `sonnet`, `haiku`; explicit IDs pass through |
92
+ | Google | `agy` (Antigravity CLI) or `gemini` | Antigravity print mode (`stream-json`) or Gemini ACP | Live from `agy models` or the ACP session |
93
+
94
+ `/models` refreshes the catalog supplied by each CLI. An explicit model ID that is
95
+ not in the catalog is passed to the CLI for validation instead of being rejected by
96
+ a bundled list. A model available in a desktop app may still be absent from the
97
+ installed CLI's catalog or account access.
98
+
99
+ Native connections are read-only by default. Codex starts with a read-only sandbox
100
+ and asks before operations outside it; approval requests appear in Thwip with the
101
+ command or file list and default to denial. Claude Code and the Antigravity CLI run
102
+ in their non-interactive print modes, where tools that would need an approval are
103
+ declined by the CLI itself. Each turn sends the portable text conversation to a
104
+ fresh native session, so native reasoning and tool state do not carry between turns.
105
+ Ctrl+C interrupts the current turn and stops the child process; the unanswered
106
+ message is removed so it can be re-sent or handed to another provider.
107
+
108
+ Native billing and usage limits are managed by each CLI account. Thwip records the
109
+ token counts the CLIs report but does not estimate cost for them. Codex and Claude
110
+ Code also report their account usage windows (5 hour and 7 day) after each response;
111
+ `/limits` and `/status` show the used percentage and reset time. When a CLI reports
112
+ an exhausted usage limit, the standard failover prompt offers the other connected
113
+ providers.
114
+
115
+ `/native codex` remains available as a launcher: it saves the Thwip session and
116
+ replaces Thwip with the Codex CLI itself in the selected project. Conversation
117
+ history is not transferred by the launcher; use `/session load` after restarting
118
+ Thwip to resume.
78
119
 
79
120
  ## Auditable handoffs
80
121
 
@@ -113,11 +154,11 @@ See [research and prior art](docs/handoff-research.md) for the differentiation r
113
154
 
114
155
  | Company | Agent | Capabilities |
115
156
  |:---|:---|:---|
116
- | Anthropic | Claude API (Fable 5, Opus 5, Sonnet 5, Haiku 4.5) | Chat, File Edit, Code Run, Terminal, Git |
117
- | Google | Gemini API (3.1 Pro Preview, 3.7 Flash, 3.5 Flash-Lite) | Chat, File Edit, Code Run, Terminal, Git |
118
- | OpenAI | OpenAI API (GPT-5.6 Sol, Terra, Luna) | Chat, File Edit, Code Run, Terminal, Git |
157
+ | Anthropic | Claude Code sign-in, or Claude API (Fable 5, Opus 5, Sonnet 5, Haiku 4.5) | Chat, File Edit, Code Run, Terminal, Git |
158
+ | Google | Antigravity or Gemini CLI sign-in, or Gemini API (3.1 Pro Preview, 3.7 Flash, 3.5 Flash-Lite) | Chat, File Edit, Code Run, Terminal, Git |
159
+ | OpenAI | Codex CLI sign-in, or OpenAI API (GPT-5.6 Sol, Terra, Luna) | Chat, File Edit, Code Run, Terminal, Git |
119
160
  | DeepSeek | DeepSeek V3 / R1 Reasoner | Chat, File Edit, Code Run, Reasoning |
120
- | Groq | Llama 3.3 70B, Mixtral | Chat, File Edit, Code Run |
161
+ | Groq | GPT-OSS 120B (default); Llama 3.3 for eligible enterprise accounts only | Chat, File Edit, Code Run |
121
162
  | Ollama | Local Models (Llama 3.3, Qwen Coder, DeepSeek R1) | Chat, File Edit, Code Run (Local, Offline) |
122
163
  | OpenRouter | Multi-Company Models | Gateway Routing |
123
164
 
@@ -127,7 +168,7 @@ See [research and prior art](docs/handoff-research.md) for the differentiation r
127
168
 
128
169
  thwip auto-detects existing API keys from environment variables and existing agent configs (`~/.claude.json`, `~/.gemini/config.json`). Configuration can also be set manually:
129
170
 
130
- Installed apps, CLI sign-ins, and API access are separate. thwip can report a detected Claude, Gemini, or Codex CLI login, but the current SDK adapters require a provider API key. A ChatGPT, Claude, or Google subscription does not automatically provide a reusable third-party API key. Ollama needs no key when its local server is running.
171
+ Installed CLI sign-ins and API access are separate paths. A signed-in Codex, Claude Code, or Antigravity CLI is used directly through its own protocol (see above). The direct SDK adapters for Anthropic, Google, OpenAI, DeepSeek, Groq, and OpenRouter require a provider API key. Ollama needs no key when its local server is running.
131
172
 
132
173
  ```toml
133
174
  [defaults]
@@ -0,0 +1,58 @@
1
+ # Verification status
2
+
3
+ Verified locally on 2026-09-24 for the unreleased v1.4.0 source. The Python suite
4
+ passes 216 offline tests. Exhaustive behavior across every provider and
5
+ configuration has not been established.
6
+
7
+ ## Native CLI connections (2026-09-24)
8
+
9
+ Live checks were run on macOS through a pseudo-terminal driving the real REPL with
10
+ the installed Codex CLI 0.152.1, Claude Code 2.1.281, and Antigravity CLI 1.2.8,
11
+ each using its existing sign-in. No API keys were configured.
12
+
13
+ | Flow | Result |
14
+ | --- | --- |
15
+ | Startup discovery | All three CLIs connected; live model lists shown (Codex 4 models, Antigravity 14, Claude aliases) |
16
+ | Chat turn per provider | Claude Code, Codex, and Antigravity each answered; text streamed into the Live view |
17
+ | Context across `/switch` | Codex and Antigravity both recalled the answer given by the previous provider |
18
+ | Codex approval request | A write command outside the read-only sandbox produced a permission prompt; denial left the workspace unchanged |
19
+ | Ctrl+C during a response | Turn cancelled, child process terminated, REPL continued, unanswered message removed |
20
+ | `/session save` and `/session load` | Session with a native provider saved and reloaded in a fresh run |
21
+ | `/models`, `/models <provider>`, `/models <tier>` | Live catalogs listed with `CLI account` in place of API pricing |
22
+ | `thwip --version`, `--help`, `--project` | Handled without starting the REPL; invalid project exits with code 2 |
23
+ | `/limits` and `/status` usage windows | Codex and Claude Code account windows (5h, 7d) displayed with reset times after live turns |
24
+ | Leftover processes | None after each run |
25
+
26
+ The defect that blocked the previous attempt was the Codex sandbox value: the
27
+ adapter sent `readOnly` and Codex rejected `thread/start` with an invalid-request
28
+ error. The protocol enum is `read-only`. A regression test now checks the request.
29
+
30
+ Remaining limitations: the Antigravity CLI showed intermittent network resets to
31
+ Google's backend during testing, which surface as turn errors; Claude Code's model
32
+ aliases are a curated list because the CLI exposes no model listing; native usage
33
+ limit failover is unit-tested from error text, not observed live; the real Gemini
34
+ CLI ACP path is covered by mocked tests only because it is not installed here.
35
+
36
+ | Area | Evidence | Remaining limitation |
37
+ | --- | --- | --- |
38
+ | Commands and aliases | Offline command smoke tests, invalid input cases | Most smoke tests check exceptions, not all rendered content |
39
+ | Sessions | Save/load, separate fresh conversations, malformed metadata, permissions, project rebinding | Concurrent writes to the same explicitly named session are not coordinated |
40
+ | Provider switching and handoff | All seven provider catalogs, portable history, bounded failover | Live account quotas and model availability unverified |
41
+ | Tool execution | Real temporary-file operations, path containment, process timeout/cancellation, invalid arguments | Shell and code tools retain local user privileges; output capture memory is unbounded |
42
+ | Native Codex launcher | Save-before-launch, consent, flags, missing binary, terminal checks, failures | Mocked process replacement; no conversation transfer |
43
+ | Native CLI connections | Live REPL runs above; mocked protocol tests for approvals, failures, limits, prompt building, discovery parsing | Gemini ACP path mocked only; live limit failover not observed |
44
+ | Provider responses | Mocked native tool continuations; DeepSeek/Groq/OpenRouter streaming, usage-only chunks, 429/500 errors and serialized tool arguments | Other streaming and error branches still have coverage gaps |
45
+ | Display/config/auth | Configuration validation, display settings, credential boundaries, short-key masking | No full terminal/platform matrix |
46
+ | Usage | Atomic writes, malformed records, valid totals | Unknown catalog pricing may appear as zero estimated cost |
47
+ | Website | Production build and deterministic demo completion/replay test | Real browser, layout, clipboard, keyboard and accessibility checks remain incomplete |
48
+ | Dependencies | npm audit: zero advisories; Python installed-dependency audit: none found | Python audit skipped legacy local `thwip 1.0.0` metadata |
49
+ | Packaging | Wheel/source build and metadata checks | Publication is separately verified through the release workflow |
50
+
51
+ The Python suite measured 69% statement coverage with 160 passing tests before
52
+ the final credential-masking tests were added. Coverage is diagnostic evidence,
53
+ not proof that every feature works. The website test uses a minimal DOM stand-in
54
+ and controlled timers, not a browser.
55
+
56
+ This pass fixed default-session save collisions, malformed session/message
57
+ metadata acceptance, tool-argument display crashes, Rich markup interpretation
58
+ in action output, and short-key masking leakage.
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "thwip-cli"
7
- version = "1.2.0"
7
+ version = "1.4.0"
8
8
  description = "Universal coding agent multiplexer: detect, switch, and route between AI coding agents seamlessly"
9
9
  readme = "README.md"
10
10
  license = "MIT"
@@ -0,0 +1,205 @@
1
+ import json
2
+ from io import StringIO
3
+ from types import SimpleNamespace
4
+
5
+ import pytest
6
+ from rich.console import Console
7
+
8
+ from thwip.agents import AgentRegistry
9
+ from thwip.agents.base import AgentDone, Capability, LimitHit, LimitStatus, TextDelta
10
+ from thwip.cli import ThwipCLI
11
+ from thwip.config import ThwipConfig, get_usage_path
12
+ from thwip.limits import UsageTracker
13
+ from thwip.session import Session
14
+ from thwip.tools import ToolManager
15
+
16
+
17
+ @pytest.fixture
18
+ def cli(tmp_path, monkeypatch):
19
+ monkeypatch.setenv('THWIP_CONFIG_DIR', str(tmp_path / 'config'))
20
+ cli = ThwipCLI.__new__(ThwipCLI)
21
+ cli.config = ThwipConfig(project=str(tmp_path), auto_save=False)
22
+ cli.registry = AgentRegistry(cli.config)
23
+ async def no_native_discovery(project):
24
+ return None
25
+ monkeypatch.setattr(cli.registry, 'connect_native_agents', no_native_discovery)
26
+ for a in cli.registry.list_agents():
27
+ monkeypatch.setattr(a, 'is_installed', lambda: True)
28
+ monkeypatch.setattr(a, 'is_configured', lambda: False)
29
+ if a.name == 'ollama':
30
+ a._cached_models = a.get_handoff_models()
31
+ cli.current_agent = cli.registry.get_agent('claude')
32
+ cli.session = Session(project_path=str(tmp_path), current_agent='claude', current_model=cli.current_agent.get_default_model())
33
+ cli.usage_tracker = UsageTracker()
34
+ cli.tool_manager = ToolManager(str(tmp_path))
35
+ cli.detector = SimpleNamespace(scan_all=list)
36
+ monkeypatch.setattr('builtins.input', lambda *args: '')
37
+ monkeypatch.setattr('getpass.getpass', lambda *args: '')
38
+ console = Console(file=StringIO(), width=100, color_system=None)
39
+ monkeypatch.setattr('thwip.cli.console', console)
40
+ monkeypatch.setattr('thwip.theme.console', console)
41
+ return cli
42
+
43
+
44
+ @pytest.mark.parametrize('command', [
45
+ '/help', '/h', '/about', '/guide', '/g', '/info', '/agents', '/a', '/list',
46
+ '/models', '/m', '/models flagship', '/models balanced', '/models fast', '/models google',
47
+ '/tools', '/t', '/status', '/limits', '/detect', '/history', '/clear', '/reset', '/cost',
48
+ '/project', '/session save audit', '/session load missing', '/session list', '/session clear',
49
+ '/key', '/key google', '/key invalid', '/key openai REJECTED_TEST_KEY',
50
+ '/handoff', '/handoff invalid', '/handoff google invalid', '/switch', '/switch invalid',
51
+ '/switch google invalid', '/unknown', '/q', '/exit', '/quit',
52
+ ])
53
+ @pytest.mark.asyncio
54
+ async def test_command_no_exception(cli, command):
55
+ await cli.handle_command(command)
56
+
57
+
58
+ @pytest.mark.parametrize('provider', ['claude','google','openai','deepseek','groq','ollama','openrouter'])
59
+ @pytest.mark.asyncio
60
+ async def test_all_catalogued_handoffs(cli, provider):
61
+ before = list(cli.session.messages)
62
+ for model in cli.registry.get_agent(provider).get_handoff_models():
63
+ await cli.handle_command(f'/handoff {provider} {model.id}')
64
+ assert cli.session.messages == before
65
+
66
+
67
+ @pytest.mark.asyncio
68
+ async def test_project_path_with_spaces(cli, tmp_path):
69
+ project = tmp_path / 'project with spaces'
70
+ project.mkdir()
71
+ await cli.handle_command(f'/project {project}')
72
+ assert cli.session.project_path == str(project)
73
+
74
+
75
+ def test_usage_file_private(cli):
76
+ cli.usage_tracker.record_usage('openai', 'gpt-5.6-terra', 10, 5)
77
+ assert get_usage_path().stat().st_mode & 0o777 == 0o600
78
+
79
+
80
+ @pytest.mark.parametrize('tool,args', [
81
+ ('read_file', {'file_path': 12}),
82
+ ('list_files', {'sub_dir': ['bad']}),
83
+ ('run_python', {'code': None}),
84
+ ])
85
+ def test_invalid_tool_args_return_error_not_exception(cli, tool, args):
86
+ result = cli.tool_manager.execute_tool(tool, args)
87
+ assert isinstance(result, str)
88
+
89
+
90
+ @pytest.mark.asyncio
91
+ async def test_history_displays_literal_markup(cli):
92
+ cli.session.add_user_message('Show [/not-a-tag] literally')
93
+ await cli.handle_command('/history')
94
+
95
+
96
+ @pytest.mark.parametrize('value', ['wrong', None, -1, True])
97
+ def test_corrupted_handoff_counter_rejected_or_sanitized(cli, value):
98
+ path = cli.session.save('damaged')
99
+ data = json.loads(path.read_text())
100
+ data['observed_tool_results'] = value
101
+ path.write_text(json.dumps(data))
102
+ loaded = Session.load('damaged')
103
+ assert loaded is None or (type(loaded.observed_tool_results) is int and loaded.observed_tool_results >= 0)
104
+
105
+
106
+ def test_invalid_config_type_rejected_or_defaulted(cli):
107
+ cli.config._apply_toml({'defaults': {'project': 123, 'confirm_tools': 'false'}})
108
+ assert isinstance(cli.config.project, str)
109
+ assert type(cli.config.confirm_tools) is bool
110
+
111
+
112
+ @pytest.mark.asyncio
113
+ async def test_session_roundtrip_rebinds_tools(cli, tmp_path):
114
+ target = tmp_path / 'other'
115
+ target.mkdir()
116
+ saved = Session(project_path=str(target), current_agent='google', current_model='missing')
117
+ saved.save('loadable')
118
+ await cli.handle_command('/session load loadable')
119
+ assert cli.tool_manager.file_editor.project_path == target
120
+ assert cli.current_agent.name == 'google'
121
+ assert cli.session.current_model == cli.current_agent.get_default_model()
122
+
123
+
124
+ def test_all_tools_and_safety(cli, tmp_path):
125
+ tm = cli.tool_manager
126
+ assert 'Successfully' in tm.execute_tool('write_file', {'file_path':'a.txt','content':'one'})
127
+ assert tm.execute_tool('read_file', {'file_path':'a.txt'}) == 'one'
128
+ assert 'Successfully' in tm.execute_tool('edit_file', {'file_path':'a.txt','old_str':'one','new_str':'two'})
129
+ assert 'a.txt' in tm.execute_tool('list_files', {})
130
+ assert '42' in tm.execute_tool('run_python', {'code':'print(6 * 7)'})
131
+ assert 'Exit code: 0' in tm.execute_tool('run_command', {'command':'printf audit-ok'})
132
+ assert isinstance(tm.execute_tool('git_status', {}), str)
133
+ assert isinstance(tm.execute_tool('git_diff', {}), str)
134
+ assert 'outside' in tm.execute_tool('read_file', {'file_path':'../outside'})
135
+ outside = tmp_path.parent / 'audit-outside-file'
136
+ outside.write_text('not accessible')
137
+ (tmp_path / 'link').symlink_to(outside)
138
+ assert 'outside' in tm.execute_tool('read_file', {'file_path':'link'})
139
+ assert 'Error' in tm.execute_tool('unknown', {})
140
+ assert {t['name'] for t in tm.get_anthropic_tools()} == {t['function']['name'] for t in tm.get_openai_tools()}
141
+
142
+
143
+ @pytest.mark.asyncio
144
+ async def test_failover_does_not_retry_already_failed_provider(cli, monkeypatch):
145
+ agents = [cli.registry.get_agent('claude'), cli.registry.get_agent('google')]
146
+ attempts = []
147
+ for agent in agents:
148
+ monkeypatch.setattr(agent, 'is_configured', lambda: True)
149
+ monkeypatch.setattr(agent, 'get_capabilities_for_model', lambda model: {Capability.CHAT})
150
+
151
+ def make_chat(name):
152
+ async def chat(**kwargs):
153
+ attempts.append(name)
154
+ # Stop safely after four simulated rate limits; no actual network calls.
155
+ if len(attempts) <= 4:
156
+ yield LimitHit(error_type=LimitStatus.RATE_LIMITED, message='test quota')
157
+ else:
158
+ yield TextDelta(content='stop test')
159
+ yield AgentDone()
160
+ return chat
161
+
162
+ monkeypatch.setattr(agent, 'chat', make_chat(agent.name))
163
+ monkeypatch.setattr(cli.registry, 'get_ready_agents', lambda: agents)
164
+ cli.config.fallback.chain = ['claude', 'google']
165
+ cli.config.limits.auto_switch = True
166
+ await cli.process_user_message('hello')
167
+ assert len(attempts) <= len(agents), f'Repeated failed providers: {attempts}'
168
+
169
+
170
+ def test_main_handles_version_and_help_without_starting_the_repl(capsys):
171
+ from thwip import __version__
172
+ from thwip import cli as cli_module
173
+
174
+ with pytest.raises(SystemExit) as exit_info:
175
+ cli_module.main(['--version'])
176
+ assert exit_info.value.code == 0 and f'thwip {__version__}' in capsys.readouterr().out
177
+ with pytest.raises(SystemExit) as exit_info:
178
+ cli_module.main(['--help'])
179
+ assert exit_info.value.code == 0 and '--project' in capsys.readouterr().out
180
+ with pytest.raises(SystemExit) as exit_info:
181
+ cli_module.main(['--project', '/definitely/missing/dir'])
182
+ assert exit_info.value.code == 2
183
+
184
+
185
+ @pytest.mark.asyncio
186
+ async def test_failed_turn_removes_unanswered_user_message(cli, monkeypatch):
187
+ class Broken:
188
+ name = 'openai'
189
+ display_name = 'Broken'
190
+ company = 'OpenAI'
191
+ native_tools = True
192
+ project = '.'
193
+ def is_configured(self):
194
+ return True
195
+ def is_installed(self):
196
+ return True
197
+ def get_capabilities_for_model(self, model):
198
+ return set()
199
+ async def chat(self, **kwargs):
200
+ raise RuntimeError('Native CLI rejected thread/start')
201
+ yield
202
+ cli.current_agent = Broken()
203
+ cli.session.current_agent = 'openai'
204
+ await cli.process_user_message('hello')
205
+ assert cli.session.messages == []
@@ -75,6 +75,30 @@ class UnconfiguredAgent(ToolCallingAgent):
75
75
  return False
76
76
 
77
77
 
78
+ @pytest.mark.asyncio
79
+ @pytest.mark.parametrize("arguments", [None, [], "bad", {"file_path": "[broken]"}])
80
+ async def test_invalid_tool_arguments_return_errors_without_crashing(tmp_path, arguments):
81
+ class InvalidToolAgent(ToolCallingAgent):
82
+ async def chat(self, messages, **kwargs):
83
+ self.calls.append(messages)
84
+ if len(self.calls) == 1:
85
+ yield ToolUseStart(tool_id="bad", tool_name="read_file", args=arguments)
86
+ yield AgentDone()
87
+ else:
88
+ assert "Error" in messages[-1]["content"]
89
+ yield TextDelta(content="Recovered from invalid tool arguments.")
90
+ yield AgentDone()
91
+
92
+ cli = ThwipCLI.__new__(ThwipCLI)
93
+ cli.config = SimpleNamespace(stream=True, confirm_tools=True)
94
+ cli.session = Session(current_agent="fake", current_model="fake-model")
95
+ cli.current_agent = InvalidToolAgent()
96
+ cli.tool_manager = ToolManager(tmp_path)
97
+ cli.usage_tracker = FakeUsageTracker()
98
+ await cli.process_user_message("Read")
99
+ assert cli.session.messages[-1].content == "Recovered from invalid tool arguments."
100
+
101
+
78
102
  @pytest.mark.asyncio
79
103
  async def test_unconfigured_agent_shows_setup_guidance(tmp_path):
80
104
  cli = ThwipCLI.__new__(ThwipCLI)
@@ -91,11 +115,12 @@ async def test_unconfigured_agent_shows_setup_guidance(tmp_path):
91
115
  assert len(cli.session.messages) == 0
92
116
 
93
117
 
94
- def test_inline_api_key_is_rejected():
118
+ @pytest.mark.asyncio
119
+ async def test_inline_api_key_is_rejected():
95
120
  cli = ThwipCLI.__new__(ThwipCLI)
96
121
  cli.config = SimpleNamespace(keys={}, key_sources={}, save=lambda: None)
97
122
 
98
- cli.cmd_auth_config("openai", "secret-value")
123
+ await cli.cmd_auth_config("openai", "secret-value")
99
124
 
100
125
  assert cli.config.keys == {}
101
126
  assert cli.config.key_sources == {}
@@ -0,0 +1,75 @@
1
+ """Exercise streaming completion and provider error boundaries without network calls."""
2
+
3
+ import json
4
+ from types import SimpleNamespace
5
+
6
+ import httpx
7
+ import openai
8
+ import pytest
9
+
10
+ from thwip.agents.base import AgentDone, LimitHit, LimitStatus, TextDelta
11
+ from thwip.agents.deepseek_agent import DeepSeekAgent
12
+ from thwip.agents.groq_agent import GroqAgent
13
+ from thwip.agents.openrouter_agent import OpenRouterAgent
14
+
15
+
16
+ @pytest.mark.parametrize("agent_type", [DeepSeekAgent, GroqAgent, OpenRouterAgent])
17
+ @pytest.mark.asyncio
18
+ async def test_tool_continuation_serializes_arguments_without_mutating_history(agent_type):
19
+ original = {"role": "assistant", "content": "", "tool_calls": [{
20
+ "id": "call1", "type": "function", "function": {
21
+ "name": "read_file", "arguments": {"file_path": "note.txt"},
22
+ },
23
+ }], "_native_state": {"other": "private"}}
24
+
25
+ async def create(**kwargs):
26
+ sent = kwargs["messages"][0]
27
+ assert "_native_state" not in sent
28
+ assert json.loads(sent["tool_calls"][0]["function"]["arguments"]) == {"file_path": "note.txt"}
29
+ return SimpleNamespace(choices=[SimpleNamespace(message=SimpleNamespace(content="done", tool_calls=[]))], usage=None)
30
+
31
+ agent = agent_type(api_key="test")
32
+ agent._client = SimpleNamespace(chat=SimpleNamespace(completions=SimpleNamespace(create=create)))
33
+ events = [event async for event in agent.chat([original], stream=False)]
34
+ assert any(isinstance(event, AgentDone) for event in events)
35
+ assert isinstance(original["tool_calls"][0]["function"]["arguments"], dict)
36
+ assert "_native_state" in original
37
+
38
+
39
+ @pytest.mark.parametrize("agent_type", [DeepSeekAgent, GroqAgent, OpenRouterAgent])
40
+ @pytest.mark.asyncio
41
+ async def test_stream_text_and_usage_only_final_chunk(agent_type):
42
+ async def chunks():
43
+ yield SimpleNamespace(choices=[SimpleNamespace(delta=SimpleNamespace(
44
+ content="hello", tool_calls=None))], usage=None)
45
+ yield SimpleNamespace(choices=[], usage=SimpleNamespace(prompt_tokens=9, completion_tokens=3))
46
+
47
+ async def create(**kwargs):
48
+ assert kwargs["stream"] is True
49
+ return chunks()
50
+
51
+ agent = agent_type(api_key="test")
52
+ agent._client = SimpleNamespace(chat=SimpleNamespace(completions=SimpleNamespace(create=create)))
53
+ events = [event async for event in agent.chat([{"role": "user", "content": "hi"}])]
54
+ assert isinstance(events[0], TextDelta) and events[0].content == "hello"
55
+ assert isinstance(events[-1], AgentDone)
56
+ assert events[-1].usage.input_tokens == 9
57
+ assert events[-1].usage.output_tokens == 3
58
+ assert agent.check_limits() == LimitStatus.OK
59
+
60
+
61
+ @pytest.mark.parametrize("agent_type", [DeepSeekAgent, GroqAgent, OpenRouterAgent])
62
+ @pytest.mark.parametrize("status", [429, 500])
63
+ @pytest.mark.asyncio
64
+ async def test_provider_errors_emit_failure_not_success(agent_type, status):
65
+ response = httpx.Response(status, request=httpx.Request("POST", "https://example.invalid"))
66
+ error_type = openai.RateLimitError if status == 429 else openai.APIStatusError
67
+
68
+ async def create(**kwargs):
69
+ raise error_type("simulated failure", response=response, body=None)
70
+
71
+ agent = agent_type(api_key="test")
72
+ agent._client = SimpleNamespace(chat=SimpleNamespace(completions=SimpleNamespace(create=create)))
73
+ events = [event async for event in agent.chat([])]
74
+ assert len(events) == 1 and isinstance(events[0], LimitHit)
75
+ assert events[0].error_type == (LimitStatus.RATE_LIMITED if status == 429 else LimitStatus.UNKNOWN)