thwip-cli 1.2.0__tar.gz → 1.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. {thwip_cli-1.2.0 → thwip_cli-1.3.0}/PKG-INFO +19 -4
  2. {thwip_cli-1.2.0 → thwip_cli-1.3.0}/README.md +18 -3
  3. thwip_cli-1.3.0/docs/verification.md +28 -0
  4. {thwip_cli-1.2.0 → thwip_cli-1.3.0}/pyproject.toml +1 -1
  5. thwip_cli-1.3.0/tests/test_audit_regressions.py +164 -0
  6. {thwip_cli-1.2.0 → thwip_cli-1.3.0}/tests/test_cli.py +24 -0
  7. thwip_cli-1.3.0/tests/test_compatible_streaming.py +75 -0
  8. thwip_cli-1.3.0/tests/test_native_launcher.py +69 -0
  9. thwip_cli-1.3.0/tests/test_repair_verification.py +177 -0
  10. {thwip_cli-1.2.0 → thwip_cli-1.3.0}/tests/test_session.py +41 -0
  11. thwip_cli-1.3.0/tests/test_utils.py +17 -0
  12. {thwip_cli-1.2.0 → thwip_cli-1.3.0}/thwip/__init__.py +1 -1
  13. {thwip_cli-1.2.0 → thwip_cli-1.3.0}/thwip/agents/base.py +1 -0
  14. thwip_cli-1.3.0/thwip/agents/chat_messages.py +20 -0
  15. {thwip_cli-1.2.0 → thwip_cli-1.3.0}/thwip/agents/claude_agent.py +6 -1
  16. {thwip_cli-1.2.0 → thwip_cli-1.3.0}/thwip/agents/deepseek_agent.py +2 -1
  17. {thwip_cli-1.2.0 → thwip_cli-1.3.0}/thwip/agents/google_agent.py +5 -1
  18. {thwip_cli-1.2.0 → thwip_cli-1.3.0}/thwip/agents/groq_agent.py +5 -3
  19. {thwip_cli-1.2.0 → thwip_cli-1.3.0}/thwip/agents/openai_agent.py +8 -1
  20. {thwip_cli-1.2.0 → thwip_cli-1.3.0}/thwip/agents/openrouter_agent.py +2 -1
  21. {thwip_cli-1.2.0 → thwip_cli-1.3.0}/thwip/cli.py +135 -38
  22. {thwip_cli-1.2.0 → thwip_cli-1.3.0}/thwip/config.py +27 -0
  23. {thwip_cli-1.2.0 → thwip_cli-1.3.0}/thwip/detector.py +16 -10
  24. {thwip_cli-1.2.0 → thwip_cli-1.3.0}/thwip/limits.py +33 -4
  25. {thwip_cli-1.2.0 → thwip_cli-1.3.0}/thwip/session.py +29 -1
  26. {thwip_cli-1.2.0 → thwip_cli-1.3.0}/thwip/shortcuts.py +1 -0
  27. {thwip_cli-1.2.0 → thwip_cli-1.3.0}/thwip/theme.py +17 -11
  28. {thwip_cli-1.2.0 → thwip_cli-1.3.0}/thwip/tools/__init__.py +22 -1
  29. {thwip_cli-1.2.0 → thwip_cli-1.3.0}/thwip/tools/code_runner.py +4 -6
  30. {thwip_cli-1.2.0 → thwip_cli-1.3.0}/thwip/tools/file_editor.py +1 -1
  31. {thwip_cli-1.2.0 → thwip_cli-1.3.0}/thwip/tools/terminal.py +35 -5
  32. {thwip_cli-1.2.0 → thwip_cli-1.3.0}/thwip/utils.py +1 -1
  33. {thwip_cli-1.2.0 → thwip_cli-1.3.0}/uv.lock +1 -1
  34. {thwip_cli-1.2.0 → thwip_cli-1.3.0}/.github/workflows/publish.yml +0 -0
  35. {thwip_cli-1.2.0 → thwip_cli-1.3.0}/.gitignore +0 -0
  36. {thwip_cli-1.2.0 → thwip_cli-1.3.0}/LICENSE +0 -0
  37. {thwip_cli-1.2.0 → thwip_cli-1.3.0}/docs/handoff-research.md +0 -0
  38. {thwip_cli-1.2.0 → thwip_cli-1.3.0}/install.sh +0 -0
  39. {thwip_cli-1.2.0 → thwip_cli-1.3.0}/tests/test_agents.py +0 -0
  40. {thwip_cli-1.2.0 → thwip_cli-1.3.0}/tests/test_config.py +0 -0
  41. {thwip_cli-1.2.0 → thwip_cli-1.3.0}/tests/test_detector.py +0 -0
  42. {thwip_cli-1.2.0 → thwip_cli-1.3.0}/tests/test_handoff.py +0 -0
  43. {thwip_cli-1.2.0 → thwip_cli-1.3.0}/tests/test_tools.py +0 -0
  44. {thwip_cli-1.2.0 → thwip_cli-1.3.0}/thwip/__main__.py +0 -0
  45. {thwip_cli-1.2.0 → thwip_cli-1.3.0}/thwip/agents/__init__.py +0 -0
  46. {thwip_cli-1.2.0 → thwip_cli-1.3.0}/thwip/agents/ollama_agent.py +0 -0
  47. {thwip_cli-1.2.0 → thwip_cli-1.3.0}/thwip/handoff.py +0 -0
  48. {thwip_cli-1.2.0 → thwip_cli-1.3.0}/thwip/tools/git_ops.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: thwip-cli
3
- Version: 1.2.0
3
+ Version: 1.3.0
4
4
  Summary: Universal coding agent multiplexer: detect, switch, and route between AI coding agents seamlessly
5
5
  Project-URL: Homepage, https://github.com/tanmayhutt/thwip-cli
6
6
  Project-URL: Repository, https://github.com/tanmayhutt/thwip-cli
@@ -91,6 +91,7 @@ thwip
91
91
  |:---|:---|
92
92
  | `/switch [agent] [model]` | Switch active agent or model mid-conversation |
93
93
  | `/handoff [agent] [model]` | Preview a target locally without switching or sending data |
94
+ | `/native codex` | Save and leave Thwip for the installed Codex CLI using its own authentication |
94
95
  | `/agents` | Show all detected coding agents, company status, and capabilities |
95
96
  | `/models [agent]` | List available models for current or target agent |
96
97
  | `/key [provider]` | Enter an API key securely without placing it in prompt history |
@@ -105,12 +106,26 @@ thwip
105
106
  | `/cost` | Show estimated session and cumulative cost |
106
107
  | `/project [path]` | View or change project working directory |
107
108
  | `Ctrl + S` | Quick switch agent prompt |
108
-
109
- Short aliases are available for frequent commands: `/a`, `/m`, `/s`, `/k`, `/g`, and `/t`.
110
109
  | `Ctrl + T` | Show agent status |
111
110
  | `Ctrl + H` | View history |
112
111
  | `/quit` | Exit thwip |
113
112
 
113
+ Short aliases are available for frequent commands: `/a`, `/m`, `/s`, `/k`, `/g`, and `/t`.
114
+
115
+ ### Existing Codex login
116
+
117
+ Use `/native codex` to open the installed native CLI without configuring a Thwip API
118
+ key. After confirmation, Thwip saves its session and replaces itself with Codex in
119
+ the selected project, using a read-only sandbox and on-request approvals. Codex
120
+ owns authentication, model selection, permissions, and usage limits. An expired
121
+ or missing login must be resolved in Codex itself.
122
+
123
+ This is a launcher, not an in-REPL provider adapter. Conversation history is not
124
+ transferred, native activity is not included in Thwip usage totals, and exiting
125
+ Codex does not automatically restart Thwip. Restart `thwip` and use `/session load`
126
+ to resume the saved Thwip conversation. Claude and Gemini native launchers are not
127
+ implemented. Direct API mode remains unchanged.
128
+
114
129
  ---
115
130
 
116
131
  ## Auditable handoffs
@@ -154,7 +169,7 @@ See [research and prior art](docs/handoff-research.md) for the differentiation r
154
169
  | Google | Gemini API (3.1 Pro Preview, 3.7 Flash, 3.5 Flash-Lite) | Chat, File Edit, Code Run, Terminal, Git |
155
170
  | OpenAI | OpenAI API (GPT-5.6 Sol, Terra, Luna) | Chat, File Edit, Code Run, Terminal, Git |
156
171
  | DeepSeek | DeepSeek V3 / R1 Reasoner | Chat, File Edit, Code Run, Reasoning |
157
- | Groq | Llama 3.3 70B, Mixtral | Chat, File Edit, Code Run |
172
+ | Groq | GPT-OSS 120B (default); Llama 3.3 for eligible enterprise accounts only | Chat, File Edit, Code Run |
158
173
  | Ollama | Local Models (Llama 3.3, Qwen Coder, DeepSeek R1) | Chat, File Edit, Code Run (Local, Offline) |
159
174
  | OpenRouter | Multi-Company Models | Gateway Routing |
160
175
 
@@ -54,6 +54,7 @@ thwip
54
54
  |:---|:---|
55
55
  | `/switch [agent] [model]` | Switch active agent or model mid-conversation |
56
56
  | `/handoff [agent] [model]` | Preview a target locally without switching or sending data |
57
+ | `/native codex` | Save and leave Thwip for the installed Codex CLI using its own authentication |
57
58
  | `/agents` | Show all detected coding agents, company status, and capabilities |
58
59
  | `/models [agent]` | List available models for current or target agent |
59
60
  | `/key [provider]` | Enter an API key securely without placing it in prompt history |
@@ -68,12 +69,26 @@ thwip
68
69
  | `/cost` | Show estimated session and cumulative cost |
69
70
  | `/project [path]` | View or change project working directory |
70
71
  | `Ctrl + S` | Quick switch agent prompt |
71
-
72
- Short aliases are available for frequent commands: `/a`, `/m`, `/s`, `/k`, `/g`, and `/t`.
73
72
  | `Ctrl + T` | Show agent status |
74
73
  | `Ctrl + H` | View history |
75
74
  | `/quit` | Exit thwip |
76
75
 
76
+ Short aliases are available for frequent commands: `/a`, `/m`, `/s`, `/k`, `/g`, and `/t`.
77
+
78
+ ### Existing Codex login
79
+
80
+ Use `/native codex` to open the installed native CLI without configuring a Thwip API
81
+ key. After confirmation, Thwip saves its session and replaces itself with Codex in
82
+ the selected project, using a read-only sandbox and on-request approvals. Codex
83
+ owns authentication, model selection, permissions, and usage limits. An expired
84
+ or missing login must be resolved in Codex itself.
85
+
86
+ This is a launcher, not an in-REPL provider adapter. Conversation history is not
87
+ transferred, native activity is not included in Thwip usage totals, and exiting
88
+ Codex does not automatically restart Thwip. Restart `thwip` and use `/session load`
89
+ to resume the saved Thwip conversation. Claude and Gemini native launchers are not
90
+ implemented. Direct API mode remains unchanged.
91
+
77
92
  ---
78
93
 
79
94
  ## Auditable handoffs
@@ -117,7 +132,7 @@ See [research and prior art](docs/handoff-research.md) for the differentiation r
117
132
  | Google | Gemini API (3.1 Pro Preview, 3.7 Flash, 3.5 Flash-Lite) | Chat, File Edit, Code Run, Terminal, Git |
118
133
  | OpenAI | OpenAI API (GPT-5.6 Sol, Terra, Luna) | Chat, File Edit, Code Run, Terminal, Git |
119
134
  | DeepSeek | DeepSeek V3 / R1 Reasoner | Chat, File Edit, Code Run, Reasoning |
120
- | Groq | Llama 3.3 70B, Mixtral | Chat, File Edit, Code Run |
135
+ | Groq | GPT-OSS 120B (default); Llama 3.3 for eligible enterprise accounts only | Chat, File Edit, Code Run |
121
136
  | Ollama | Local Models (Llama 3.3, Qwen Coder, DeepSeek R1) | Chat, File Edit, Code Run (Local, Offline) |
122
137
  | OpenRouter | Multi-Company Models | Gateway Routing |
123
138
 
@@ -0,0 +1,28 @@
1
+ # Verification status
2
+
3
+ Verified locally on 2026-09-07 for v1.3.0. The Python suite passes 186 tests.
4
+ Exhaustive behavior across every provider and
5
+ configuration has not been established.
6
+
7
+ | Area | Evidence | Remaining limitation |
8
+ | --- | --- | --- |
9
+ | Commands and aliases | Offline command smoke tests, invalid input cases | Most smoke tests check exceptions, not all rendered content |
10
+ | Sessions | Save/load, separate fresh conversations, malformed metadata, permissions, project rebinding | Concurrent writes to the same explicitly named session are not coordinated |
11
+ | Provider switching and handoff | All seven provider catalogs, portable history, bounded failover | Live account quotas and model availability unverified |
12
+ | Tool execution | Real temporary-file operations, path containment, process timeout/cancellation, invalid arguments | Shell and code tools retain local user privileges; output capture memory is unbounded |
13
+ | Native Codex launcher | Save-before-launch, consent, flags, missing binary, terminal checks, failures | Mocked process replacement; live native interaction unverified; no conversation transfer |
14
+ | Provider responses | Mocked native tool continuations; DeepSeek/Groq/OpenRouter streaming, usage-only chunks, 429/500 errors and serialized tool arguments | Other streaming and error branches still have coverage gaps |
15
+ | Display/config/auth | Configuration validation, display settings, credential boundaries, short-key masking | No full terminal/platform matrix |
16
+ | Usage | Atomic writes, malformed records, valid totals | Unknown catalog pricing may appear as zero estimated cost |
17
+ | Website | Production build and deterministic demo completion/replay test | Real browser, layout, clipboard, keyboard and accessibility checks remain incomplete |
18
+ | Dependencies | npm audit: zero advisories; Python installed-dependency audit: none found | Python audit skipped legacy local `thwip 1.0.0` metadata |
19
+ | Packaging | Wheel/source build and metadata checks | Publication is separately verified through the release workflow |
20
+
21
+ The Python suite measured 69% statement coverage with 160 passing tests before
22
+ the final credential-masking tests were added. Coverage is diagnostic evidence,
23
+ not proof that every feature works. The website test uses a minimal DOM stand-in
24
+ and controlled timers, not a browser.
25
+
26
+ This pass fixed default-session save collisions, malformed session/message
27
+ metadata acceptance, tool-argument display crashes, Rich markup interpretation
28
+ in action output, and short-key masking leakage.
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "thwip-cli"
7
- version = "1.2.0"
7
+ version = "1.3.0"
8
8
  description = "Universal coding agent multiplexer: detect, switch, and route between AI coding agents seamlessly"
9
9
  readme = "README.md"
10
10
  license = "MIT"
@@ -0,0 +1,164 @@
1
+ import json
2
+ from io import StringIO
3
+ from types import SimpleNamespace
4
+
5
+ import pytest
6
+ from rich.console import Console
7
+
8
+ from thwip.agents import AgentRegistry
9
+ from thwip.agents.base import AgentDone, Capability, LimitHit, LimitStatus, TextDelta
10
+ from thwip.cli import ThwipCLI
11
+ from thwip.config import ThwipConfig, get_usage_path
12
+ from thwip.limits import UsageTracker
13
+ from thwip.session import Session
14
+ from thwip.tools import ToolManager
15
+
16
+
17
+ @pytest.fixture
18
+ def cli(tmp_path, monkeypatch):
19
+ monkeypatch.setenv('THWIP_CONFIG_DIR', str(tmp_path / 'config'))
20
+ cli = ThwipCLI.__new__(ThwipCLI)
21
+ cli.config = ThwipConfig(project=str(tmp_path), auto_save=False)
22
+ cli.registry = AgentRegistry(cli.config)
23
+ for a in cli.registry.list_agents():
24
+ monkeypatch.setattr(a, 'is_installed', lambda: True)
25
+ monkeypatch.setattr(a, 'is_configured', lambda: False)
26
+ if a.name == 'ollama':
27
+ a._cached_models = a.get_handoff_models()
28
+ cli.current_agent = cli.registry.get_agent('claude')
29
+ cli.session = Session(project_path=str(tmp_path), current_agent='claude', current_model=cli.current_agent.get_default_model())
30
+ cli.usage_tracker = UsageTracker()
31
+ cli.tool_manager = ToolManager(str(tmp_path))
32
+ cli.detector = SimpleNamespace(scan_all=list)
33
+ monkeypatch.setattr('builtins.input', lambda *args: '')
34
+ monkeypatch.setattr('getpass.getpass', lambda *args: '')
35
+ console = Console(file=StringIO(), width=100, color_system=None)
36
+ monkeypatch.setattr('thwip.cli.console', console)
37
+ monkeypatch.setattr('thwip.theme.console', console)
38
+ return cli
39
+
40
+
41
+ @pytest.mark.parametrize('command', [
42
+ '/help', '/h', '/about', '/guide', '/g', '/info', '/agents', '/a', '/list',
43
+ '/models', '/m', '/models flagship', '/models balanced', '/models fast', '/models google',
44
+ '/tools', '/t', '/status', '/limits', '/detect', '/history', '/clear', '/reset', '/cost',
45
+ '/project', '/session save audit', '/session load missing', '/session list', '/session clear',
46
+ '/key', '/key google', '/key invalid', '/key openai REJECTED_TEST_KEY',
47
+ '/handoff', '/handoff invalid', '/handoff google invalid', '/switch', '/switch invalid',
48
+ '/switch google invalid', '/unknown', '/q', '/exit', '/quit',
49
+ ])
50
+ @pytest.mark.asyncio
51
+ async def test_command_no_exception(cli, command):
52
+ await cli.handle_command(command)
53
+
54
+
55
+ @pytest.mark.parametrize('provider', ['claude','google','openai','deepseek','groq','ollama','openrouter'])
56
+ @pytest.mark.asyncio
57
+ async def test_all_catalogued_handoffs(cli, provider):
58
+ before = list(cli.session.messages)
59
+ for model in cli.registry.get_agent(provider).get_handoff_models():
60
+ await cli.handle_command(f'/handoff {provider} {model.id}')
61
+ assert cli.session.messages == before
62
+
63
+
64
+ @pytest.mark.asyncio
65
+ async def test_project_path_with_spaces(cli, tmp_path):
66
+ project = tmp_path / 'project with spaces'
67
+ project.mkdir()
68
+ await cli.handle_command(f'/project {project}')
69
+ assert cli.session.project_path == str(project)
70
+
71
+
72
+ def test_usage_file_private(cli):
73
+ cli.usage_tracker.record_usage('openai', 'gpt-5.6-terra', 10, 5)
74
+ assert get_usage_path().stat().st_mode & 0o777 == 0o600
75
+
76
+
77
+ @pytest.mark.parametrize('tool,args', [
78
+ ('read_file', {'file_path': 12}),
79
+ ('list_files', {'sub_dir': ['bad']}),
80
+ ('run_python', {'code': None}),
81
+ ])
82
+ def test_invalid_tool_args_return_error_not_exception(cli, tool, args):
83
+ result = cli.tool_manager.execute_tool(tool, args)
84
+ assert isinstance(result, str)
85
+
86
+
87
+ @pytest.mark.asyncio
88
+ async def test_history_displays_literal_markup(cli):
89
+ cli.session.add_user_message('Show [/not-a-tag] literally')
90
+ await cli.handle_command('/history')
91
+
92
+
93
+ @pytest.mark.parametrize('value', ['wrong', None, -1, True])
94
+ def test_corrupted_handoff_counter_rejected_or_sanitized(cli, value):
95
+ path = cli.session.save('damaged')
96
+ data = json.loads(path.read_text())
97
+ data['observed_tool_results'] = value
98
+ path.write_text(json.dumps(data))
99
+ loaded = Session.load('damaged')
100
+ assert loaded is None or (type(loaded.observed_tool_results) is int and loaded.observed_tool_results >= 0)
101
+
102
+
103
+ def test_invalid_config_type_rejected_or_defaulted(cli):
104
+ cli.config._apply_toml({'defaults': {'project': 123, 'confirm_tools': 'false'}})
105
+ assert isinstance(cli.config.project, str)
106
+ assert type(cli.config.confirm_tools) is bool
107
+
108
+
109
+ @pytest.mark.asyncio
110
+ async def test_session_roundtrip_rebinds_tools(cli, tmp_path):
111
+ target = tmp_path / 'other'
112
+ target.mkdir()
113
+ saved = Session(project_path=str(target), current_agent='google', current_model='missing')
114
+ saved.save('loadable')
115
+ await cli.handle_command('/session load loadable')
116
+ assert cli.tool_manager.file_editor.project_path == target
117
+ assert cli.current_agent.name == 'google'
118
+ assert cli.session.current_model == cli.current_agent.get_default_model()
119
+
120
+
121
+ def test_all_tools_and_safety(cli, tmp_path):
122
+ tm = cli.tool_manager
123
+ assert 'Successfully' in tm.execute_tool('write_file', {'file_path':'a.txt','content':'one'})
124
+ assert tm.execute_tool('read_file', {'file_path':'a.txt'}) == 'one'
125
+ assert 'Successfully' in tm.execute_tool('edit_file', {'file_path':'a.txt','old_str':'one','new_str':'two'})
126
+ assert 'a.txt' in tm.execute_tool('list_files', {})
127
+ assert '42' in tm.execute_tool('run_python', {'code':'print(6 * 7)'})
128
+ assert 'Exit code: 0' in tm.execute_tool('run_command', {'command':'printf audit-ok'})
129
+ assert isinstance(tm.execute_tool('git_status', {}), str)
130
+ assert isinstance(tm.execute_tool('git_diff', {}), str)
131
+ assert 'outside' in tm.execute_tool('read_file', {'file_path':'../outside'})
132
+ outside = tmp_path.parent / 'audit-outside-file'
133
+ outside.write_text('not accessible')
134
+ (tmp_path / 'link').symlink_to(outside)
135
+ assert 'outside' in tm.execute_tool('read_file', {'file_path':'link'})
136
+ assert 'Error' in tm.execute_tool('unknown', {})
137
+ assert {t['name'] for t in tm.get_anthropic_tools()} == {t['function']['name'] for t in tm.get_openai_tools()}
138
+
139
+
140
+ @pytest.mark.asyncio
141
+ async def test_failover_does_not_retry_already_failed_provider(cli, monkeypatch):
142
+ agents = [cli.registry.get_agent('claude'), cli.registry.get_agent('google')]
143
+ attempts = []
144
+ for agent in agents:
145
+ monkeypatch.setattr(agent, 'is_configured', lambda: True)
146
+ monkeypatch.setattr(agent, 'get_capabilities_for_model', lambda model: {Capability.CHAT})
147
+
148
+ def make_chat(name):
149
+ async def chat(**kwargs):
150
+ attempts.append(name)
151
+ # Stop safely after four simulated rate limits; no actual network calls.
152
+ if len(attempts) <= 4:
153
+ yield LimitHit(error_type=LimitStatus.RATE_LIMITED, message='test quota')
154
+ else:
155
+ yield TextDelta(content='stop test')
156
+ yield AgentDone()
157
+ return chat
158
+
159
+ monkeypatch.setattr(agent, 'chat', make_chat(agent.name))
160
+ monkeypatch.setattr(cli.registry, 'get_ready_agents', lambda: agents)
161
+ cli.config.fallback.chain = ['claude', 'google']
162
+ cli.config.limits.auto_switch = True
163
+ await cli.process_user_message('hello')
164
+ assert len(attempts) <= len(agents), f'Repeated failed providers: {attempts}'
@@ -75,6 +75,30 @@ class UnconfiguredAgent(ToolCallingAgent):
75
75
  return False
76
76
 
77
77
 
78
+ @pytest.mark.asyncio
79
+ @pytest.mark.parametrize("arguments", [None, [], "bad", {"file_path": "[broken]"}])
80
+ async def test_invalid_tool_arguments_return_errors_without_crashing(tmp_path, arguments):
81
+ class InvalidToolAgent(ToolCallingAgent):
82
+ async def chat(self, messages, **kwargs):
83
+ self.calls.append(messages)
84
+ if len(self.calls) == 1:
85
+ yield ToolUseStart(tool_id="bad", tool_name="read_file", args=arguments)
86
+ yield AgentDone()
87
+ else:
88
+ assert "Error" in messages[-1]["content"]
89
+ yield TextDelta(content="Recovered from invalid tool arguments.")
90
+ yield AgentDone()
91
+
92
+ cli = ThwipCLI.__new__(ThwipCLI)
93
+ cli.config = SimpleNamespace(stream=True, confirm_tools=True)
94
+ cli.session = Session(current_agent="fake", current_model="fake-model")
95
+ cli.current_agent = InvalidToolAgent()
96
+ cli.tool_manager = ToolManager(tmp_path)
97
+ cli.usage_tracker = FakeUsageTracker()
98
+ await cli.process_user_message("Read")
99
+ assert cli.session.messages[-1].content == "Recovered from invalid tool arguments."
100
+
101
+
78
102
  @pytest.mark.asyncio
79
103
  async def test_unconfigured_agent_shows_setup_guidance(tmp_path):
80
104
  cli = ThwipCLI.__new__(ThwipCLI)
@@ -0,0 +1,75 @@
1
+ """Exercise streaming completion and provider error boundaries without network calls."""
2
+
3
+ import json
4
+ from types import SimpleNamespace
5
+
6
+ import httpx
7
+ import openai
8
+ import pytest
9
+
10
+ from thwip.agents.base import AgentDone, LimitHit, LimitStatus, TextDelta
11
+ from thwip.agents.deepseek_agent import DeepSeekAgent
12
+ from thwip.agents.groq_agent import GroqAgent
13
+ from thwip.agents.openrouter_agent import OpenRouterAgent
14
+
15
+
16
+ @pytest.mark.parametrize("agent_type", [DeepSeekAgent, GroqAgent, OpenRouterAgent])
17
+ @pytest.mark.asyncio
18
+ async def test_tool_continuation_serializes_arguments_without_mutating_history(agent_type):
19
+ original = {"role": "assistant", "content": "", "tool_calls": [{
20
+ "id": "call1", "type": "function", "function": {
21
+ "name": "read_file", "arguments": {"file_path": "note.txt"},
22
+ },
23
+ }], "_native_state": {"other": "private"}}
24
+
25
+ async def create(**kwargs):
26
+ sent = kwargs["messages"][0]
27
+ assert "_native_state" not in sent
28
+ assert json.loads(sent["tool_calls"][0]["function"]["arguments"]) == {"file_path": "note.txt"}
29
+ return SimpleNamespace(choices=[SimpleNamespace(message=SimpleNamespace(content="done", tool_calls=[]))], usage=None)
30
+
31
+ agent = agent_type(api_key="test")
32
+ agent._client = SimpleNamespace(chat=SimpleNamespace(completions=SimpleNamespace(create=create)))
33
+ events = [event async for event in agent.chat([original], stream=False)]
34
+ assert any(isinstance(event, AgentDone) for event in events)
35
+ assert isinstance(original["tool_calls"][0]["function"]["arguments"], dict)
36
+ assert "_native_state" in original
37
+
38
+
39
+ @pytest.mark.parametrize("agent_type", [DeepSeekAgent, GroqAgent, OpenRouterAgent])
40
+ @pytest.mark.asyncio
41
+ async def test_stream_text_and_usage_only_final_chunk(agent_type):
42
+ async def chunks():
43
+ yield SimpleNamespace(choices=[SimpleNamespace(delta=SimpleNamespace(
44
+ content="hello", tool_calls=None))], usage=None)
45
+ yield SimpleNamespace(choices=[], usage=SimpleNamespace(prompt_tokens=9, completion_tokens=3))
46
+
47
+ async def create(**kwargs):
48
+ assert kwargs["stream"] is True
49
+ return chunks()
50
+
51
+ agent = agent_type(api_key="test")
52
+ agent._client = SimpleNamespace(chat=SimpleNamespace(completions=SimpleNamespace(create=create)))
53
+ events = [event async for event in agent.chat([{"role": "user", "content": "hi"}])]
54
+ assert isinstance(events[0], TextDelta) and events[0].content == "hello"
55
+ assert isinstance(events[-1], AgentDone)
56
+ assert events[-1].usage.input_tokens == 9
57
+ assert events[-1].usage.output_tokens == 3
58
+ assert agent.check_limits() == LimitStatus.OK
59
+
60
+
61
+ @pytest.mark.parametrize("agent_type", [DeepSeekAgent, GroqAgent, OpenRouterAgent])
62
+ @pytest.mark.parametrize("status", [429, 500])
63
+ @pytest.mark.asyncio
64
+ async def test_provider_errors_emit_failure_not_success(agent_type, status):
65
+ response = httpx.Response(status, request=httpx.Request("POST", "https://example.invalid"))
66
+ error_type = openai.RateLimitError if status == 429 else openai.APIStatusError
67
+
68
+ async def create(**kwargs):
69
+ raise error_type("simulated failure", response=response, body=None)
70
+
71
+ agent = agent_type(api_key="test")
72
+ agent._client = SimpleNamespace(chat=SimpleNamespace(completions=SimpleNamespace(create=create)))
73
+ events = [event async for event in agent.chat([])]
74
+ assert len(events) == 1 and isinstance(events[0], LimitHit)
75
+ assert events[0].error_type == (LimitStatus.RATE_LIMITED if status == 429 else LimitStatus.UNKNOWN)
@@ -0,0 +1,69 @@
1
+ """Native launch boundaries without executing a provider or reading credentials."""
2
+
3
+ from types import SimpleNamespace
4
+
5
+ import pytest
6
+
7
+ from thwip.cli import ThwipCLI
8
+
9
+
10
+ @pytest.fixture
11
+ def launcher(tmp_path, monkeypatch):
12
+ events = []
13
+ cli = ThwipCLI.__new__(ThwipCLI)
14
+ cli.session = SimpleNamespace(project_path=str(tmp_path))
15
+ cli.session.save = lambda: events.append("save") or tmp_path / "session.json"
16
+ monkeypatch.setattr("thwip.cli.shutil.which", lambda name: "/installed/codex")
17
+ monkeypatch.setattr("thwip.cli.sys.stdin.isatty", lambda: True)
18
+ monkeypatch.setattr("thwip.cli.sys.stdout.isatty", lambda: True)
19
+ monkeypatch.setattr("thwip.cli.console.input", lambda prompt: "yes")
20
+ monkeypatch.setattr("thwip.cli.print_info", lambda message: None)
21
+ monkeypatch.setattr("thwip.cli.print_error", lambda message: events.append("error"))
22
+ monkeypatch.setattr("thwip.cli.os.execv", lambda path, args: events.append((path, args)))
23
+ return cli, events
24
+
25
+
26
+ @pytest.mark.asyncio
27
+ async def test_native_command_saves_before_replacing_process(launcher):
28
+ cli, events = launcher
29
+ await cli.handle_command("/native codex")
30
+ assert events == ["save", ("/installed/codex", [
31
+ "/installed/codex", "--cd", cli.session.project_path,
32
+ "--sandbox", "read-only", "--ask-for-approval", "on-request",
33
+ ])]
34
+
35
+
36
+ @pytest.mark.parametrize("command", ["/native", "/native claude", "/native codex --dangerously-bypass-approvals-and-sandbox"])
37
+ @pytest.mark.asyncio
38
+ async def test_native_rejects_unsupported_arguments(launcher, command):
39
+ cli, events = launcher
40
+ await cli.handle_command(command)
41
+ assert events == []
42
+
43
+
44
+ def test_native_decline_does_not_save_or_launch(launcher, monkeypatch):
45
+ cli, events = launcher
46
+ monkeypatch.setattr("thwip.cli.console.input", lambda prompt: "")
47
+ cli.cmd_native("codex")
48
+ assert events == []
49
+
50
+
51
+ @pytest.mark.parametrize("failure", ["missing", "noninteractive", "project", "save", "exec"])
52
+ def test_native_failure_stays_in_thwip(launcher, monkeypatch, failure):
53
+ cli, events = launcher
54
+
55
+ def fail(*args):
56
+ raise OSError("private detail")
57
+
58
+ if failure == "missing":
59
+ monkeypatch.setattr("thwip.cli.shutil.which", lambda name: None)
60
+ elif failure == "noninteractive":
61
+ monkeypatch.setattr("thwip.cli.sys.stdin.isatty", lambda: False)
62
+ elif failure == "project":
63
+ cli.session.project_path += "/missing"
64
+ elif failure == "save":
65
+ cli.session.save = fail
66
+ else:
67
+ monkeypatch.setattr("thwip.cli.os.execv", fail)
68
+ cli.cmd_native("codex")
69
+ assert events == (["save", "error"] if failure == "exec" else ["error"])
@@ -0,0 +1,177 @@
1
+ """Focused verification of repaired settings and native continuation."""
2
+
3
+ import asyncio
4
+ import json
5
+ import time
6
+ from types import SimpleNamespace
7
+
8
+ import pytest
9
+ from rich.text import Text
10
+
11
+ from thwip.agents.base import AgentDone
12
+ from thwip.agents.claude_agent import ClaudeAgent
13
+ from thwip.agents.google_agent import GoogleAgent
14
+ from thwip.agents.openai_agent import OpenAIAgent
15
+ from thwip.cli import ThwipCLI
16
+ from thwip.config import ThwipConfig
17
+ from thwip.detector import SystemDetector
18
+ from thwip.limits import UsageTracker
19
+ from thwip.session import Session
20
+ from thwip.tools.terminal import TerminalRunner
21
+
22
+
23
+ @pytest.mark.parametrize("field,value", [
24
+ ("estimated_cost", float("nan")), ("estimated_cost", float("inf")),
25
+ ("estimated_cost", True), ("estimated_cost", 10 ** 400),
26
+ ("last_limit_hit_timestamp", -1), ("last_limit_hit_timestamp", "bad"),
27
+ ("last_limit_hit_timestamp", float("nan")), ("last_error", []),
28
+ ])
29
+ def test_usage_rejects_corrupt_values_without_losing_other_agents(tmp_path, monkeypatch, field, value):
30
+ path = tmp_path / "usage.json"
31
+ path.write_text(json.dumps({"bad": {field: value}, "good": {"estimated_cost": 0.5}}))
32
+ monkeypatch.setattr("thwip.limits.get_usage_path", lambda: path)
33
+ tracker = UsageTracker()
34
+ assert set(tracker.stats) == {"good"}
35
+ assert tracker.get_summary()["total_cost"] == 0.5
36
+
37
+
38
+ @pytest.mark.asyncio
39
+ @pytest.mark.parametrize("provider,auth", [("ollama", "none"), ("openai", "subscription"), ("google", "oauth"), ("claude", "none")])
40
+ async def test_setup_guidance_matches_transport(monkeypatch, provider, auth):
41
+ panels = []
42
+ monkeypatch.setattr("thwip.cli.console.print", lambda panel: panels.append(panel))
43
+ cli = ThwipCLI.__new__(ThwipCLI)
44
+ cli.current_agent = SimpleNamespace(
45
+ name=provider, display_name="Provider [literal]", auth_method=auth,
46
+ is_configured=lambda: False,
47
+ )
48
+ cli.session = Session(current_model="model [literal]")
49
+ await cli.process_user_message("hello")
50
+ text = panels[0].renderable.plain
51
+ assert not cli.session.messages
52
+ if provider == "ollama":
53
+ assert "ollama serve" in text and "No API key is required" in text
54
+ assert "/key" not in text
55
+ else:
56
+ assert "model [literal]" in text and f"/key {provider}" in text
57
+ assert ("sign-in was detected" in text) == (auth in ("oauth", "subscription"))
58
+
59
+
60
+ def test_display_switches_hide_status_fields():
61
+ cli = ThwipCLI.__new__(ThwipCLI)
62
+ cli.config = ThwipConfig()
63
+ cli.current_agent = OpenAIAgent(api_key="test")
64
+ cli.session = Session(current_model=cli.current_agent.get_default_model())
65
+ cli.session.add_assistant_message("done", "openai", cli.session.current_model, tokens=42)
66
+ cli.usage_tracker = SimpleNamespace(get_summary=lambda: {"total_cost": 1.25})
67
+ visible = cli._render_status().plain
68
+ assert "42 tok" in visible and "$1.2500" in visible and "[chat]" in visible
69
+ for name in ("show_agent_badge", "show_token_count", "show_cost", "show_capabilities"):
70
+ setattr(cli.config.display, name, False)
71
+ assert cli._render_status().plain == ""
72
+ cli.config.display.markdown = False
73
+ assert isinstance(cli._render_response("**literal**"), Text)
74
+ assert cli._render_response("**literal**").plain == "**literal**"
75
+
76
+
77
+ @pytest.mark.parametrize("content,expected", [
78
+ ('{"userID":"metadata-only"}', False),
79
+ ('{"apiKey":""}', False),
80
+ ('{"apiKey":42}', False),
81
+ ('{"apiKey":"test-key"}', True),
82
+ ('{"credentials":{"apiKey":"test-key"}}', True),
83
+ ])
84
+ def test_detector_requires_actual_key_field(tmp_path, monkeypatch, content, expected):
85
+ path = tmp_path / "claude.json"
86
+ path.write_text(content)
87
+ monkeypatch.delenv("ANTHROPIC_API_KEY", raising=False)
88
+ monkeypatch.setattr("thwip.detector.KNOWN_CLI_TOOLS", [{
89
+ "name": "Claude", "company": "Anthropic", "binaries": [], "category": "CLI Agent",
90
+ "key_env": "ANTHROPIC_API_KEY", "config_files": [str(path)], "capabilities": [],
91
+ }])
92
+ monkeypatch.setattr("thwip.detector.KNOWN_APPS", [])
93
+ monkeypatch.setattr("thwip.detector.subprocess.run", lambda *a, **k: SimpleNamespace(returncode=1))
94
+ found = SystemDetector().scan_all()
95
+ assert bool(found) is expected
96
+
97
+
98
+ @pytest.mark.asyncio
99
+ async def test_openai_preserves_reasoning_and_function_items():
100
+ captured = []
101
+ items = [SimpleNamespace(type="reasoning", id="r1", encrypted_content="opaque", summary=[]),
102
+ SimpleNamespace(type="function_call", call_id="c1", name="read_file", arguments='{"file_path":"x"}')]
103
+
104
+ async def create(**kwargs):
105
+ captured.append(kwargs)
106
+ return SimpleNamespace(output_text="", output=items, usage=None)
107
+
108
+ agent = OpenAIAgent(api_key="test")
109
+ agent._client = SimpleNamespace(responses=SimpleNamespace(create=create))
110
+ events = [e async for e in agent.chat(messages=[{"role": "user", "content": "read"}], stream=False)]
111
+ done = next(e for e in events if isinstance(e, AgentDone))
112
+ messages = [{"role": "assistant", "content": "", "_native_state": done.native_state},
113
+ {"role": "tool", "tool_call_id": "c1", "content": "result"}]
114
+ _ = [e async for e in agent.chat(messages=messages, stream=False)]
115
+ assert captured[-1]["input"][:2] == [vars(item) for item in items]
116
+ assert captured[-1]["input"][2]["type"] == "function_call_output"
117
+ assert captured[-1]["store"] is False
118
+
119
+
120
+ @pytest.mark.asyncio
121
+ async def test_claude_preserves_thinking_signature():
122
+ captured = []
123
+ blocks = [SimpleNamespace(type="thinking", thinking="thought", signature="signature"),
124
+ SimpleNamespace(type="tool_use", id="c1", name="read_file", input={"file_path":"x"})]
125
+
126
+ async def create(**kwargs):
127
+ captured.append(kwargs)
128
+ return SimpleNamespace(content=blocks, usage=SimpleNamespace(input_tokens=1, output_tokens=1), stop_reason="tool_use")
129
+
130
+ agent = ClaudeAgent(api_key="test")
131
+ agent._client = SimpleNamespace(messages=SimpleNamespace(create=create))
132
+ events = [e async for e in agent.chat(messages=[], stream=False)]
133
+ done = next(e for e in events if isinstance(e, AgentDone))
134
+ _ = [e async for e in agent.chat(messages=[{"role":"assistant", "_native_state":done.native_state}], stream=False)]
135
+ assert captured[-1]["messages"][0]["content"] == [vars(block) for block in blocks]
136
+
137
+
138
+ @pytest.mark.asyncio
139
+ async def test_google_preserves_thought_signature():
140
+ from google.genai import types
141
+
142
+ captured = []
143
+ content = types.Content(role="model", parts=[types.Part(
144
+ function_call=types.FunctionCall(name="read_file", args={"file_path":"x"}),
145
+ thought_signature=b"opaque-signature",
146
+ )])
147
+
148
+ def generate(**kwargs):
149
+ captured.append(kwargs)
150
+ return SimpleNamespace(candidates=[SimpleNamespace(content=content)], usage_metadata=None)
151
+
152
+ agent = GoogleAgent(api_key="test")
153
+ agent._client = SimpleNamespace(models=SimpleNamespace(generate_content=generate))
154
+ events = [e async for e in agent.chat(messages=[], stream=False)]
155
+ done = next(e for e in events if isinstance(e, AgentDone))
156
+ _ = [e async for e in agent.chat(messages=[{"role":"assistant", "_native_state":done.native_state}], stream=False)]
157
+ assert captured[-1]["contents"][0].parts[0].thought_signature == b"opaque-signature"
158
+
159
+
160
+ def test_sync_timeout_stops_child_writes(tmp_path):
161
+ runner = TerminalRunner(tmp_path)
162
+ result = runner.run_command("sleep 0.3; printf unwanted > late.txt", timeout=0.03)
163
+ assert "timed out" in result
164
+ time.sleep(0.4)
165
+ assert not (tmp_path / "late.txt").exists()
166
+
167
+
168
+ @pytest.mark.asyncio
169
+ async def test_async_cancellation_stops_child_writes(tmp_path):
170
+ runner = TerminalRunner(tmp_path)
171
+ task = asyncio.create_task(runner.run_command_async("sleep 0.3; printf unwanted > late.txt"))
172
+ await asyncio.sleep(0.03)
173
+ task.cancel()
174
+ with pytest.raises(asyncio.CancelledError):
175
+ await task
176
+ await asyncio.sleep(0.4)
177
+ assert not (tmp_path / "late.txt").exists()