thwip-cli 1.2.0__tar.gz → 1.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {thwip_cli-1.2.0 → thwip_cli-1.3.0}/PKG-INFO +19 -4
- {thwip_cli-1.2.0 → thwip_cli-1.3.0}/README.md +18 -3
- thwip_cli-1.3.0/docs/verification.md +28 -0
- {thwip_cli-1.2.0 → thwip_cli-1.3.0}/pyproject.toml +1 -1
- thwip_cli-1.3.0/tests/test_audit_regressions.py +164 -0
- {thwip_cli-1.2.0 → thwip_cli-1.3.0}/tests/test_cli.py +24 -0
- thwip_cli-1.3.0/tests/test_compatible_streaming.py +75 -0
- thwip_cli-1.3.0/tests/test_native_launcher.py +69 -0
- thwip_cli-1.3.0/tests/test_repair_verification.py +177 -0
- {thwip_cli-1.2.0 → thwip_cli-1.3.0}/tests/test_session.py +41 -0
- thwip_cli-1.3.0/tests/test_utils.py +17 -0
- {thwip_cli-1.2.0 → thwip_cli-1.3.0}/thwip/__init__.py +1 -1
- {thwip_cli-1.2.0 → thwip_cli-1.3.0}/thwip/agents/base.py +1 -0
- thwip_cli-1.3.0/thwip/agents/chat_messages.py +20 -0
- {thwip_cli-1.2.0 → thwip_cli-1.3.0}/thwip/agents/claude_agent.py +6 -1
- {thwip_cli-1.2.0 → thwip_cli-1.3.0}/thwip/agents/deepseek_agent.py +2 -1
- {thwip_cli-1.2.0 → thwip_cli-1.3.0}/thwip/agents/google_agent.py +5 -1
- {thwip_cli-1.2.0 → thwip_cli-1.3.0}/thwip/agents/groq_agent.py +5 -3
- {thwip_cli-1.2.0 → thwip_cli-1.3.0}/thwip/agents/openai_agent.py +8 -1
- {thwip_cli-1.2.0 → thwip_cli-1.3.0}/thwip/agents/openrouter_agent.py +2 -1
- {thwip_cli-1.2.0 → thwip_cli-1.3.0}/thwip/cli.py +135 -38
- {thwip_cli-1.2.0 → thwip_cli-1.3.0}/thwip/config.py +27 -0
- {thwip_cli-1.2.0 → thwip_cli-1.3.0}/thwip/detector.py +16 -10
- {thwip_cli-1.2.0 → thwip_cli-1.3.0}/thwip/limits.py +33 -4
- {thwip_cli-1.2.0 → thwip_cli-1.3.0}/thwip/session.py +29 -1
- {thwip_cli-1.2.0 → thwip_cli-1.3.0}/thwip/shortcuts.py +1 -0
- {thwip_cli-1.2.0 → thwip_cli-1.3.0}/thwip/theme.py +17 -11
- {thwip_cli-1.2.0 → thwip_cli-1.3.0}/thwip/tools/__init__.py +22 -1
- {thwip_cli-1.2.0 → thwip_cli-1.3.0}/thwip/tools/code_runner.py +4 -6
- {thwip_cli-1.2.0 → thwip_cli-1.3.0}/thwip/tools/file_editor.py +1 -1
- {thwip_cli-1.2.0 → thwip_cli-1.3.0}/thwip/tools/terminal.py +35 -5
- {thwip_cli-1.2.0 → thwip_cli-1.3.0}/thwip/utils.py +1 -1
- {thwip_cli-1.2.0 → thwip_cli-1.3.0}/uv.lock +1 -1
- {thwip_cli-1.2.0 → thwip_cli-1.3.0}/.github/workflows/publish.yml +0 -0
- {thwip_cli-1.2.0 → thwip_cli-1.3.0}/.gitignore +0 -0
- {thwip_cli-1.2.0 → thwip_cli-1.3.0}/LICENSE +0 -0
- {thwip_cli-1.2.0 → thwip_cli-1.3.0}/docs/handoff-research.md +0 -0
- {thwip_cli-1.2.0 → thwip_cli-1.3.0}/install.sh +0 -0
- {thwip_cli-1.2.0 → thwip_cli-1.3.0}/tests/test_agents.py +0 -0
- {thwip_cli-1.2.0 → thwip_cli-1.3.0}/tests/test_config.py +0 -0
- {thwip_cli-1.2.0 → thwip_cli-1.3.0}/tests/test_detector.py +0 -0
- {thwip_cli-1.2.0 → thwip_cli-1.3.0}/tests/test_handoff.py +0 -0
- {thwip_cli-1.2.0 → thwip_cli-1.3.0}/tests/test_tools.py +0 -0
- {thwip_cli-1.2.0 → thwip_cli-1.3.0}/thwip/__main__.py +0 -0
- {thwip_cli-1.2.0 → thwip_cli-1.3.0}/thwip/agents/__init__.py +0 -0
- {thwip_cli-1.2.0 → thwip_cli-1.3.0}/thwip/agents/ollama_agent.py +0 -0
- {thwip_cli-1.2.0 → thwip_cli-1.3.0}/thwip/handoff.py +0 -0
- {thwip_cli-1.2.0 → thwip_cli-1.3.0}/thwip/tools/git_ops.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: thwip-cli
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.3.0
|
|
4
4
|
Summary: Universal coding agent multiplexer: detect, switch, and route between AI coding agents seamlessly
|
|
5
5
|
Project-URL: Homepage, https://github.com/tanmayhutt/thwip-cli
|
|
6
6
|
Project-URL: Repository, https://github.com/tanmayhutt/thwip-cli
|
|
@@ -91,6 +91,7 @@ thwip
|
|
|
91
91
|
|:---|:---|
|
|
92
92
|
| `/switch [agent] [model]` | Switch active agent or model mid-conversation |
|
|
93
93
|
| `/handoff [agent] [model]` | Preview a target locally without switching or sending data |
|
|
94
|
+
| `/native codex` | Save and leave Thwip for the installed Codex CLI using its own authentication |
|
|
94
95
|
| `/agents` | Show all detected coding agents, company status, and capabilities |
|
|
95
96
|
| `/models [agent]` | List available models for current or target agent |
|
|
96
97
|
| `/key [provider]` | Enter an API key securely without placing it in prompt history |
|
|
@@ -105,12 +106,26 @@ thwip
|
|
|
105
106
|
| `/cost` | Show estimated session and cumulative cost |
|
|
106
107
|
| `/project [path]` | View or change project working directory |
|
|
107
108
|
| `Ctrl + S` | Quick switch agent prompt |
|
|
108
|
-
|
|
109
|
-
Short aliases are available for frequent commands: `/a`, `/m`, `/s`, `/k`, `/g`, and `/t`.
|
|
110
109
|
| `Ctrl + T` | Show agent status |
|
|
111
110
|
| `Ctrl + H` | View history |
|
|
112
111
|
| `/quit` | Exit thwip |
|
|
113
112
|
|
|
113
|
+
Short aliases are available for frequent commands: `/a`, `/m`, `/s`, `/k`, `/g`, and `/t`.
|
|
114
|
+
|
|
115
|
+
### Existing Codex login
|
|
116
|
+
|
|
117
|
+
Use `/native codex` to open the installed native CLI without configuring a Thwip API
|
|
118
|
+
key. After confirmation, Thwip saves its session and replaces itself with Codex in
|
|
119
|
+
the selected project, using a read-only sandbox and on-request approvals. Codex
|
|
120
|
+
owns authentication, model selection, permissions, and usage limits. An expired
|
|
121
|
+
or missing login must be resolved in Codex itself.
|
|
122
|
+
|
|
123
|
+
This is a launcher, not an in-REPL provider adapter. Conversation history is not
|
|
124
|
+
transferred, native activity is not included in Thwip usage totals, and exiting
|
|
125
|
+
Codex does not automatically restart Thwip. Restart `thwip` and use `/session load`
|
|
126
|
+
to resume the saved Thwip conversation. Claude and Gemini native launchers are not
|
|
127
|
+
implemented. Direct API mode remains unchanged.
|
|
128
|
+
|
|
114
129
|
---
|
|
115
130
|
|
|
116
131
|
## Auditable handoffs
|
|
@@ -154,7 +169,7 @@ See [research and prior art](docs/handoff-research.md) for the differentiation r
|
|
|
154
169
|
| Google | Gemini API (3.1 Pro Preview, 3.7 Flash, 3.5 Flash-Lite) | Chat, File Edit, Code Run, Terminal, Git |
|
|
155
170
|
| OpenAI | OpenAI API (GPT-5.6 Sol, Terra, Luna) | Chat, File Edit, Code Run, Terminal, Git |
|
|
156
171
|
| DeepSeek | DeepSeek V3 / R1 Reasoner | Chat, File Edit, Code Run, Reasoning |
|
|
157
|
-
| Groq | Llama 3.3
|
|
172
|
+
| Groq | GPT-OSS 120B (default); Llama 3.3 for eligible enterprise accounts only | Chat, File Edit, Code Run |
|
|
158
173
|
| Ollama | Local Models (Llama 3.3, Qwen Coder, DeepSeek R1) | Chat, File Edit, Code Run (Local, Offline) |
|
|
159
174
|
| OpenRouter | Multi-Company Models | Gateway Routing |
|
|
160
175
|
|
|
@@ -54,6 +54,7 @@ thwip
|
|
|
54
54
|
|:---|:---|
|
|
55
55
|
| `/switch [agent] [model]` | Switch active agent or model mid-conversation |
|
|
56
56
|
| `/handoff [agent] [model]` | Preview a target locally without switching or sending data |
|
|
57
|
+
| `/native codex` | Save and leave Thwip for the installed Codex CLI using its own authentication |
|
|
57
58
|
| `/agents` | Show all detected coding agents, company status, and capabilities |
|
|
58
59
|
| `/models [agent]` | List available models for current or target agent |
|
|
59
60
|
| `/key [provider]` | Enter an API key securely without placing it in prompt history |
|
|
@@ -68,12 +69,26 @@ thwip
|
|
|
68
69
|
| `/cost` | Show estimated session and cumulative cost |
|
|
69
70
|
| `/project [path]` | View or change project working directory |
|
|
70
71
|
| `Ctrl + S` | Quick switch agent prompt |
|
|
71
|
-
|
|
72
|
-
Short aliases are available for frequent commands: `/a`, `/m`, `/s`, `/k`, `/g`, and `/t`.
|
|
73
72
|
| `Ctrl + T` | Show agent status |
|
|
74
73
|
| `Ctrl + H` | View history |
|
|
75
74
|
| `/quit` | Exit thwip |
|
|
76
75
|
|
|
76
|
+
Short aliases are available for frequent commands: `/a`, `/m`, `/s`, `/k`, `/g`, and `/t`.
|
|
77
|
+
|
|
78
|
+
### Existing Codex login
|
|
79
|
+
|
|
80
|
+
Use `/native codex` to open the installed native CLI without configuring a Thwip API
|
|
81
|
+
key. After confirmation, Thwip saves its session and replaces itself with Codex in
|
|
82
|
+
the selected project, using a read-only sandbox and on-request approvals. Codex
|
|
83
|
+
owns authentication, model selection, permissions, and usage limits. An expired
|
|
84
|
+
or missing login must be resolved in Codex itself.
|
|
85
|
+
|
|
86
|
+
This is a launcher, not an in-REPL provider adapter. Conversation history is not
|
|
87
|
+
transferred, native activity is not included in Thwip usage totals, and exiting
|
|
88
|
+
Codex does not automatically restart Thwip. Restart `thwip` and use `/session load`
|
|
89
|
+
to resume the saved Thwip conversation. Claude and Gemini native launchers are not
|
|
90
|
+
implemented. Direct API mode remains unchanged.
|
|
91
|
+
|
|
77
92
|
---
|
|
78
93
|
|
|
79
94
|
## Auditable handoffs
|
|
@@ -117,7 +132,7 @@ See [research and prior art](docs/handoff-research.md) for the differentiation r
|
|
|
117
132
|
| Google | Gemini API (3.1 Pro Preview, 3.7 Flash, 3.5 Flash-Lite) | Chat, File Edit, Code Run, Terminal, Git |
|
|
118
133
|
| OpenAI | OpenAI API (GPT-5.6 Sol, Terra, Luna) | Chat, File Edit, Code Run, Terminal, Git |
|
|
119
134
|
| DeepSeek | DeepSeek V3 / R1 Reasoner | Chat, File Edit, Code Run, Reasoning |
|
|
120
|
-
| Groq | Llama 3.3
|
|
135
|
+
| Groq | GPT-OSS 120B (default); Llama 3.3 for eligible enterprise accounts only | Chat, File Edit, Code Run |
|
|
121
136
|
| Ollama | Local Models (Llama 3.3, Qwen Coder, DeepSeek R1) | Chat, File Edit, Code Run (Local, Offline) |
|
|
122
137
|
| OpenRouter | Multi-Company Models | Gateway Routing |
|
|
123
138
|
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
# Verification status
|
|
2
|
+
|
|
3
|
+
Verified locally on 2026-09-07 for v1.3.0. The Python suite passes 186 tests.
|
|
4
|
+
Exhaustive behavior across every provider and
|
|
5
|
+
configuration has not been established.
|
|
6
|
+
|
|
7
|
+
| Area | Evidence | Remaining limitation |
|
|
8
|
+
| --- | --- | --- |
|
|
9
|
+
| Commands and aliases | Offline command smoke tests, invalid input cases | Most smoke tests check exceptions, not all rendered content |
|
|
10
|
+
| Sessions | Save/load, separate fresh conversations, malformed metadata, permissions, project rebinding | Concurrent writes to the same explicitly named session are not coordinated |
|
|
11
|
+
| Provider switching and handoff | All seven provider catalogs, portable history, bounded failover | Live account quotas and model availability unverified |
|
|
12
|
+
| Tool execution | Real temporary-file operations, path containment, process timeout/cancellation, invalid arguments | Shell and code tools retain local user privileges; output capture memory is unbounded |
|
|
13
|
+
| Native Codex launcher | Save-before-launch, consent, flags, missing binary, terminal checks, failures | Mocked process replacement; live native interaction unverified; no conversation transfer |
|
|
14
|
+
| Provider responses | Mocked native tool continuations; DeepSeek/Groq/OpenRouter streaming, usage-only chunks, 429/500 errors and serialized tool arguments | Other streaming and error branches still have coverage gaps |
|
|
15
|
+
| Display/config/auth | Configuration validation, display settings, credential boundaries, short-key masking | No full terminal/platform matrix |
|
|
16
|
+
| Usage | Atomic writes, malformed records, valid totals | Unknown catalog pricing may appear as zero estimated cost |
|
|
17
|
+
| Website | Production build and deterministic demo completion/replay test | Real browser, layout, clipboard, keyboard and accessibility checks remain incomplete |
|
|
18
|
+
| Dependencies | npm audit: zero advisories; Python installed-dependency audit: none found | Python audit skipped legacy local `thwip 1.0.0` metadata |
|
|
19
|
+
| Packaging | Wheel/source build and metadata checks | Publication is separately verified through the release workflow |
|
|
20
|
+
|
|
21
|
+
The Python suite measured 69% statement coverage with 160 passing tests before
|
|
22
|
+
the final credential-masking tests were added. Coverage is diagnostic evidence,
|
|
23
|
+
not proof that every feature works. The website test uses a minimal DOM stand-in
|
|
24
|
+
and controlled timers, not a browser.
|
|
25
|
+
|
|
26
|
+
This pass fixed default-session save collisions, malformed session/message
|
|
27
|
+
metadata acceptance, tool-argument display crashes, Rich markup interpretation
|
|
28
|
+
in action output, and short-key masking leakage.
|
|
@@ -0,0 +1,164 @@
|
|
|
1
|
+
import json
|
|
2
|
+
from io import StringIO
|
|
3
|
+
from types import SimpleNamespace
|
|
4
|
+
|
|
5
|
+
import pytest
|
|
6
|
+
from rich.console import Console
|
|
7
|
+
|
|
8
|
+
from thwip.agents import AgentRegistry
|
|
9
|
+
from thwip.agents.base import AgentDone, Capability, LimitHit, LimitStatus, TextDelta
|
|
10
|
+
from thwip.cli import ThwipCLI
|
|
11
|
+
from thwip.config import ThwipConfig, get_usage_path
|
|
12
|
+
from thwip.limits import UsageTracker
|
|
13
|
+
from thwip.session import Session
|
|
14
|
+
from thwip.tools import ToolManager
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
@pytest.fixture
|
|
18
|
+
def cli(tmp_path, monkeypatch):
|
|
19
|
+
monkeypatch.setenv('THWIP_CONFIG_DIR', str(tmp_path / 'config'))
|
|
20
|
+
cli = ThwipCLI.__new__(ThwipCLI)
|
|
21
|
+
cli.config = ThwipConfig(project=str(tmp_path), auto_save=False)
|
|
22
|
+
cli.registry = AgentRegistry(cli.config)
|
|
23
|
+
for a in cli.registry.list_agents():
|
|
24
|
+
monkeypatch.setattr(a, 'is_installed', lambda: True)
|
|
25
|
+
monkeypatch.setattr(a, 'is_configured', lambda: False)
|
|
26
|
+
if a.name == 'ollama':
|
|
27
|
+
a._cached_models = a.get_handoff_models()
|
|
28
|
+
cli.current_agent = cli.registry.get_agent('claude')
|
|
29
|
+
cli.session = Session(project_path=str(tmp_path), current_agent='claude', current_model=cli.current_agent.get_default_model())
|
|
30
|
+
cli.usage_tracker = UsageTracker()
|
|
31
|
+
cli.tool_manager = ToolManager(str(tmp_path))
|
|
32
|
+
cli.detector = SimpleNamespace(scan_all=list)
|
|
33
|
+
monkeypatch.setattr('builtins.input', lambda *args: '')
|
|
34
|
+
monkeypatch.setattr('getpass.getpass', lambda *args: '')
|
|
35
|
+
console = Console(file=StringIO(), width=100, color_system=None)
|
|
36
|
+
monkeypatch.setattr('thwip.cli.console', console)
|
|
37
|
+
monkeypatch.setattr('thwip.theme.console', console)
|
|
38
|
+
return cli
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
@pytest.mark.parametrize('command', [
|
|
42
|
+
'/help', '/h', '/about', '/guide', '/g', '/info', '/agents', '/a', '/list',
|
|
43
|
+
'/models', '/m', '/models flagship', '/models balanced', '/models fast', '/models google',
|
|
44
|
+
'/tools', '/t', '/status', '/limits', '/detect', '/history', '/clear', '/reset', '/cost',
|
|
45
|
+
'/project', '/session save audit', '/session load missing', '/session list', '/session clear',
|
|
46
|
+
'/key', '/key google', '/key invalid', '/key openai REJECTED_TEST_KEY',
|
|
47
|
+
'/handoff', '/handoff invalid', '/handoff google invalid', '/switch', '/switch invalid',
|
|
48
|
+
'/switch google invalid', '/unknown', '/q', '/exit', '/quit',
|
|
49
|
+
])
|
|
50
|
+
@pytest.mark.asyncio
|
|
51
|
+
async def test_command_no_exception(cli, command):
|
|
52
|
+
await cli.handle_command(command)
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
@pytest.mark.parametrize('provider', ['claude','google','openai','deepseek','groq','ollama','openrouter'])
|
|
56
|
+
@pytest.mark.asyncio
|
|
57
|
+
async def test_all_catalogued_handoffs(cli, provider):
|
|
58
|
+
before = list(cli.session.messages)
|
|
59
|
+
for model in cli.registry.get_agent(provider).get_handoff_models():
|
|
60
|
+
await cli.handle_command(f'/handoff {provider} {model.id}')
|
|
61
|
+
assert cli.session.messages == before
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
@pytest.mark.asyncio
|
|
65
|
+
async def test_project_path_with_spaces(cli, tmp_path):
|
|
66
|
+
project = tmp_path / 'project with spaces'
|
|
67
|
+
project.mkdir()
|
|
68
|
+
await cli.handle_command(f'/project {project}')
|
|
69
|
+
assert cli.session.project_path == str(project)
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def test_usage_file_private(cli):
|
|
73
|
+
cli.usage_tracker.record_usage('openai', 'gpt-5.6-terra', 10, 5)
|
|
74
|
+
assert get_usage_path().stat().st_mode & 0o777 == 0o600
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
@pytest.mark.parametrize('tool,args', [
|
|
78
|
+
('read_file', {'file_path': 12}),
|
|
79
|
+
('list_files', {'sub_dir': ['bad']}),
|
|
80
|
+
('run_python', {'code': None}),
|
|
81
|
+
])
|
|
82
|
+
def test_invalid_tool_args_return_error_not_exception(cli, tool, args):
|
|
83
|
+
result = cli.tool_manager.execute_tool(tool, args)
|
|
84
|
+
assert isinstance(result, str)
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
@pytest.mark.asyncio
|
|
88
|
+
async def test_history_displays_literal_markup(cli):
|
|
89
|
+
cli.session.add_user_message('Show [/not-a-tag] literally')
|
|
90
|
+
await cli.handle_command('/history')
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
@pytest.mark.parametrize('value', ['wrong', None, -1, True])
|
|
94
|
+
def test_corrupted_handoff_counter_rejected_or_sanitized(cli, value):
|
|
95
|
+
path = cli.session.save('damaged')
|
|
96
|
+
data = json.loads(path.read_text())
|
|
97
|
+
data['observed_tool_results'] = value
|
|
98
|
+
path.write_text(json.dumps(data))
|
|
99
|
+
loaded = Session.load('damaged')
|
|
100
|
+
assert loaded is None or (type(loaded.observed_tool_results) is int and loaded.observed_tool_results >= 0)
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def test_invalid_config_type_rejected_or_defaulted(cli):
|
|
104
|
+
cli.config._apply_toml({'defaults': {'project': 123, 'confirm_tools': 'false'}})
|
|
105
|
+
assert isinstance(cli.config.project, str)
|
|
106
|
+
assert type(cli.config.confirm_tools) is bool
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
@pytest.mark.asyncio
|
|
110
|
+
async def test_session_roundtrip_rebinds_tools(cli, tmp_path):
|
|
111
|
+
target = tmp_path / 'other'
|
|
112
|
+
target.mkdir()
|
|
113
|
+
saved = Session(project_path=str(target), current_agent='google', current_model='missing')
|
|
114
|
+
saved.save('loadable')
|
|
115
|
+
await cli.handle_command('/session load loadable')
|
|
116
|
+
assert cli.tool_manager.file_editor.project_path == target
|
|
117
|
+
assert cli.current_agent.name == 'google'
|
|
118
|
+
assert cli.session.current_model == cli.current_agent.get_default_model()
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def test_all_tools_and_safety(cli, tmp_path):
|
|
122
|
+
tm = cli.tool_manager
|
|
123
|
+
assert 'Successfully' in tm.execute_tool('write_file', {'file_path':'a.txt','content':'one'})
|
|
124
|
+
assert tm.execute_tool('read_file', {'file_path':'a.txt'}) == 'one'
|
|
125
|
+
assert 'Successfully' in tm.execute_tool('edit_file', {'file_path':'a.txt','old_str':'one','new_str':'two'})
|
|
126
|
+
assert 'a.txt' in tm.execute_tool('list_files', {})
|
|
127
|
+
assert '42' in tm.execute_tool('run_python', {'code':'print(6 * 7)'})
|
|
128
|
+
assert 'Exit code: 0' in tm.execute_tool('run_command', {'command':'printf audit-ok'})
|
|
129
|
+
assert isinstance(tm.execute_tool('git_status', {}), str)
|
|
130
|
+
assert isinstance(tm.execute_tool('git_diff', {}), str)
|
|
131
|
+
assert 'outside' in tm.execute_tool('read_file', {'file_path':'../outside'})
|
|
132
|
+
outside = tmp_path.parent / 'audit-outside-file'
|
|
133
|
+
outside.write_text('not accessible')
|
|
134
|
+
(tmp_path / 'link').symlink_to(outside)
|
|
135
|
+
assert 'outside' in tm.execute_tool('read_file', {'file_path':'link'})
|
|
136
|
+
assert 'Error' in tm.execute_tool('unknown', {})
|
|
137
|
+
assert {t['name'] for t in tm.get_anthropic_tools()} == {t['function']['name'] for t in tm.get_openai_tools()}
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
@pytest.mark.asyncio
|
|
141
|
+
async def test_failover_does_not_retry_already_failed_provider(cli, monkeypatch):
|
|
142
|
+
agents = [cli.registry.get_agent('claude'), cli.registry.get_agent('google')]
|
|
143
|
+
attempts = []
|
|
144
|
+
for agent in agents:
|
|
145
|
+
monkeypatch.setattr(agent, 'is_configured', lambda: True)
|
|
146
|
+
monkeypatch.setattr(agent, 'get_capabilities_for_model', lambda model: {Capability.CHAT})
|
|
147
|
+
|
|
148
|
+
def make_chat(name):
|
|
149
|
+
async def chat(**kwargs):
|
|
150
|
+
attempts.append(name)
|
|
151
|
+
# Stop safely after four simulated rate limits; no actual network calls.
|
|
152
|
+
if len(attempts) <= 4:
|
|
153
|
+
yield LimitHit(error_type=LimitStatus.RATE_LIMITED, message='test quota')
|
|
154
|
+
else:
|
|
155
|
+
yield TextDelta(content='stop test')
|
|
156
|
+
yield AgentDone()
|
|
157
|
+
return chat
|
|
158
|
+
|
|
159
|
+
monkeypatch.setattr(agent, 'chat', make_chat(agent.name))
|
|
160
|
+
monkeypatch.setattr(cli.registry, 'get_ready_agents', lambda: agents)
|
|
161
|
+
cli.config.fallback.chain = ['claude', 'google']
|
|
162
|
+
cli.config.limits.auto_switch = True
|
|
163
|
+
await cli.process_user_message('hello')
|
|
164
|
+
assert len(attempts) <= len(agents), f'Repeated failed providers: {attempts}'
|
|
@@ -75,6 +75,30 @@ class UnconfiguredAgent(ToolCallingAgent):
|
|
|
75
75
|
return False
|
|
76
76
|
|
|
77
77
|
|
|
78
|
+
@pytest.mark.asyncio
|
|
79
|
+
@pytest.mark.parametrize("arguments", [None, [], "bad", {"file_path": "[broken]"}])
|
|
80
|
+
async def test_invalid_tool_arguments_return_errors_without_crashing(tmp_path, arguments):
|
|
81
|
+
class InvalidToolAgent(ToolCallingAgent):
|
|
82
|
+
async def chat(self, messages, **kwargs):
|
|
83
|
+
self.calls.append(messages)
|
|
84
|
+
if len(self.calls) == 1:
|
|
85
|
+
yield ToolUseStart(tool_id="bad", tool_name="read_file", args=arguments)
|
|
86
|
+
yield AgentDone()
|
|
87
|
+
else:
|
|
88
|
+
assert "Error" in messages[-1]["content"]
|
|
89
|
+
yield TextDelta(content="Recovered from invalid tool arguments.")
|
|
90
|
+
yield AgentDone()
|
|
91
|
+
|
|
92
|
+
cli = ThwipCLI.__new__(ThwipCLI)
|
|
93
|
+
cli.config = SimpleNamespace(stream=True, confirm_tools=True)
|
|
94
|
+
cli.session = Session(current_agent="fake", current_model="fake-model")
|
|
95
|
+
cli.current_agent = InvalidToolAgent()
|
|
96
|
+
cli.tool_manager = ToolManager(tmp_path)
|
|
97
|
+
cli.usage_tracker = FakeUsageTracker()
|
|
98
|
+
await cli.process_user_message("Read")
|
|
99
|
+
assert cli.session.messages[-1].content == "Recovered from invalid tool arguments."
|
|
100
|
+
|
|
101
|
+
|
|
78
102
|
@pytest.mark.asyncio
|
|
79
103
|
async def test_unconfigured_agent_shows_setup_guidance(tmp_path):
|
|
80
104
|
cli = ThwipCLI.__new__(ThwipCLI)
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
"""Exercise streaming completion and provider error boundaries without network calls."""
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
from types import SimpleNamespace
|
|
5
|
+
|
|
6
|
+
import httpx
|
|
7
|
+
import openai
|
|
8
|
+
import pytest
|
|
9
|
+
|
|
10
|
+
from thwip.agents.base import AgentDone, LimitHit, LimitStatus, TextDelta
|
|
11
|
+
from thwip.agents.deepseek_agent import DeepSeekAgent
|
|
12
|
+
from thwip.agents.groq_agent import GroqAgent
|
|
13
|
+
from thwip.agents.openrouter_agent import OpenRouterAgent
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
@pytest.mark.parametrize("agent_type", [DeepSeekAgent, GroqAgent, OpenRouterAgent])
|
|
17
|
+
@pytest.mark.asyncio
|
|
18
|
+
async def test_tool_continuation_serializes_arguments_without_mutating_history(agent_type):
|
|
19
|
+
original = {"role": "assistant", "content": "", "tool_calls": [{
|
|
20
|
+
"id": "call1", "type": "function", "function": {
|
|
21
|
+
"name": "read_file", "arguments": {"file_path": "note.txt"},
|
|
22
|
+
},
|
|
23
|
+
}], "_native_state": {"other": "private"}}
|
|
24
|
+
|
|
25
|
+
async def create(**kwargs):
|
|
26
|
+
sent = kwargs["messages"][0]
|
|
27
|
+
assert "_native_state" not in sent
|
|
28
|
+
assert json.loads(sent["tool_calls"][0]["function"]["arguments"]) == {"file_path": "note.txt"}
|
|
29
|
+
return SimpleNamespace(choices=[SimpleNamespace(message=SimpleNamespace(content="done", tool_calls=[]))], usage=None)
|
|
30
|
+
|
|
31
|
+
agent = agent_type(api_key="test")
|
|
32
|
+
agent._client = SimpleNamespace(chat=SimpleNamespace(completions=SimpleNamespace(create=create)))
|
|
33
|
+
events = [event async for event in agent.chat([original], stream=False)]
|
|
34
|
+
assert any(isinstance(event, AgentDone) for event in events)
|
|
35
|
+
assert isinstance(original["tool_calls"][0]["function"]["arguments"], dict)
|
|
36
|
+
assert "_native_state" in original
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
@pytest.mark.parametrize("agent_type", [DeepSeekAgent, GroqAgent, OpenRouterAgent])
|
|
40
|
+
@pytest.mark.asyncio
|
|
41
|
+
async def test_stream_text_and_usage_only_final_chunk(agent_type):
|
|
42
|
+
async def chunks():
|
|
43
|
+
yield SimpleNamespace(choices=[SimpleNamespace(delta=SimpleNamespace(
|
|
44
|
+
content="hello", tool_calls=None))], usage=None)
|
|
45
|
+
yield SimpleNamespace(choices=[], usage=SimpleNamespace(prompt_tokens=9, completion_tokens=3))
|
|
46
|
+
|
|
47
|
+
async def create(**kwargs):
|
|
48
|
+
assert kwargs["stream"] is True
|
|
49
|
+
return chunks()
|
|
50
|
+
|
|
51
|
+
agent = agent_type(api_key="test")
|
|
52
|
+
agent._client = SimpleNamespace(chat=SimpleNamespace(completions=SimpleNamespace(create=create)))
|
|
53
|
+
events = [event async for event in agent.chat([{"role": "user", "content": "hi"}])]
|
|
54
|
+
assert isinstance(events[0], TextDelta) and events[0].content == "hello"
|
|
55
|
+
assert isinstance(events[-1], AgentDone)
|
|
56
|
+
assert events[-1].usage.input_tokens == 9
|
|
57
|
+
assert events[-1].usage.output_tokens == 3
|
|
58
|
+
assert agent.check_limits() == LimitStatus.OK
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
@pytest.mark.parametrize("agent_type", [DeepSeekAgent, GroqAgent, OpenRouterAgent])
|
|
62
|
+
@pytest.mark.parametrize("status", [429, 500])
|
|
63
|
+
@pytest.mark.asyncio
|
|
64
|
+
async def test_provider_errors_emit_failure_not_success(agent_type, status):
|
|
65
|
+
response = httpx.Response(status, request=httpx.Request("POST", "https://example.invalid"))
|
|
66
|
+
error_type = openai.RateLimitError if status == 429 else openai.APIStatusError
|
|
67
|
+
|
|
68
|
+
async def create(**kwargs):
|
|
69
|
+
raise error_type("simulated failure", response=response, body=None)
|
|
70
|
+
|
|
71
|
+
agent = agent_type(api_key="test")
|
|
72
|
+
agent._client = SimpleNamespace(chat=SimpleNamespace(completions=SimpleNamespace(create=create)))
|
|
73
|
+
events = [event async for event in agent.chat([])]
|
|
74
|
+
assert len(events) == 1 and isinstance(events[0], LimitHit)
|
|
75
|
+
assert events[0].error_type == (LimitStatus.RATE_LIMITED if status == 429 else LimitStatus.UNKNOWN)
|
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
"""Native launch boundaries without executing a provider or reading credentials."""
|
|
2
|
+
|
|
3
|
+
from types import SimpleNamespace
|
|
4
|
+
|
|
5
|
+
import pytest
|
|
6
|
+
|
|
7
|
+
from thwip.cli import ThwipCLI
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
@pytest.fixture
|
|
11
|
+
def launcher(tmp_path, monkeypatch):
|
|
12
|
+
events = []
|
|
13
|
+
cli = ThwipCLI.__new__(ThwipCLI)
|
|
14
|
+
cli.session = SimpleNamespace(project_path=str(tmp_path))
|
|
15
|
+
cli.session.save = lambda: events.append("save") or tmp_path / "session.json"
|
|
16
|
+
monkeypatch.setattr("thwip.cli.shutil.which", lambda name: "/installed/codex")
|
|
17
|
+
monkeypatch.setattr("thwip.cli.sys.stdin.isatty", lambda: True)
|
|
18
|
+
monkeypatch.setattr("thwip.cli.sys.stdout.isatty", lambda: True)
|
|
19
|
+
monkeypatch.setattr("thwip.cli.console.input", lambda prompt: "yes")
|
|
20
|
+
monkeypatch.setattr("thwip.cli.print_info", lambda message: None)
|
|
21
|
+
monkeypatch.setattr("thwip.cli.print_error", lambda message: events.append("error"))
|
|
22
|
+
monkeypatch.setattr("thwip.cli.os.execv", lambda path, args: events.append((path, args)))
|
|
23
|
+
return cli, events
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
@pytest.mark.asyncio
|
|
27
|
+
async def test_native_command_saves_before_replacing_process(launcher):
|
|
28
|
+
cli, events = launcher
|
|
29
|
+
await cli.handle_command("/native codex")
|
|
30
|
+
assert events == ["save", ("/installed/codex", [
|
|
31
|
+
"/installed/codex", "--cd", cli.session.project_path,
|
|
32
|
+
"--sandbox", "read-only", "--ask-for-approval", "on-request",
|
|
33
|
+
])]
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
@pytest.mark.parametrize("command", ["/native", "/native claude", "/native codex --dangerously-bypass-approvals-and-sandbox"])
|
|
37
|
+
@pytest.mark.asyncio
|
|
38
|
+
async def test_native_rejects_unsupported_arguments(launcher, command):
|
|
39
|
+
cli, events = launcher
|
|
40
|
+
await cli.handle_command(command)
|
|
41
|
+
assert events == []
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def test_native_decline_does_not_save_or_launch(launcher, monkeypatch):
|
|
45
|
+
cli, events = launcher
|
|
46
|
+
monkeypatch.setattr("thwip.cli.console.input", lambda prompt: "")
|
|
47
|
+
cli.cmd_native("codex")
|
|
48
|
+
assert events == []
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
@pytest.mark.parametrize("failure", ["missing", "noninteractive", "project", "save", "exec"])
|
|
52
|
+
def test_native_failure_stays_in_thwip(launcher, monkeypatch, failure):
|
|
53
|
+
cli, events = launcher
|
|
54
|
+
|
|
55
|
+
def fail(*args):
|
|
56
|
+
raise OSError("private detail")
|
|
57
|
+
|
|
58
|
+
if failure == "missing":
|
|
59
|
+
monkeypatch.setattr("thwip.cli.shutil.which", lambda name: None)
|
|
60
|
+
elif failure == "noninteractive":
|
|
61
|
+
monkeypatch.setattr("thwip.cli.sys.stdin.isatty", lambda: False)
|
|
62
|
+
elif failure == "project":
|
|
63
|
+
cli.session.project_path += "/missing"
|
|
64
|
+
elif failure == "save":
|
|
65
|
+
cli.session.save = fail
|
|
66
|
+
else:
|
|
67
|
+
monkeypatch.setattr("thwip.cli.os.execv", fail)
|
|
68
|
+
cli.cmd_native("codex")
|
|
69
|
+
assert events == (["save", "error"] if failure == "exec" else ["error"])
|
|
@@ -0,0 +1,177 @@
|
|
|
1
|
+
"""Focused verification of repaired settings and native continuation."""
|
|
2
|
+
|
|
3
|
+
import asyncio
|
|
4
|
+
import json
|
|
5
|
+
import time
|
|
6
|
+
from types import SimpleNamespace
|
|
7
|
+
|
|
8
|
+
import pytest
|
|
9
|
+
from rich.text import Text
|
|
10
|
+
|
|
11
|
+
from thwip.agents.base import AgentDone
|
|
12
|
+
from thwip.agents.claude_agent import ClaudeAgent
|
|
13
|
+
from thwip.agents.google_agent import GoogleAgent
|
|
14
|
+
from thwip.agents.openai_agent import OpenAIAgent
|
|
15
|
+
from thwip.cli import ThwipCLI
|
|
16
|
+
from thwip.config import ThwipConfig
|
|
17
|
+
from thwip.detector import SystemDetector
|
|
18
|
+
from thwip.limits import UsageTracker
|
|
19
|
+
from thwip.session import Session
|
|
20
|
+
from thwip.tools.terminal import TerminalRunner
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
@pytest.mark.parametrize("field,value", [
|
|
24
|
+
("estimated_cost", float("nan")), ("estimated_cost", float("inf")),
|
|
25
|
+
("estimated_cost", True), ("estimated_cost", 10 ** 400),
|
|
26
|
+
("last_limit_hit_timestamp", -1), ("last_limit_hit_timestamp", "bad"),
|
|
27
|
+
("last_limit_hit_timestamp", float("nan")), ("last_error", []),
|
|
28
|
+
])
|
|
29
|
+
def test_usage_rejects_corrupt_values_without_losing_other_agents(tmp_path, monkeypatch, field, value):
|
|
30
|
+
path = tmp_path / "usage.json"
|
|
31
|
+
path.write_text(json.dumps({"bad": {field: value}, "good": {"estimated_cost": 0.5}}))
|
|
32
|
+
monkeypatch.setattr("thwip.limits.get_usage_path", lambda: path)
|
|
33
|
+
tracker = UsageTracker()
|
|
34
|
+
assert set(tracker.stats) == {"good"}
|
|
35
|
+
assert tracker.get_summary()["total_cost"] == 0.5
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
@pytest.mark.asyncio
|
|
39
|
+
@pytest.mark.parametrize("provider,auth", [("ollama", "none"), ("openai", "subscription"), ("google", "oauth"), ("claude", "none")])
|
|
40
|
+
async def test_setup_guidance_matches_transport(monkeypatch, provider, auth):
|
|
41
|
+
panels = []
|
|
42
|
+
monkeypatch.setattr("thwip.cli.console.print", lambda panel: panels.append(panel))
|
|
43
|
+
cli = ThwipCLI.__new__(ThwipCLI)
|
|
44
|
+
cli.current_agent = SimpleNamespace(
|
|
45
|
+
name=provider, display_name="Provider [literal]", auth_method=auth,
|
|
46
|
+
is_configured=lambda: False,
|
|
47
|
+
)
|
|
48
|
+
cli.session = Session(current_model="model [literal]")
|
|
49
|
+
await cli.process_user_message("hello")
|
|
50
|
+
text = panels[0].renderable.plain
|
|
51
|
+
assert not cli.session.messages
|
|
52
|
+
if provider == "ollama":
|
|
53
|
+
assert "ollama serve" in text and "No API key is required" in text
|
|
54
|
+
assert "/key" not in text
|
|
55
|
+
else:
|
|
56
|
+
assert "model [literal]" in text and f"/key {provider}" in text
|
|
57
|
+
assert ("sign-in was detected" in text) == (auth in ("oauth", "subscription"))
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def test_display_switches_hide_status_fields():
|
|
61
|
+
cli = ThwipCLI.__new__(ThwipCLI)
|
|
62
|
+
cli.config = ThwipConfig()
|
|
63
|
+
cli.current_agent = OpenAIAgent(api_key="test")
|
|
64
|
+
cli.session = Session(current_model=cli.current_agent.get_default_model())
|
|
65
|
+
cli.session.add_assistant_message("done", "openai", cli.session.current_model, tokens=42)
|
|
66
|
+
cli.usage_tracker = SimpleNamespace(get_summary=lambda: {"total_cost": 1.25})
|
|
67
|
+
visible = cli._render_status().plain
|
|
68
|
+
assert "42 tok" in visible and "$1.2500" in visible and "[chat]" in visible
|
|
69
|
+
for name in ("show_agent_badge", "show_token_count", "show_cost", "show_capabilities"):
|
|
70
|
+
setattr(cli.config.display, name, False)
|
|
71
|
+
assert cli._render_status().plain == ""
|
|
72
|
+
cli.config.display.markdown = False
|
|
73
|
+
assert isinstance(cli._render_response("**literal**"), Text)
|
|
74
|
+
assert cli._render_response("**literal**").plain == "**literal**"
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
@pytest.mark.parametrize("content,expected", [
|
|
78
|
+
('{"userID":"metadata-only"}', False),
|
|
79
|
+
('{"apiKey":""}', False),
|
|
80
|
+
('{"apiKey":42}', False),
|
|
81
|
+
('{"apiKey":"test-key"}', True),
|
|
82
|
+
('{"credentials":{"apiKey":"test-key"}}', True),
|
|
83
|
+
])
|
|
84
|
+
def test_detector_requires_actual_key_field(tmp_path, monkeypatch, content, expected):
|
|
85
|
+
path = tmp_path / "claude.json"
|
|
86
|
+
path.write_text(content)
|
|
87
|
+
monkeypatch.delenv("ANTHROPIC_API_KEY", raising=False)
|
|
88
|
+
monkeypatch.setattr("thwip.detector.KNOWN_CLI_TOOLS", [{
|
|
89
|
+
"name": "Claude", "company": "Anthropic", "binaries": [], "category": "CLI Agent",
|
|
90
|
+
"key_env": "ANTHROPIC_API_KEY", "config_files": [str(path)], "capabilities": [],
|
|
91
|
+
}])
|
|
92
|
+
monkeypatch.setattr("thwip.detector.KNOWN_APPS", [])
|
|
93
|
+
monkeypatch.setattr("thwip.detector.subprocess.run", lambda *a, **k: SimpleNamespace(returncode=1))
|
|
94
|
+
found = SystemDetector().scan_all()
|
|
95
|
+
assert bool(found) is expected
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
@pytest.mark.asyncio
|
|
99
|
+
async def test_openai_preserves_reasoning_and_function_items():
|
|
100
|
+
captured = []
|
|
101
|
+
items = [SimpleNamespace(type="reasoning", id="r1", encrypted_content="opaque", summary=[]),
|
|
102
|
+
SimpleNamespace(type="function_call", call_id="c1", name="read_file", arguments='{"file_path":"x"}')]
|
|
103
|
+
|
|
104
|
+
async def create(**kwargs):
|
|
105
|
+
captured.append(kwargs)
|
|
106
|
+
return SimpleNamespace(output_text="", output=items, usage=None)
|
|
107
|
+
|
|
108
|
+
agent = OpenAIAgent(api_key="test")
|
|
109
|
+
agent._client = SimpleNamespace(responses=SimpleNamespace(create=create))
|
|
110
|
+
events = [e async for e in agent.chat(messages=[{"role": "user", "content": "read"}], stream=False)]
|
|
111
|
+
done = next(e for e in events if isinstance(e, AgentDone))
|
|
112
|
+
messages = [{"role": "assistant", "content": "", "_native_state": done.native_state},
|
|
113
|
+
{"role": "tool", "tool_call_id": "c1", "content": "result"}]
|
|
114
|
+
_ = [e async for e in agent.chat(messages=messages, stream=False)]
|
|
115
|
+
assert captured[-1]["input"][:2] == [vars(item) for item in items]
|
|
116
|
+
assert captured[-1]["input"][2]["type"] == "function_call_output"
|
|
117
|
+
assert captured[-1]["store"] is False
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
@pytest.mark.asyncio
|
|
121
|
+
async def test_claude_preserves_thinking_signature():
|
|
122
|
+
captured = []
|
|
123
|
+
blocks = [SimpleNamespace(type="thinking", thinking="thought", signature="signature"),
|
|
124
|
+
SimpleNamespace(type="tool_use", id="c1", name="read_file", input={"file_path":"x"})]
|
|
125
|
+
|
|
126
|
+
async def create(**kwargs):
|
|
127
|
+
captured.append(kwargs)
|
|
128
|
+
return SimpleNamespace(content=blocks, usage=SimpleNamespace(input_tokens=1, output_tokens=1), stop_reason="tool_use")
|
|
129
|
+
|
|
130
|
+
agent = ClaudeAgent(api_key="test")
|
|
131
|
+
agent._client = SimpleNamespace(messages=SimpleNamespace(create=create))
|
|
132
|
+
events = [e async for e in agent.chat(messages=[], stream=False)]
|
|
133
|
+
done = next(e for e in events if isinstance(e, AgentDone))
|
|
134
|
+
_ = [e async for e in agent.chat(messages=[{"role":"assistant", "_native_state":done.native_state}], stream=False)]
|
|
135
|
+
assert captured[-1]["messages"][0]["content"] == [vars(block) for block in blocks]
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
@pytest.mark.asyncio
|
|
139
|
+
async def test_google_preserves_thought_signature():
|
|
140
|
+
from google.genai import types
|
|
141
|
+
|
|
142
|
+
captured = []
|
|
143
|
+
content = types.Content(role="model", parts=[types.Part(
|
|
144
|
+
function_call=types.FunctionCall(name="read_file", args={"file_path":"x"}),
|
|
145
|
+
thought_signature=b"opaque-signature",
|
|
146
|
+
)])
|
|
147
|
+
|
|
148
|
+
def generate(**kwargs):
|
|
149
|
+
captured.append(kwargs)
|
|
150
|
+
return SimpleNamespace(candidates=[SimpleNamespace(content=content)], usage_metadata=None)
|
|
151
|
+
|
|
152
|
+
agent = GoogleAgent(api_key="test")
|
|
153
|
+
agent._client = SimpleNamespace(models=SimpleNamespace(generate_content=generate))
|
|
154
|
+
events = [e async for e in agent.chat(messages=[], stream=False)]
|
|
155
|
+
done = next(e for e in events if isinstance(e, AgentDone))
|
|
156
|
+
_ = [e async for e in agent.chat(messages=[{"role":"assistant", "_native_state":done.native_state}], stream=False)]
|
|
157
|
+
assert captured[-1]["contents"][0].parts[0].thought_signature == b"opaque-signature"
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def test_sync_timeout_stops_child_writes(tmp_path):
|
|
161
|
+
runner = TerminalRunner(tmp_path)
|
|
162
|
+
result = runner.run_command("sleep 0.3; printf unwanted > late.txt", timeout=0.03)
|
|
163
|
+
assert "timed out" in result
|
|
164
|
+
time.sleep(0.4)
|
|
165
|
+
assert not (tmp_path / "late.txt").exists()
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
@pytest.mark.asyncio
|
|
169
|
+
async def test_async_cancellation_stops_child_writes(tmp_path):
|
|
170
|
+
runner = TerminalRunner(tmp_path)
|
|
171
|
+
task = asyncio.create_task(runner.run_command_async("sleep 0.3; printf unwanted > late.txt"))
|
|
172
|
+
await asyncio.sleep(0.03)
|
|
173
|
+
task.cancel()
|
|
174
|
+
with pytest.raises(asyncio.CancelledError):
|
|
175
|
+
await task
|
|
176
|
+
await asyncio.sleep(0.4)
|
|
177
|
+
assert not (tmp_path / "late.txt").exists()
|