agent-shell-py 0.2.5__tar.gz → 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/AGENTS.md +53 -1
- agent_shell_py-0.2.5/README.md → agent_shell_py-0.3.0/PKG-INFO +49 -1
- agent_shell_py-0.2.5/PKG-INFO → agent_shell_py-0.3.0/README.md +40 -10
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/docs/development/info.md +3 -2
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/skills/invoking-cli-agents/SKILL.md +45 -3
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/skills/invoking-cli-agents/api-reference.md +50 -4
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/src/agent_shell/_version.py +2 -2
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/src/agent_shell/adapters/claude_code_adapter.py +35 -20
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/src/agent_shell/adapters/codex_adapter.py +33 -16
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/src/agent_shell/adapters/copilot_cli_adapter.py +36 -20
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/src/agent_shell/adapters/cursor_adapter.py +33 -16
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/src/agent_shell/adapters/grok_adapter.py +33 -18
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/src/agent_shell/adapters/opencode_adapter.py +34 -17
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/src/agent_shell/adapters/pi_adapter.py +34 -17
- agent_shell_py-0.3.0/src/agent_shell/adapters/process_failure.py +35 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/src/agent_shell/adapters/response.py +6 -0
- agent_shell_py-0.3.0/src/agent_shell/execution.py +341 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/src/agent_shell/models/agent.py +10 -1
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/src/agent_shell/process_cleanup.py +39 -20
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/src/agent_shell/shell.py +32 -7
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/e2e/test_cursor_e2e.py +14 -4
- agent_shell_py-0.3.0/tests/e2e/test_execution_host_e2e.py +46 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/e2e/test_health_check_e2e.py +2 -3
- agent_shell_py-0.3.0/tests/integration/test_execution_host.py +310 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/integration/test_process_lifecycle.py +95 -9
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_cancel.py +4 -6
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_codex_cancel.py +4 -6
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_copilot_cli_cancel.py +4 -6
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_cursor_cancel.py +4 -6
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_grok_cancel.py +4 -6
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_opencode_cancel.py +4 -6
- agent_shell_py-0.3.0/tests/unit/test_pi_cancel.py +19 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_process_cleanup.py +32 -36
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_process_group_registration.py +7 -5
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_shell.py +32 -5
- agent_shell_py-0.2.5/tests/unit/test_pi_cancel.py +0 -21
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/.github/workflows/build.yml +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/.github/workflows/ci.yml +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/.github/workflows/publish.yml +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/.gitignore +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/.python-version +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/LICENSE +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/docs/assets/skill_banner.png +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/docs/development/agent_parameter_comparison.md +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/docs/development/disabled_tools.md +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/docs/development/total_token_count.md +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/pyproject.toml +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/skills/delegating-code-review/SKILL.md +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/src/agent_shell/__init__.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/src/agent_shell/adapters/__init__.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/src/agent_shell/adapters/agent_adapter_protocol.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/src/agent_shell/adapters/health.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/src/agent_shell/adapters/model_discovery.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/src/agent_shell/adapters/outcome.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/src/agent_shell/adapters/stderr_format.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/src/agent_shell/adapters/tool_denial.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/src/agent_shell/models/__init__.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/__init__.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/conftest.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/e2e/__init__.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/e2e/test_claude_code_e2e.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/e2e/test_codex_e2e.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/e2e/test_copilot_cli_e2e.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/e2e/test_grok_e2e.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/e2e/test_model_discovery_e2e.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/e2e/test_opencode_e2e.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/e2e/test_pi_e2e.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/integration/__init__.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/integration/test_claude_code_integration.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/integration/test_claude_code_mcp_integration.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/integration/test_codex_integration.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/integration/test_codex_mcp_integration.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/integration/test_copilot_cli_integration.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/integration/test_copilot_cli_mcp_integration.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/integration/test_cursor_integration.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/integration/test_cursor_mcp_integration.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/integration/test_grok_integration.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/integration/test_grok_mcp_integration.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/integration/test_health_check_integration.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/integration/test_model_discovery_integration.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/integration/test_opencode_integration.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/integration/test_opencode_mcp_integration.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/integration/test_pi_integration.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/integration/test_pi_mcp_integration.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/__init__.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/adapter_matrix.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/codex_fixtures.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/copilot_fixtures.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/cursor_fixtures.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/fixtures.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/grok_fixtures.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/opencode_fixtures.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/pi_fixtures.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_adapter_transport.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_codex_execute.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_codex_parse_event.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_codex_warnings.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_copilot_cli_execute.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_copilot_cli_parse_event.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_copilot_cli_stream.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_cursor_execute.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_cursor_parse_event.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_cursor_warnings.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_execute.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_execute_outcome.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_grok_execute.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_grok_parse_event.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_grok_warnings.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_health_probe.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_mcp_server_spec.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_model_discovery.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_models.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_opencode_execute.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_opencode_parse_event.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_opencode_spawn.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_opencode_stream.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_parse_event.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_pi_execute.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_pi_parse_event.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_pi_warnings.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_response_aggregation.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_shell_cancellation.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_shell_mcp.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_stderr_format.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_stream.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_tool_denial.py +0 -0
- {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/uv.lock +0 -0
|
@@ -8,6 +8,8 @@ A lightweight, async Python package that executes CLI coding agents headlessly a
|
|
|
8
8
|
classDiagram
|
|
9
9
|
class AgentShell {
|
|
10
10
|
-AgentAdapter _adapter
|
|
11
|
+
+ExecutionHost execution_host
|
|
12
|
+
+IsolationPolicy isolation_policy
|
|
11
13
|
+execute(cwd, prompt, ...) AgentResponse
|
|
12
14
|
+stream(cwd, prompt, ...) AsyncIterator~StreamEvent~
|
|
13
15
|
+health_check(cwd, model, timeout) HealthCheckResult
|
|
@@ -49,6 +51,8 @@ classDiagram
|
|
|
49
51
|
+str session_id
|
|
50
52
|
+float duration
|
|
51
53
|
+int output_tokens
|
|
54
|
+
+int returncode
|
|
55
|
+
+int signal
|
|
52
56
|
}
|
|
53
57
|
|
|
54
58
|
class MCPServerSpec {
|
|
@@ -88,8 +92,39 @@ classDiagram
|
|
|
88
92
|
+str session_id
|
|
89
93
|
+int output_tokens
|
|
90
94
|
+str error
|
|
95
|
+
+int returncode
|
|
96
|
+
+int signal
|
|
91
97
|
}
|
|
92
98
|
|
|
99
|
+
class ExecutionHost {
|
|
100
|
+
<<Protocol>>
|
|
101
|
+
+launch(command, cwd, env, stdin, isolation_policy) RunHandle
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
class NativeExecutionHost {
|
|
105
|
+
+launch(command, cwd, env, stdin, isolation_policy) NativeRunHandle
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
class IsolationPolicy {
|
|
109
|
+
<<Protocol>>
|
|
110
|
+
+prepare(command, env) PreparedLaunch
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
class NoIsolation
|
|
114
|
+
class LinuxPidNamespaceIsolation
|
|
115
|
+
|
|
116
|
+
class RunHandle {
|
|
117
|
+
<<Protocol>>
|
|
118
|
+
+int pid
|
|
119
|
+
+int returncode
|
|
120
|
+
+wait() int
|
|
121
|
+
+communicate(input) tuple
|
|
122
|
+
+cancel() None
|
|
123
|
+
+release() None
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
class NativeRunHandle
|
|
127
|
+
|
|
93
128
|
class AgentType {
|
|
94
129
|
<<StrEnum>>
|
|
95
130
|
CLAUDE_CODE
|
|
@@ -98,9 +133,17 @@ classDiagram
|
|
|
98
133
|
CODEX
|
|
99
134
|
PI
|
|
100
135
|
CURSOR
|
|
136
|
+
GROK
|
|
101
137
|
}
|
|
102
138
|
|
|
103
139
|
AgentShell --> AgentAdapter : delegates to
|
|
140
|
+
AgentShell --> ExecutionHost : selects
|
|
141
|
+
AgentShell --> IsolationPolicy : selects
|
|
142
|
+
NativeExecutionHost ..|> ExecutionHost : satisfies
|
|
143
|
+
NativeExecutionHost --> NativeRunHandle : creates
|
|
144
|
+
NativeRunHandle ..|> RunHandle : satisfies
|
|
145
|
+
NoIsolation ..|> IsolationPolicy : satisfies
|
|
146
|
+
LinuxPidNamespaceIsolation ..|> IsolationPolicy : satisfies
|
|
104
147
|
AgentShell --> AgentType : resolves via
|
|
105
148
|
ClaudeCodeAdapter ..|> AgentAdapter : satisfies
|
|
106
149
|
AgentShell ..> AgentResponse : returns on success
|
|
@@ -112,7 +155,16 @@ classDiagram
|
|
|
112
155
|
ClaudeCodeAdapter ..> StreamEvent : parses NDJSON into
|
|
113
156
|
```
|
|
114
157
|
|
|
115
|
-
The adapter pattern uses Python's `Protocol` (structural typing) rather than ABC, so adapters satisfy the contract implicitly without inheritance. Each adapter
|
|
158
|
+
The adapter pattern uses Python's `Protocol` (structural typing) rather than ABC, so adapters satisfy the contract implicitly without inheritance. Each adapter translates agent-specific CLI flags and NDJSON output into the shared `StreamEvent`/`AgentResponse` models, while the selected `ExecutionHost` owns process creation and returns a per-run `RunHandle`.
|
|
159
|
+
|
|
160
|
+
Execution location and protection are separate axes. Existing callers default to
|
|
161
|
+
`NativeExecutionHost()` plus `NoIsolation()`. `LinuxPidNamespaceIsolation` is an opt-in direct
|
|
162
|
+
signal boundary: a tiny init/reaper is PID 1 and the CLI is PID 2 or later, so child-namespace
|
|
163
|
+
processes cannot see or signal AgentShell's ancestors. It requires Linux, `unshare`, and enabled
|
|
164
|
+
unprivileged user/PID namespaces; an unavailable explicit request raises
|
|
165
|
+
`IsolationUnavailableError` and never falls back. This is not a general sandbox and does not
|
|
166
|
+
restrict filesystem, credentials, network, tools, or resources. The host/policy applies to
|
|
167
|
+
`execute()`, `stream()`, and `health_check()`; model discovery and MCP configuration remain local.
|
|
116
168
|
|
|
117
169
|
`output_tokens` is a cost measure — the billed output-token count, which **includes reasoning tokens** (billed at the output rate). Each adapter normalises this so the value is consistent across agents (e.g. OpenCode reports reasoning in a sibling field, so its adapter adds it back).
|
|
118
170
|
|
|
@@ -1,3 +1,12 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: agent-shell-py
|
|
3
|
+
Version: 0.3.0
|
|
4
|
+
Summary: A lightweight abstraction for executing CLI coding agents headlessly
|
|
5
|
+
License-Expression: MIT
|
|
6
|
+
License-File: LICENSE
|
|
7
|
+
Requires-Python: >=3.12
|
|
8
|
+
Description-Content-Type: text/markdown
|
|
9
|
+
|
|
1
10
|
# Agent Shell
|
|
2
11
|
Agent Shell is a light weight abstraction for executing a cli coding agent headlessly
|
|
3
12
|
and returning the output that can be used programatically as a unified contract
|
|
@@ -10,6 +19,8 @@ and returning the output that can be used programatically as a unified contract
|
|
|
10
19
|
behind a common adapter protocol.
|
|
11
20
|
- **Execute or stream** — get one `AgentResponse` (raises `AgentExecutionError` on a failed run),
|
|
12
21
|
or async-iterate normalized `StreamEvent`s with optional thinking/reasoning.
|
|
22
|
+
- **Composable execution policy** — preserve native execution by default, or opt into Linux PID
|
|
23
|
+
namespace isolation without changing an agent adapter.
|
|
13
24
|
- **Session resumption** — continue any conversation by passing back its `session_id`.
|
|
14
25
|
- **Normalized cost & tokens** — consistent `cost` and `output_tokens` (reasoning included)
|
|
15
26
|
regardless of how each CLI reports them.
|
|
@@ -70,6 +81,40 @@ separately.
|
|
|
70
81
|
|
|
71
82
|
## Examples
|
|
72
83
|
|
|
84
|
+
### Execution host and isolation
|
|
85
|
+
|
|
86
|
+
Existing callers remain unchanged. Omitting both settings means
|
|
87
|
+
`NativeExecutionHost()` plus `NoIsolation()`:
|
|
88
|
+
|
|
89
|
+
```python
|
|
90
|
+
shell = AgentShell(agent_type=AgentType.CLAUDE_CODE)
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
To protect the AgentShell owner from broad same-user cleanup commands such as `pkill -f`,
|
|
94
|
+
explicitly request Linux PID namespace isolation:
|
|
95
|
+
|
|
96
|
+
```python
|
|
97
|
+
from agent_shell.execution import LinuxPidNamespaceIsolation
|
|
98
|
+
|
|
99
|
+
shell = AgentShell(
|
|
100
|
+
agent_type=AgentType.CLAUDE_CODE,
|
|
101
|
+
isolation_policy=LinuxPidNamespaceIsolation(),
|
|
102
|
+
)
|
|
103
|
+
```
|
|
104
|
+
|
|
105
|
+
The policy runs a tiny namespace init as PID 1 and the real CLI as PID 2 or later. Processes in
|
|
106
|
+
that child namespace cannot see or signal AgentShell's ancestor processes. The feature requires
|
|
107
|
+
Linux, the `unshare` command, and kernel support for unprivileged user/PID namespaces. If any are
|
|
108
|
+
unavailable, launching raises `IsolationUnavailableError`; it never silently falls back.
|
|
109
|
+
|
|
110
|
+
This is **direct-signal protection, not a sandbox**. It does not restrict files, credentials,
|
|
111
|
+
network access, tools, or resource consumption. Any background descendants are also terminated
|
|
112
|
+
when the isolated namespace ends. `execute()`, `stream()`, and `health_check()` use the selected
|
|
113
|
+
host/policy; model discovery and MCP configuration remain local management operations.
|
|
114
|
+
|
|
115
|
+
`NativeExecutionHost` is currently the only host implementation. Host and isolation are separate
|
|
116
|
+
so future tmux/Herdr hosts can compose with policies without creating one class per combination.
|
|
117
|
+
|
|
73
118
|
### Execute
|
|
74
119
|
|
|
75
120
|
```python
|
|
@@ -108,7 +153,9 @@ follow_up = await shell.execute(
|
|
|
108
153
|
`execute()` raises `AgentExecutionError` instead of returning when a run failed — an `error`
|
|
109
154
|
event was emitted, the terminal `result` had `content == "error"`, or no terminal `result`
|
|
110
155
|
arrived at all. `str(e)` is the bare reason; the exception also carries whatever partial
|
|
111
|
-
`response`/`cost`/`session_id`/`duration`/`output_tokens` the run produced before failing.
|
|
156
|
+
`response`/`cost`/`session_id`/`duration`/`output_tokens` the run produced before failing. When
|
|
157
|
+
the CLI process itself exits unsuccessfully, `returncode` is also populated; signal termination
|
|
158
|
+
uses Python's negative-returncode convention and supplies the positive signal number separately.
|
|
112
159
|
|
|
113
160
|
```python
|
|
114
161
|
from agent_shell.models.agent import AgentExecutionError
|
|
@@ -117,6 +164,7 @@ try:
|
|
|
117
164
|
response = await shell.execute(cwd="/path/to/project", prompt="Fix the failing test")
|
|
118
165
|
except AgentExecutionError as e:
|
|
119
166
|
print(f"run failed: {e}") # e.g. "500 model name=qwen3.6-27b-8Q failed to load"
|
|
167
|
+
print(e.returncode, e.signal) # e.g. -15, 15 for SIGTERM; otherwise None when unavailable
|
|
120
168
|
```
|
|
121
169
|
|
|
122
170
|
### Stream
|
|
@@ -1,12 +1,3 @@
|
|
|
1
|
-
Metadata-Version: 2.4
|
|
2
|
-
Name: agent-shell-py
|
|
3
|
-
Version: 0.2.5
|
|
4
|
-
Summary: A lightweight abstraction for executing CLI coding agents headlessly
|
|
5
|
-
License-Expression: MIT
|
|
6
|
-
License-File: LICENSE
|
|
7
|
-
Requires-Python: >=3.12
|
|
8
|
-
Description-Content-Type: text/markdown
|
|
9
|
-
|
|
10
1
|
# Agent Shell
|
|
11
2
|
Agent Shell is a light weight abstraction for executing a cli coding agent headlessly
|
|
12
3
|
and returning the output that can be used programatically as a unified contract
|
|
@@ -19,6 +10,8 @@ and returning the output that can be used programatically as a unified contract
|
|
|
19
10
|
behind a common adapter protocol.
|
|
20
11
|
- **Execute or stream** — get one `AgentResponse` (raises `AgentExecutionError` on a failed run),
|
|
21
12
|
or async-iterate normalized `StreamEvent`s with optional thinking/reasoning.
|
|
13
|
+
- **Composable execution policy** — preserve native execution by default, or opt into Linux PID
|
|
14
|
+
namespace isolation without changing an agent adapter.
|
|
22
15
|
- **Session resumption** — continue any conversation by passing back its `session_id`.
|
|
23
16
|
- **Normalized cost & tokens** — consistent `cost` and `output_tokens` (reasoning included)
|
|
24
17
|
regardless of how each CLI reports them.
|
|
@@ -79,6 +72,40 @@ separately.
|
|
|
79
72
|
|
|
80
73
|
## Examples
|
|
81
74
|
|
|
75
|
+
### Execution host and isolation
|
|
76
|
+
|
|
77
|
+
Existing callers remain unchanged. Omitting both settings means
|
|
78
|
+
`NativeExecutionHost()` plus `NoIsolation()`:
|
|
79
|
+
|
|
80
|
+
```python
|
|
81
|
+
shell = AgentShell(agent_type=AgentType.CLAUDE_CODE)
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
To protect the AgentShell owner from broad same-user cleanup commands such as `pkill -f`,
|
|
85
|
+
explicitly request Linux PID namespace isolation:
|
|
86
|
+
|
|
87
|
+
```python
|
|
88
|
+
from agent_shell.execution import LinuxPidNamespaceIsolation
|
|
89
|
+
|
|
90
|
+
shell = AgentShell(
|
|
91
|
+
agent_type=AgentType.CLAUDE_CODE,
|
|
92
|
+
isolation_policy=LinuxPidNamespaceIsolation(),
|
|
93
|
+
)
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
The policy runs a tiny namespace init as PID 1 and the real CLI as PID 2 or later. Processes in
|
|
97
|
+
that child namespace cannot see or signal AgentShell's ancestor processes. The feature requires
|
|
98
|
+
Linux, the `unshare` command, and kernel support for unprivileged user/PID namespaces. If any are
|
|
99
|
+
unavailable, launching raises `IsolationUnavailableError`; it never silently falls back.
|
|
100
|
+
|
|
101
|
+
This is **direct-signal protection, not a sandbox**. It does not restrict files, credentials,
|
|
102
|
+
network access, tools, or resource consumption. Any background descendants are also terminated
|
|
103
|
+
when the isolated namespace ends. `execute()`, `stream()`, and `health_check()` use the selected
|
|
104
|
+
host/policy; model discovery and MCP configuration remain local management operations.
|
|
105
|
+
|
|
106
|
+
`NativeExecutionHost` is currently the only host implementation. Host and isolation are separate
|
|
107
|
+
so future tmux/Herdr hosts can compose with policies without creating one class per combination.
|
|
108
|
+
|
|
82
109
|
### Execute
|
|
83
110
|
|
|
84
111
|
```python
|
|
@@ -117,7 +144,9 @@ follow_up = await shell.execute(
|
|
|
117
144
|
`execute()` raises `AgentExecutionError` instead of returning when a run failed — an `error`
|
|
118
145
|
event was emitted, the terminal `result` had `content == "error"`, or no terminal `result`
|
|
119
146
|
arrived at all. `str(e)` is the bare reason; the exception also carries whatever partial
|
|
120
|
-
`response`/`cost`/`session_id`/`duration`/`output_tokens` the run produced before failing.
|
|
147
|
+
`response`/`cost`/`session_id`/`duration`/`output_tokens` the run produced before failing. When
|
|
148
|
+
the CLI process itself exits unsuccessfully, `returncode` is also populated; signal termination
|
|
149
|
+
uses Python's negative-returncode convention and supplies the positive signal number separately.
|
|
121
150
|
|
|
122
151
|
```python
|
|
123
152
|
from agent_shell.models.agent import AgentExecutionError
|
|
@@ -126,6 +155,7 @@ try:
|
|
|
126
155
|
response = await shell.execute(cwd="/path/to/project", prompt="Fix the failing test")
|
|
127
156
|
except AgentExecutionError as e:
|
|
128
157
|
print(f"run failed: {e}") # e.g. "500 model name=qwen3.6-27b-8Q failed to load"
|
|
158
|
+
print(e.returncode, e.signal) # e.g. -15, 15 for SIGTERM; otherwise None when unavailable
|
|
129
159
|
```
|
|
130
160
|
|
|
131
161
|
### Stream
|
|
@@ -61,8 +61,9 @@ uv run pytest tests/e2e -v
|
|
|
61
61
|
|
|
62
62
|
> [!WARNING]
|
|
63
63
|
> E2E tests may mutate real user configuration files. MCP tests can call an agent's real
|
|
64
|
-
>
|
|
65
|
-
> `~/.config/opencode/opencode.json`, `~/.copilot/mcp-config.json`,
|
|
64
|
+
> MCP commands or edit its config directly, affecting files such as `~/.claude.json`,
|
|
65
|
+
> `~/.config/opencode/opencode.json`, `~/.copilot/mcp-config.json`, `~/.cursor/mcp.json`,
|
|
66
|
+
> or Codex configuration.
|
|
66
67
|
> Tests use unique names and `finally` cleanup where implemented, but forced termination,
|
|
67
68
|
> a CLI crash, or a machine failure can prevent cleanup. The CLI may also rewrite config
|
|
68
69
|
> formatting even when the temporary entry is removed. Review the selected E2E test before
|
|
@@ -6,7 +6,7 @@ description: >-
|
|
|
6
6
|
restricting tools, or checking agent/model health. Supports Claude Code, OpenCode,
|
|
7
7
|
Copilot CLI, Codex, Pi, Cursor, and Grok. Keywords: AgentShell, list_models, headless
|
|
8
8
|
agent, model discovery, subprocess, allowed_tools, disallowed_tools, session_id, cost,
|
|
9
|
-
output_tokens.
|
|
9
|
+
output_tokens, execution host, PID namespace, isolation policy.
|
|
10
10
|
---
|
|
11
11
|
|
|
12
12
|
# Invoking CLI Agents with AgentShell
|
|
@@ -44,7 +44,44 @@ uv add agent-shell-py
|
|
|
44
44
|
|
|
45
45
|
AgentShell has two invocation methods — `execute()` collects a complete response, `stream()`
|
|
46
46
|
yields events in real-time — plus helpers for model discovery, health checks, and MCP server
|
|
47
|
-
management. All are async.
|
|
47
|
+
management. All are async. Invocation has three independent choices:
|
|
48
|
+
|
|
49
|
+
- `agent_type`: which CLI (Claude Code, Codex, etc.)
|
|
50
|
+
- `execution_host`: where/how the process is owned (`NativeExecutionHost` today)
|
|
51
|
+
- `isolation_policy`: what protection surrounds it (`NoIsolation` or Linux PID isolation)
|
|
52
|
+
|
|
53
|
+
### Execution Host and Isolation Policy
|
|
54
|
+
|
|
55
|
+
Existing code is backward-compatible: omitting both execution settings selects native execution
|
|
56
|
+
with no isolation.
|
|
57
|
+
|
|
58
|
+
```python
|
|
59
|
+
shell = AgentShell(agent_type=AgentType.CODEX)
|
|
60
|
+
# Equivalent to NativeExecutionHost() + NoIsolation()
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
Opt into direct-signal protection when an agent may run broad same-user cleanup commands such as
|
|
64
|
+
`pkill -f`:
|
|
65
|
+
|
|
66
|
+
```python
|
|
67
|
+
from agent_shell.execution import LinuxPidNamespaceIsolation
|
|
68
|
+
|
|
69
|
+
shell = AgentShell(
|
|
70
|
+
agent_type=AgentType.CODEX,
|
|
71
|
+
isolation_policy=LinuxPidNamespaceIsolation(),
|
|
72
|
+
)
|
|
73
|
+
```
|
|
74
|
+
|
|
75
|
+
The Linux policy puts a tiny init/reaper at namespace PID 1 and the actual CLI at PID 2 or later.
|
|
76
|
+
The CLI cannot see or signal AgentShell's ancestor processes. It requires Linux, `unshare`, and
|
|
77
|
+
kernel support for unprivileged user/PID namespaces. An unavailable requested policy raises
|
|
78
|
+
`IsolationUnavailableError` before the CLI starts; AgentShell never silently falls back.
|
|
79
|
+
|
|
80
|
+
This policy is **not a sandbox**: filesystem, credentials, network, agent tools, and resource use
|
|
81
|
+
remain available. Background descendants cannot outlive the isolated namespace. It applies to
|
|
82
|
+
`execute()`, `stream()`, and `health_check()`; `list_models()` and MCP configuration are local
|
|
83
|
+
management operations. `NativeExecutionHost` is the only shipped host today. Tmux and Herdr are
|
|
84
|
+
future host possibilities, not current APIs.
|
|
48
85
|
|
|
49
86
|
### Discover Available Model Strings
|
|
50
87
|
|
|
@@ -117,6 +154,7 @@ async for event in shell.stream(
|
|
|
117
154
|
print(event.content)
|
|
118
155
|
elif event.type == "error":
|
|
119
156
|
print(f"[error] {event.content}")
|
|
157
|
+
print(event.returncode, event.signal) # populated for process-level failures
|
|
120
158
|
elif event.type == "result":
|
|
121
159
|
print(f"Done ({event.content}). Cost: ${event.cost:.4f}, {event.output_tokens} tok")
|
|
122
160
|
```
|
|
@@ -187,7 +225,7 @@ Capabilities differ by agent. `output_tokens` is populated on all of them; the r
|
|
|
187
225
|
| Copilot CLI | ✅ | ⚠️ `bash`, `edit` only | ✅ | ❌ `0.0` | ✅ real | ✅ |
|
|
188
226
|
| Codex | ❌ | ⚠️ `web_search` only | ✅ | ❌ `0.0` | ❌ `0.0` | ✅ |
|
|
189
227
|
| Pi | ✅ | ⚠️ `bash`, `edit`, `read` | ✅ | ⚠️ paid providers only | ❌ `0.0` | ❌ raises |
|
|
190
|
-
| Cursor | ❌ warns | ❌ none — warns | ❌ warns | ❌ `0.0` | ✅ real |
|
|
228
|
+
| Cursor | ❌ warns | ❌ none — warns | ❌ warns | ❌ `0.0` | ✅ real | ✅ user-scope |
|
|
191
229
|
| Grok | ✅ | ✅ all canonical | ✅ | ⚠️ may be `0.0` | ✅ real | ✅ user-scope |
|
|
192
230
|
|
|
193
231
|
A `✅` for `allowed_tools` means the flag is passed — but it only *enforces* with
|
|
@@ -271,6 +309,7 @@ except AgentExecutionError as e:
|
|
|
271
309
|
print(f"failed: {e}") # str(e) == e.reason, the bare cause
|
|
272
310
|
print(e.response) # text produced before the failure, if any
|
|
273
311
|
print(e.cost, e.session_id, e.duration, e.output_tokens)
|
|
312
|
+
print(e.returncode, e.signal) # -15 and 15 for SIGTERM; None if not process-derived
|
|
274
313
|
else:
|
|
275
314
|
print(response.response)
|
|
276
315
|
```
|
|
@@ -347,6 +386,7 @@ logging.getLogger("agent_shell").addHandler(logging.StreamHandler())
|
|
|
347
386
|
| See agent thinking | `include_thinking=True` in `stream()` |
|
|
348
387
|
| Check an agent/model works | `await shell.health_check(cwd, model=...)` |
|
|
349
388
|
| Cancel a running agent | `KeyboardInterrupt` (handled automatically) |
|
|
389
|
+
| Protect the owner from broad process kills | `isolation_policy=LinuxPidNamespaceIsolation()` |
|
|
350
390
|
|
|
351
391
|
## Common Mistakes
|
|
352
392
|
|
|
@@ -366,6 +406,8 @@ logging.getLogger("agent_shell").addHandler(logging.StreamHandler())
|
|
|
366
406
|
| Not catching `AgentExecutionError` | `execute()` raises on a failed run — catch it |
|
|
367
407
|
| Ignoring `UserWarning` on a deny | An unenforceable deny is warned, not applied — the tool is NOT blocked |
|
|
368
408
|
| Ignoring `session_id` for multi-step work | Without it, each call starts fresh |
|
|
409
|
+
| Treating PID isolation as a sandbox | It only blocks direct signalling of ancestors; separately restrict files, network, credentials, tools, and resources |
|
|
410
|
+
| Assuming isolation silently degrades | Requested isolation raises `IsolationUnavailableError` when unavailable; catch it or fail the operation |
|
|
369
411
|
|
|
370
412
|
## API Reference
|
|
371
413
|
|
|
@@ -4,6 +4,7 @@
|
|
|
4
4
|
`MCPServerSpec`, `HealthCheckResult`
|
|
5
5
|
- [StreamEvent types](#event-types)
|
|
6
6
|
- [AgentShell class](#agentshell-class) — invocation, model discovery, health, MCP management
|
|
7
|
+
- [Execution hosts and isolation](#execution-hosts-and-isolation)
|
|
7
8
|
- [AgentAdapter protocol](#agentadapter-protocol)
|
|
8
9
|
- [Agent-specific notes](#agent-specific-notes)
|
|
9
10
|
|
|
@@ -58,6 +59,8 @@ class AgentExecutionError(Exception):
|
|
|
58
59
|
session_id: str | None = None,
|
|
59
60
|
duration: float = 0.0,
|
|
60
61
|
output_tokens: int = 0,
|
|
62
|
+
returncode: int | None = None, # raw process status; negative means signal
|
|
63
|
+
signal: int | None = None, # positive signal number for signal termination
|
|
61
64
|
): ...
|
|
62
65
|
```
|
|
63
66
|
|
|
@@ -75,6 +78,8 @@ class StreamEvent:
|
|
|
75
78
|
session_id: str | None = None # On session-start and "result" events
|
|
76
79
|
output_tokens: int = 0 # Cumulative generated tokens (on "result" events)
|
|
77
80
|
error: str | None = None # Why a failing "result" failed, when recoverable (Pi)
|
|
81
|
+
returncode: int | None = None # Set on process-level "error" events
|
|
82
|
+
signal: int | None = None # Positive signal number when returncode is negative
|
|
78
83
|
```
|
|
79
84
|
|
|
80
85
|
### MCPServerSpec
|
|
@@ -124,7 +129,8 @@ Canonical event types emitted by `stream()`:
|
|
|
124
129
|
|
|
125
130
|
> A `result` event carries `cost`, `duration`, `output_tokens` and `session_id` on the agents
|
|
126
131
|
> that report them. On a failing result, `error` holds the reason when the adapter recovered a
|
|
127
|
-
> structured one (Pi); it is `None` otherwise.
|
|
132
|
+
> structured one (Pi); it is `None` otherwise. A process-level `error` also carries
|
|
133
|
+
> `returncode`; if a signal terminated it, `returncode` is negative and `signal` is positive.
|
|
128
134
|
|
|
129
135
|
> Codex emits the session-start event as `type="session"` (not `"system"`). If you branch on
|
|
130
136
|
> the session event across agents, match both.
|
|
@@ -140,7 +146,12 @@ Canonical event types emitted by `stream()`:
|
|
|
140
146
|
from agent_shell.shell import AgentShell
|
|
141
147
|
|
|
142
148
|
class AgentShell:
|
|
143
|
-
def __init__(
|
|
149
|
+
def __init__(
|
|
150
|
+
self,
|
|
151
|
+
agent_type: AgentType,
|
|
152
|
+
execution_host: ExecutionHost | None = None, # default NativeExecutionHost()
|
|
153
|
+
isolation_policy: IsolationPolicy | None = None, # default NoIsolation()
|
|
154
|
+
): ...
|
|
144
155
|
# raises ValueError for an AgentType with no registered adapter
|
|
145
156
|
|
|
146
157
|
async def execute(
|
|
@@ -174,6 +185,38 @@ class AgentShell:
|
|
|
174
185
|
async def list_mcp_servers(self) -> list[MCPServerSpec]: ...
|
|
175
186
|
```
|
|
176
187
|
|
|
188
|
+
## Execution Hosts and Isolation
|
|
189
|
+
|
|
190
|
+
```python
|
|
191
|
+
from agent_shell.execution import (
|
|
192
|
+
ExecutionHost,
|
|
193
|
+
IsolationPolicy,
|
|
194
|
+
IsolationUnavailableError,
|
|
195
|
+
LinuxPidNamespaceIsolation,
|
|
196
|
+
NativeExecutionHost,
|
|
197
|
+
NativeRunHandle,
|
|
198
|
+
NoIsolation,
|
|
199
|
+
PreparedLaunch,
|
|
200
|
+
RunHandle,
|
|
201
|
+
)
|
|
202
|
+
```
|
|
203
|
+
|
|
204
|
+
`ExecutionHost` creates a `RunHandle` for one command. The handle exposes `pid`, `stdin`,
|
|
205
|
+
`stdout`, `stderr`, `returncode`, `wait()`, `communicate()`, `cancel()`, and `release()`.
|
|
206
|
+
`NativeExecutionHost` is currently the only concrete host. Agent adapters use handles internally;
|
|
207
|
+
existing `execute()` and `stream()` callers do not need to manage them.
|
|
208
|
+
|
|
209
|
+
`NoIsolation` preserves historical native execution. `LinuxPidNamespaceIsolation` uses rootless
|
|
210
|
+
user + PID namespaces, with a tiny PID 1 reaper and the CLI at PID 2 or later. It protects
|
|
211
|
+
AgentShell's ancestors from direct same-user signals sent inside the child namespace. It is not a
|
|
212
|
+
filesystem, credential, network, tool, resource, or general security sandbox. Background
|
|
213
|
+
descendants terminate when the namespace ends.
|
|
214
|
+
|
|
215
|
+
The Linux policy requires `unshare` and supporting kernel configuration. An explicit request that
|
|
216
|
+
cannot be satisfied raises `IsolationUnavailableError` before launching the CLI and never falls
|
|
217
|
+
back to `NoIsolation`. The host/policy selection applies to `execute()`, `stream()`, and
|
|
218
|
+
`health_check()`; model discovery and MCP configuration remain local management operations.
|
|
219
|
+
|
|
177
220
|
### Model discovery semantics
|
|
178
221
|
|
|
179
222
|
`list_models()` reads the selected CLI's account/workspace-aware catalog without sending an
|
|
@@ -273,8 +316,11 @@ class AgentAdapter(Protocol):
|
|
|
273
316
|
it rather than failing (like Pi; unlike Claude Code, OpenCode, Copilot and Codex, which all
|
|
274
317
|
reject one). A matching id is therefore not proof a prior transcript was continued.
|
|
275
318
|
- `duration` and `output_tokens` are real (`usage.outputTokens`); `cost` is always `0.0` — Cursor
|
|
276
|
-
reports no cost.
|
|
277
|
-
|
|
319
|
+
reports no cost.
|
|
320
|
+
- MCP add/remove/list are supported by directly managing user-scope `~/.cursor/mcp.json`, because
|
|
321
|
+
`cursor-agent mcp` has no add/remove subcommands and its list output lacks full configuration.
|
|
322
|
+
Writes are atomic and user-only. Same-transport updates preserve Cursor-native fields that
|
|
323
|
+
`MCPServerSpec` cannot represent.
|
|
278
324
|
|
|
279
325
|
### Grok
|
|
280
326
|
- Headless: `grok -p --output-format streaming-messages-json` (full assistant blocks; not
|
|
@@ -18,7 +18,7 @@ version_tuple: tuple[int | str, ...]
|
|
|
18
18
|
commit_id: str | None
|
|
19
19
|
__commit_id__: str | None
|
|
20
20
|
|
|
21
|
-
__version__ = version = '0.
|
|
22
|
-
__version_tuple__ = version_tuple = (0,
|
|
21
|
+
__version__ = version = '0.3.0'
|
|
22
|
+
__version_tuple__ = version_tuple = (0, 3, 0)
|
|
23
23
|
|
|
24
24
|
__commit_id__ = commit_id = None
|
{agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/src/agent_shell/adapters/claude_code_adapter.py
RENAMED
|
@@ -1,12 +1,24 @@
|
|
|
1
1
|
import asyncio
|
|
2
2
|
import codecs
|
|
3
3
|
import json
|
|
4
|
-
import os
|
|
5
4
|
import logging
|
|
5
|
+
import os
|
|
6
6
|
import warnings
|
|
7
7
|
from pathlib import Path
|
|
8
8
|
from typing import AsyncIterator
|
|
9
9
|
|
|
10
|
+
from agent_shell.adapters.health import run_health_probe
|
|
11
|
+
from agent_shell.adapters.model_discovery import decode_model_output, run_model_command
|
|
12
|
+
from agent_shell.adapters.process_failure import process_failure_event
|
|
13
|
+
from agent_shell.adapters.response import collect_response
|
|
14
|
+
from agent_shell.adapters.stderr_format import format_stderr
|
|
15
|
+
from agent_shell.adapters.tool_denial import resolve_disallowed_tools
|
|
16
|
+
from agent_shell.execution import (
|
|
17
|
+
ExecutionHost,
|
|
18
|
+
IsolationPolicy,
|
|
19
|
+
NativeExecutionHost,
|
|
20
|
+
NoIsolation,
|
|
21
|
+
)
|
|
10
22
|
from agent_shell.models.agent import (
|
|
11
23
|
AgentResponse,
|
|
12
24
|
HealthCheckResult,
|
|
@@ -15,15 +27,8 @@ from agent_shell.models.agent import (
|
|
|
15
27
|
StreamEvent,
|
|
16
28
|
)
|
|
17
29
|
from agent_shell.process_cleanup import (
|
|
18
|
-
create_grouped_process,
|
|
19
|
-
kill_process_group,
|
|
20
30
|
release_process,
|
|
21
31
|
)
|
|
22
|
-
from agent_shell.adapters.health import run_health_probe
|
|
23
|
-
from agent_shell.adapters.model_discovery import decode_model_output, run_model_command
|
|
24
|
-
from agent_shell.adapters.response import collect_response
|
|
25
|
-
from agent_shell.adapters.stderr_format import format_stderr
|
|
26
|
-
from agent_shell.adapters.tool_denial import resolve_disallowed_tools
|
|
27
32
|
|
|
28
33
|
logger = logging.getLogger("agent_shell.claude_code_adapter")
|
|
29
34
|
|
|
@@ -38,8 +43,18 @@ _DISALLOWED_TOOL_MAP = {
|
|
|
38
43
|
}
|
|
39
44
|
|
|
40
45
|
class ClaudeCodeAdapter():
|
|
41
|
-
def __init__(
|
|
46
|
+
def __init__(
|
|
47
|
+
self,
|
|
48
|
+
execution_host: ExecutionHost | None = None,
|
|
49
|
+
isolation_policy: IsolationPolicy | None = None,
|
|
50
|
+
):
|
|
42
51
|
self._active_processes = []
|
|
52
|
+
self._execution_host = (
|
|
53
|
+
execution_host if execution_host is not None else NativeExecutionHost()
|
|
54
|
+
)
|
|
55
|
+
self._isolation_policy = (
|
|
56
|
+
isolation_policy if isolation_policy is not None else NoIsolation()
|
|
57
|
+
)
|
|
43
58
|
|
|
44
59
|
async def execute(
|
|
45
60
|
self,
|
|
@@ -114,9 +129,10 @@ class ClaudeCodeAdapter():
|
|
|
114
129
|
logger.debug("Command: %s", cmd)
|
|
115
130
|
logger.info("Process started (cwd=%s)", os.path.abspath(cwd))
|
|
116
131
|
|
|
117
|
-
process = await
|
|
132
|
+
process = await self._execution_host.launch(
|
|
118
133
|
cmd,
|
|
119
134
|
cwd=os.path.abspath(cwd),
|
|
135
|
+
isolation_policy=self._isolation_policy,
|
|
120
136
|
)
|
|
121
137
|
|
|
122
138
|
self._active_processes.append(process)
|
|
@@ -174,10 +190,10 @@ class ClaudeCodeAdapter():
|
|
|
174
190
|
child_exited = True
|
|
175
191
|
|
|
176
192
|
stderr = await stderr_task
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
logger.warning("Process
|
|
180
|
-
yield
|
|
193
|
+
failure = process_failure_event(process.returncode, stderr)
|
|
194
|
+
if failure is not None:
|
|
195
|
+
logger.warning("Process failed (%d): %s", process.returncode, failure.content)
|
|
196
|
+
yield failure
|
|
181
197
|
finally:
|
|
182
198
|
# Teardown must live here, not after the read loop: on an exception, or when the
|
|
183
199
|
# consumer abandons the stream, the normal path never runs and the still-running
|
|
@@ -188,8 +204,8 @@ class ClaudeCodeAdapter():
|
|
|
188
204
|
# the child is still alive and still registered until a later turn of the loop, and
|
|
189
205
|
# if the loop is torn down first (asyncio.run cancelling pending tasks) it never
|
|
190
206
|
# runs at all. That last case is what the atexit net in process_cleanup covers.
|
|
191
|
-
release_process(process, self._active_processes, stderr_task,
|
|
192
|
-
|
|
207
|
+
await release_process(process, self._active_processes, stderr_task,
|
|
208
|
+
child_exited=child_exited)
|
|
193
209
|
|
|
194
210
|
def _parse_event(self, event: dict, include_thinking: bool) -> list[StreamEvent]:
|
|
195
211
|
t = event.get("type", "")
|
|
@@ -229,9 +245,10 @@ class ClaudeCodeAdapter():
|
|
|
229
245
|
return events
|
|
230
246
|
|
|
231
247
|
async def cancel(self) -> None:
|
|
232
|
-
|
|
233
|
-
kill_process_group(process)
|
|
248
|
+
processes = list(self._active_processes)
|
|
234
249
|
self._active_processes.clear()
|
|
250
|
+
for process in processes:
|
|
251
|
+
await process.cancel()
|
|
235
252
|
|
|
236
253
|
async def health_check(
|
|
237
254
|
self,
|
|
@@ -437,5 +454,3 @@ class ClaudeCodeAdapter():
|
|
|
437
454
|
return {"returncode": process.returncode, "stdout": stdout_text, "stderr": stderr_text}
|
|
438
455
|
|
|
439
456
|
|
|
440
|
-
|
|
441
|
-
|